{"openapi":"3.1.0","info":{"title":"Taalas API","description":"Taalas API interface for running inference against the HC1 harcore model silicon.","version":"1.0.0"},"paths":{"/health":{"get":{"tags":["Monitoring"],"summary":"Health","description":"Endpoint to check the health status of the server.\n\nReturns:\n    dict: A dictionary containing the following keys:\n        - status (str): The health status of the server, always \"healthy\".\n        - queue_size (int): The number of items in the queue managed by `queue_manager`.\n        - current_adapter (str): The name of the currently loaded LoRA model, or \"none\" if no model is loaded.","operationId":"health_health_get","responses":{"200":{"description":"Successful Response","content":{"application/json":{"schema":{"$ref":"#/components/schemas/HealthResponse"}}}}}}},"/models":{"get":{"tags":["Models"],"summary":"Models","description":"Endpoint to get the model info of server.","operationId":"models_models_get","responses":{"200":{"description":"Successful Response","content":{"application/json":{"schema":{}}}}}}},"/generate":{"post":{"tags":["Inference"],"summary":"Generate","description":"Generate text based on the input prompt.","operationId":"generate_generate_post","requestBody":{"content":{"application/json":{"schema":{"$ref":"#/components/schemas/GenerateRequest"}}},"required":true},"responses":{"200":{"description":"Successful Response","content":{"application/json":{"schema":{}}}},"422":{"description":"Validation Error","content":{"application/json":{"schema":{"$ref":"#/components/schemas/HTTPValidationError"}}}}}}}},"components":{"schemas":{"GenerateRequest":{"properties":{"prompt":{"items":{"type":"object"},"type":"array","title":"Prompt","description":"The input prompt for the language model to generate text from."},"min_output_tokens":{"type":"integer","minimum":0.0,"title":"Min Output Tokens","description":"Minimum number of tokens to output.","default":0},"max_output_tokens":{"type":"integer","minimum":0.0,"title":"Max Output Tokens","description":"Maximum number of tokens to output.","default":0},"topk":{"type":"integer","maximum":8.0,"minimum":1.0,"title":"Topk","description":"Choose top k tokens.","default":1},"model":{"type":"string","title":"Model","description":"The name of the model or adapter to use for generation.","default":"llama3.1-8B"},"stream":{"anyOf":[{"type":"boolean"},{"type":"null"}],"title":"Stream","description":"Whether to stream the response back incrementally.","default":false},"tools":{"anyOf":[{"items":{"type":"object"},"type":"array"},{"type":"null"}],"title":"Tools","description":"List of tools given to model."},"tool_choice":{"anyOf":[{"type":"string"},{"type":"null"}],"title":"Tool Choice","description":"Controls which (if any) tool is called by the model."},"logprobs":{"type":"boolean","title":"Logprobs","description":"Whether to return log probabilities of the output tokens or not.","default":false},"top_logprobs":{"type":"integer","maximum":20.0,"minimum":0.0,"title":"Top Logprobs","description":"Return output tokens with top probabilities.","default":0},"system_prompt_version":{"type":"string","title":"System Prompt Version","description":"Which system prompt to use in the backend.","default":"offline"},"tool_name":{"type":"string","title":"Tool Name","description":"Which tool was selected by router LLM model.","default":""},"context_for_tool":{"type":"string","title":"Context For Tool","description":"Context most relavant to selected tool.","default":""},"guided_json":{"anyOf":[{"type":"object"},{"type":"null"}],"title":"Guided Json","description":"A JSON schema for guided generation."},"guided_regex":{"anyOf":[{"type":"string"},{"type":"null"}],"title":"Guided Regex","description":"A regex pattern for guided generation."},"response_format":{"anyOf":[{"type":"object"},{"type":"null"}],"title":"Response Format","description":"For JSON object output."},"metadata":{"anyOf":[{"type":"object"},{"type":"null"}],"title":"Metadata","description":"Used to propagate number of prefill and decode tokens."},"echo":{"type":"boolean","title":"Echo","description":"Echo back the prompt in addition to the completion.","default":false},"stream_options":{"anyOf":[{"$ref":"#/components/schemas/StreamOptions"},{"type":"null"}],"description":"If stream: true, this will cause the server to send back usage stats as a final chunk."},"stop":{"anyOf":[{"items":{"type":"string"},"type":"array"},{"type":"string"},{"type":"null"}],"title":"Stop","description":"Up to 4 sequences where the API will stop generating further tokens. The returned text will not contain the stop sequence."},"temperature":{"anyOf":[{"type":"number","maximum":2.0,"minimum":0.0},{"type":"null"}],"title":"Temperature","description":"What sampling temperature to use, between 0 and 2.","default":1},"ignore_eos":{"type":"boolean","title":"Ignore Eos","description":"Ignore end of sequence (stop_id) and continue generation until output limit is hit.","default":false}},"additionalProperties":true,"type":"object","required":["prompt"],"title":"GenerateRequest"},"HTTPValidationError":{"properties":{"detail":{"items":{"$ref":"#/components/schemas/ValidationError"},"type":"array","title":"Detail"}},"type":"object","title":"HTTPValidationError"},"HealthResponse":{"properties":{"status":{"type":"string","title":"Status","description":"Health status of the server."},"queue_size":{"type":"integer","title":"Queue Size","description":"Number of items in the processing queue."},"current_adapter":{"type":"string","title":"Current Adapter","description":"Name of the currently loaded LoRA adapter."}},"type":"object","required":["status","queue_size","current_adapter"],"title":"HealthResponse"},"StreamOptions":{"properties":{"include_usage":{"anyOf":[{"type":"boolean"},{"type":"null"}],"title":"Include Usage","default":false},"include_obfuscation":{"anyOf":[{"type":"boolean"},{"type":"null"}],"title":"Include Obfuscation","default":false}},"type":"object","title":"StreamOptions"},"ValidationError":{"properties":{"loc":{"items":{"anyOf":[{"type":"string"},{"type":"integer"}]},"type":"array","title":"Location"},"msg":{"type":"string","title":"Message"},"type":{"type":"string","title":"Error Type"}},"type":"object","required":["loc","msg","type"],"title":"ValidationError"}}}}