3636from openai .types .completion import Completion as OpenAICompletion
3737from openai .types .completion_choice import CompletionChoice , Logprobs as CompletionLogprobs
3838from openai .types .completion_usage import CompletionUsage
39+ from openai .types .model import Model as OpenAIModel
3940from openai .types .chat .chat_completion import (
4041 ChatCompletion ,
4142 Choice as ChatCompletionChoice ,
@@ -1316,6 +1317,7 @@ def resolve_model_path(self) -> str:
13161317 class ModelOptions (BaseModel ):
13171318 path : Optional [str ] = None
13181319 from_pretrained : Optional ["ConfigFile.FromPretrainedOptions" ] = None
1320+ alias : Optional [str ] = None
13191321 n_gpu_layers : Optional [int ] = None
13201322 split_mode : Optional [int ] = None
13211323 main_gpu : Optional [int ] = None
@@ -4808,6 +4810,31 @@ class ResponsesStream:
48084810 def __init__ (self , model : Model ) -> None :
48094811 self .model = model
48104812
4813+ def model_id (self ) -> str :
4814+ model_alias = getattr (self .model , "model_alias" , None )
4815+ if isinstance (model_alias , str ) and model_alias :
4816+ return model_alias
4817+ return self .model .model_path
4818+
4819+ def model_card (self ) -> OpenAIModel :
4820+ model_path = self .model_id ()
4821+ try :
4822+ created = int (Path (model_path ).stat ().st_mtime )
4823+ except OSError :
4824+ created = int (time .time ())
4825+ return OpenAIModel (
4826+ id = model_path ,
4827+ created = created ,
4828+ object = "model" ,
4829+ owned_by = "llama-cpp-python" ,
4830+ )
4831+
4832+ def model_list (self ) -> Dict [str , Any ]:
4833+ return {
4834+ "object" : "list" ,
4835+ "data" : [self .model_card ().model_dump (mode = "json" , exclude_none = True )],
4836+ }
4837+
48114838 @staticmethod
48124839 def decode_text (data : bytes ) -> str :
48134840 return data .decode ("utf-8" , errors = "ignore" )
@@ -6831,6 +6858,7 @@ def __init__(
68316858 self ,
68326859 * ,
68336860 model_path : str ,
6861+ model_alias : Optional [str ] = None ,
68346862 n_gpu_layers : Optional [int ] = None ,
68356863 split_mode : Optional [int ] = None ,
68366864 main_gpu : Optional [int ] = None ,
@@ -6873,6 +6901,7 @@ def __init__(
68736901 llama_cpp .llama_backend_init ()
68746902 self .backend_initialized = True
68756903 self .model_path = model_path
6904+ self .model_alias = model_alias
68766905 self .prompt_chunk_size = prompt_chunk_size
68776906 self .response_schema = response_schema
68786907 model_params , self ._c_tensor_split , self ._kv_overrides_array = self .build_model_params (
@@ -8860,6 +8889,11 @@ def response_chunk_payloads(
88608889 )
88618890 )
88628891
8892+ @app .get ("/v1/models" )
8893+ async def list_models () -> Dict [str , Any ]:
8894+ service : CompletionService = app .state .service
8895+ return service .formatter .model_list ()
8896+
88638897 @app .get ("/healthz" )
88648898 async def healthz () -> Dict [str , str ]:
88658899 return {"status" : "ok" }
@@ -8878,6 +8912,7 @@ def main() -> None:
88788912 model_path = config .model .resolve_model_path ()
88798913 model = Model (
88808914 model_path = model_path ,
8915+ model_alias = config .model .alias ,
88818916 n_gpu_layers = config .model .n_gpu_layers ,
88828917 split_mode = config .model .split_mode ,
88838918 main_gpu = config .model .main_gpu ,
0 commit comments