@@ -88,9 +88,14 @@ def test_real_model(llama_cpp_model_path):
8888 context = internals .LlamaContext (model = model , params = cparams )
8989 tokens = model .tokenize (b"Hello, world!" , add_bos = True , special = True )
9090
91- assert tokens == [9707 , 11 , 1879 , 0 ]
91+ assert tokens == [9419 , 11 , 1814 , 0 ]
9292
93- tokens = model .tokenize (b"The quick brown fox jumps" , add_bos = True , special = True )
93+ tokens = model .tokenize (
94+ b"The quick brown fox jumps over the lazy dog. The quick brown fox jumps " ,
95+ add_bos = True ,
96+ special = True ,
97+ )
98+ prompt_token_count = len (tokens )
9499
95100 batch = internals .LlamaBatch (n_tokens = len (tokens ), embd = 0 , n_seq_max = 1 )
96101
@@ -111,9 +116,9 @@ def test_real_model(llama_cpp_model_path):
111116 tokens = [token_id ]
112117 result += tokens
113118
114- output = result [5 :]
119+ output = result [prompt_token_count :]
115120 output_text = model .detokenize (output , special = True )
116- assert output_text == b" over the lazy dog "
121+ assert output_text == b"5 times over the"
117122
118123
119124def test_real_llama (llama_cpp_model_path ):
@@ -129,14 +134,14 @@ def test_real_llama(llama_cpp_model_path):
129134 )
130135
131136 output = model .create_completion (
132- "The quick brown fox jumps" ,
137+ "The quick brown fox jumps over the lazy dog. The quick brown fox jumps " ,
133138 max_tokens = 4 ,
134139 top_k = 50 ,
135140 top_p = 0.9 ,
136141 temperature = 0.8 ,
137142 seed = 1337 ,
138143 )
139- assert output ["choices" ][0 ]["text" ] == " over the lazy dog "
144+ assert output ["choices" ][0 ]["text" ] == "5 times over the"
140145
141146 output = model .create_completion (
142147 "The capital of france is paris, 'true' or 'false'?:\n " ,
@@ -177,7 +182,7 @@ def logit_processor_func(input_ids, logits):
177182 state = model .save_state ()
178183
179184 output = model .create_completion (
180- "Pick a number from 1 to 10? :\n " ,
185+ "Pick a random number from 1 to 10:\n " ,
181186 max_tokens = 4 ,
182187 top_k = 50 ,
183188 top_p = 0.9 ,
@@ -189,7 +194,7 @@ def logit_processor_func(input_ids, logits):
189194 number_1 = output ["choices" ][0 ]["text" ]
190195
191196 output = model .create_completion (
192- "Pick a number from 1 to 10? :\n " ,
197+ "Pick a random number from 1 to 10:\n " ,
193198 max_tokens = 4 ,
194199 top_k = 50 ,
195200 top_p = 0.9 ,
@@ -203,7 +208,7 @@ def logit_processor_func(input_ids, logits):
203208 model .load_state (state )
204209
205210 output = model .create_completion (
206- "Pick a number from 1 to 10? :\n " ,
211+ "Pick a random number from 1 to 10:\n " ,
207212 max_tokens = 4 ,
208213 top_k = 50 ,
209214 top_p = 0.9 ,
0 commit comments