Repository navigation
Expand file tree
/
Copy pathconfig.yaml
More file actions
1011 lines (889 loc) · 41.1 KB
/
Copy pathconfig.yaml
File metadata and controls
1011 lines (889 loc) · 41.1 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
# Airlock - LiteLLM Proxy Configuration
# https://docs.litellm.ai/docs/proxy/configs
#
# This is the main configuration file for the Airlock LLM proxy.
# Copy to config.yaml and fill in your API keys (or set them as env vars).
# Pull in machine-specific MCP servers from config.local.yaml. The repository
# ships it as an empty mapping so fresh checkouts start normally; local MCP
# overrides are uncommitted edits. LiteLLM REPLACES the mcp_servers dict with
# the included file's, so an override must list ALL runtime MCP servers (incl.
# a copy of the newscatcher block below).
include: ["config.local.yaml"]
# ---------------------------------------------------------------------------
# Model definitions
# ---------------------------------------------------------------------------
model_list:
# --- Anthropic (Claude Opus 4.8 / Sonnet 4.6 / Haiku 4.5) ---
- model_name: claude-opus
litellm_params:
model: anthropic/claude-opus-4-8
api_key: os.environ/ANTHROPIC_API_KEY
- model_name: claude-sonnet
litellm_params:
model: anthropic/claude-sonnet-4-6
api_key: os.environ/ANTHROPIC_API_KEY
- model_name: claude-haiku
litellm_params:
model: anthropic/claude-haiku-4-5-20251001
api_key: os.environ/ANTHROPIC_API_KEY
# Provider-prefixed aliases (0.5.2) — stable served-by-explicit names; same
# litellm_params as the bare entries above (dual-listed).
- model_name: anthropic/claude-opus
litellm_params:
model: anthropic/claude-opus-4-8
api_key: os.environ/ANTHROPIC_API_KEY
- model_name: anthropic/claude-sonnet
litellm_params:
model: anthropic/claude-sonnet-4-6
api_key: os.environ/ANTHROPIC_API_KEY
- model_name: anthropic/claude-haiku
litellm_params:
model: anthropic/claude-haiku-4-5-20251001
api_key: os.environ/ANTHROPIC_API_KEY
# --- OpenAI (GPT-5.5 flagship / 5.4 value / 5.4 mini+nano / 5.3 codex) ---
- model_name: gpt-5-pro
litellm_params:
model: openai/gpt-5.5-pro
api_key: os.environ/OPENAI_API_KEY
- model_name: gpt-5
litellm_params:
model: openai/gpt-5.5
api_key: os.environ/OPENAI_API_KEY
# Previous flagship — still available at ~half the cost of 5.5
# ($2.50/$15 vs $5.00/$30 per 1M). Kept as the cheaper high-tier option.
- model_name: gpt-5.4
litellm_params:
model: openai/gpt-5.4
api_key: os.environ/OPENAI_API_KEY
- model_name: gpt-5-mini
litellm_params:
model: openai/gpt-5.4-mini
api_key: os.environ/OPENAI_API_KEY
- model_name: gpt-5-nano
litellm_params:
model: openai/gpt-5.4-nano
api_key: os.environ/OPENAI_API_KEY
- model_name: gpt-5-codex
litellm_params:
model: openai/gpt-5.3-codex
api_key: os.environ/OPENAI_API_KEY
# FathomDB benchmark extractor/chat model. This is an explicit alias, not a
# routing fallback: benchmark callers must request it deliberately.
- model_name: gpt-4o-mini
litellm_params:
model: openai/gpt-4o-mini
api_key: os.environ/OPENAI_API_KEY
# FathomDB benchmark embedding model. The explicit marker keeps this model
# off chat completions and authorizes it only for /v1/embeddings.
- model_name: text-embedding-3-small
litellm_params:
model: openai/text-embedding-3-small
api_key: os.environ/OPENAI_API_KEY
airlock_embeddings: true
# Provider-prefixed OpenAI aliases (0.5.2) — same litellm_params as above.
- model_name: openai/gpt-5-pro
litellm_params:
model: openai/gpt-5.5-pro
api_key: os.environ/OPENAI_API_KEY
- model_name: openai/gpt-5
litellm_params:
model: openai/gpt-5.5
api_key: os.environ/OPENAI_API_KEY
- model_name: openai/gpt-5.4
litellm_params:
model: openai/gpt-5.4
api_key: os.environ/OPENAI_API_KEY
- model_name: openai/gpt-5-mini
litellm_params:
model: openai/gpt-5.4-mini
api_key: os.environ/OPENAI_API_KEY
- model_name: openai/gpt-5-nano
litellm_params:
model: openai/gpt-5.4-nano
api_key: os.environ/OPENAI_API_KEY
- model_name: openai/gpt-5-codex
litellm_params:
model: openai/gpt-5.3-codex
api_key: os.environ/OPENAI_API_KEY
- model_name: openai/gpt-4o-mini
litellm_params:
model: openai/gpt-4o-mini
api_key: os.environ/OPENAI_API_KEY
- model_name: openai/text-embedding-3-small
litellm_params:
model: openai/text-embedding-3-small
api_key: os.environ/OPENAI_API_KEY
airlock_embeddings: true
# --- OpenAI GPT-5.6 (GA 2026-07-09) — Sol / Terra / Luna ---
# A FLAT PRICE-TIER TRIPLE over ONE capability envelope, not a mini/nano
# capability ladder: all three share 1,050,000 ctx / 128,000 max output,
# cutoff 2026-02-16, and an identical capability flag set. They differ only
# in price and quality. Luna is NOT a nano replacement — $1/$6 vs
# gpt-5.4-nano's $0.20/$1.25 (5x input).
#
# Long-context surcharge: >272K INPUT tokens bills 2x input / 1.5x output
# for the WHOLE request (sol $5->$10 in, $30->$45 out; terra $2.50->$5,
# $15->$22.50; luna $1->$2, $6->$9). litellm applies this automatically.
#
# ORDER IS LOAD-BEARING: `-sol` precedes bare `gpt-5.6` so dated snapshots
# (gpt-5.6-sol-2026-07-09) resolve to the explicit variant.
- model_name: gpt-5.6-sol
litellm_params:
model: openai/gpt-5.6-sol
api_key: os.environ/OPENAI_API_KEY
- model_name: gpt-5.6-terra
litellm_params:
model: openai/gpt-5.6-terra
api_key: os.environ/OPENAI_API_KEY
- model_name: gpt-5.6-luna
litellm_params:
model: openai/gpt-5.6-luna
api_key: os.environ/OPENAI_API_KEY
# Convenience family alias. NOTE: bare `gpt-5.6` is NOT a real OpenAI model
# id — verified against live GET /v1/models on 2026-07-20, which offers only
# sol/terra/luna. litellm's price map carries a bare entry, which is
# misleading. This alias is pinned to the EXPLICIT sol body so the string
# `gpt-5.6` is never forwarded upstream; pointing it at the floating id
# would 404 every request.
- model_name: gpt-5.6
litellm_params:
model: openai/gpt-5.6-sol
api_key: os.environ/OPENAI_API_KEY
# Provider-prefixed twins (0.5.2 pattern). LOAD-BEARING, not cosmetic:
# ModelAliasTable's _provider_body_alias uses setdefault (first-writer-wins),
# so without an explicit entry `openai/gpt-5.6-sol` would bind to whichever
# alias sharing that body appears first in this list.
- model_name: openai/gpt-5.6-sol
litellm_params:
model: openai/gpt-5.6-sol
api_key: os.environ/OPENAI_API_KEY
- model_name: openai/gpt-5.6-terra
litellm_params:
model: openai/gpt-5.6-terra
api_key: os.environ/OPENAI_API_KEY
- model_name: openai/gpt-5.6-luna
litellm_params:
model: openai/gpt-5.6-luna
api_key: os.environ/OPENAI_API_KEY
- model_name: openai/gpt-5.6
litellm_params:
model: openai/gpt-5.6-sol
api_key: os.environ/OPENAI_API_KEY
# --- Google (Gemini) ---
# gemini-pro / gemini-flash advanced to the 3.x line 2026-07-20 (O-4).
# NOTE: these are PREVIEW ids — Google withdraws them (this config previously
# carried gemini-3-pro-preview, shut down 2026-03-09), so a withdrawal breaks
# the generic alias until it is repointed. Deliberate cost increase:
# pro $1.25/$10 -> $2.00/$12, flash $0.30/$2.50 -> $0.50/$3.00 per 1M.
# gemini-3.5-flash was REJECTED for `gemini-flash`: at $1.50/$9.00 it would be
# pricier than everything else in the `low` cost tier.
- model_name: gemini-flash
litellm_params:
model: gemini/gemini-3-flash-preview
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
# Generic alias advanced to the 3.1 line (2026-07-20). `gemini-3.1-flash-lite`
# below points at the same body — this is the same generic+versioned dual
# listing as gpt-5 -> gpt-5.5. Clients are told which concrete model backed
# the alias via X-Airlock-Model-Alias, so the advance is never silent.
- model_name: gemini-flash-lite
litellm_params:
model: gemini/gemini-3.1-flash-lite
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
- model_name: gemini-pro
litellm_params:
model: gemini/gemini-3.1-pro-preview
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
# --- Google (Gemini 3.5 / 3.1) ---
- model_name: gemini-3.5-flash
litellm_params:
model: gemini/gemini-3.5-flash
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
- model_name: gemini-3.1-flash-lite
litellm_params:
model: gemini/gemini-3.1-flash-lite
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
- model_name: gemini-3-flash
litellm_params:
model: gemini/gemini-3-flash-preview
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
# NOTE: gemini-3-pro removed — gemini-3-pro-preview was shut down 2026-03-09.
# Use gemini-3.1-pro below (Google's recommended migration target).
- model_name: gemini-3.1-pro
litellm_params:
model: gemini/gemini-3.1-pro-preview
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
- model_name: gemini-3.1-pro-tools
litellm_params:
model: gemini/gemini-3.1-pro-preview-customtools
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
- model_name: gemini-coding
litellm_params:
model: enhanced/gemini-coding
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
enhanced_profile:
target_model: gemini/gemini-3.1-pro-preview-customtools
system_prompt: "CRITICAL: You are operating in a multi-turn tool-calling loop. You must retain and finalize all reasoning pathways. Do not truncate internal thoughts."
params:
thinking: true
thinking_level: "MEDIUM"
# Provider-prefixed AI Studio aliases (0.5.2) — same litellm_params (gemini/)
# as the bare entries above. The aistudio/gemini-3.5-flash and
# aistudio/gemini-3.1-pro aliases carry the airlock_batch marker (consolidating
# the legacy -aistudio twins) so the prefixed name serves sync AND batch.
- model_name: aistudio/gemini-flash
litellm_params:
model: gemini/gemini-3-flash-preview
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
- model_name: aistudio/gemini-flash-lite
litellm_params:
model: gemini/gemini-3.1-flash-lite
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
- model_name: aistudio/gemini-pro
litellm_params:
model: gemini/gemini-3.1-pro-preview
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
- model_name: aistudio/gemini-3.5-flash
litellm_params:
model: gemini/gemini-3.5-flash
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
airlock_batch:
backend: aistudio
provider_model: gemini-3.5-flash
- model_name: aistudio/gemini-3.1-flash-lite
litellm_params:
model: gemini/gemini-3.1-flash-lite
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
- model_name: aistudio/gemini-3-flash
litellm_params:
model: gemini/gemini-3-flash-preview
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
- model_name: aistudio/gemini-3.1-pro
litellm_params:
model: gemini/gemini-3.1-pro-preview
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
airlock_batch:
backend: aistudio
provider_model: gemini-3.1-pro-preview
- model_name: aistudio/gemini-3.1-pro-tools
litellm_params:
model: gemini/gemini-3.1-pro-preview-customtools
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
- model_name: aistudio/gemini-coding
litellm_params:
model: enhanced/gemini-coding
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
enhanced_profile:
target_model: gemini/gemini-3.1-pro-preview-customtools
system_prompt: "CRITICAL: You are operating in a multi-turn tool-calling loop. You must retain and finalize all reasoning pathways. Do not truncate internal thoughts."
params:
thinking: true
thinking_level: "MEDIUM"
# --- Google Vertex AI (Gemini 3.x, BATCH-capable) ---
# vertex_ai is wired in LiteLLM for the Batch API (/v1/files + /v1/batches);
# the gemini/ (AI Studio) provider above is not. Also serves sync completions.
# Auth: VERTEX_PROJECT + VERTEX_CREDENTIALS (service-account JSON) from .env;
# batch staging uses GCS_BUCKET_NAME. (Needs the `vertex` extra: google-auth.)
#
# Availability verified via the SA token (2026-06-14) — both on `global` ONLY
# (every 3.x id is 404 in us-central1 / us-east5):
# gemini-3.5-flash : 200
# gemini-3.1-pro-preview : 200 (GA id `gemini-3.1-pro` is 404 — use -preview)
# NOTE: Vertex BATCH jobs generally need a REGIONAL location; `global` is fine
# for sync but may not support BatchPredictionJob. For 3.x batch, set
# vertex_location to a region once these models are regionally available.
- model_name: gemini-3.5-flash-vertex
litellm_params:
model: vertex_ai/gemini-3.5-flash
vertex_project: os.environ/VERTEX_PROJECT
vertex_location: global
vertex_credentials: os.environ/VERTEX_CREDENTIALS
- model_name: gemini-3.1-pro-vertex
litellm_params:
model: vertex_ai/gemini-3.1-pro-preview
vertex_project: os.environ/VERTEX_PROJECT
vertex_location: global
vertex_credentials: os.environ/VERTEX_CREDENTIALS
# Provider-prefixed Vertex aliases (0.5.2) — same litellm_params as the
# -vertex entries above. NO airlock_batch marker: vertex batch is region-gated
# and vertex_location is `global` (sync/chat-only).
- model_name: vertex/gemini-3.5-flash
litellm_params:
model: vertex_ai/gemini-3.5-flash
vertex_project: os.environ/VERTEX_PROJECT
vertex_location: global
vertex_credentials: os.environ/VERTEX_CREDENTIALS
- model_name: vertex/gemini-3.1-pro
litellm_params:
model: vertex_ai/gemini-3.1-pro-preview
vertex_project: os.environ/VERTEX_PROJECT
vertex_location: global
vertex_credentials: os.environ/VERTEX_CREDENTIALS
# --- AI Studio Gemini BATCH (Airlock Batch Gateway, not LiteLLM-native) ---
# The `gemini/` (AI Studio) provider is NOT wired in LiteLLM's Batch API, so
# these aliases are handled by the Airlock Batch Gateway middleware, selected
# via ?custom_llm_provider=aistudio on /v1/files + /v1/batches.
# The `airlock_batch` marker is a SIBLING of litellm_params (NOT nested inside
# it) so it never leaks to the provider SDK on the sync path (§7.4). The sync
# completion still routes through litellm's native gemini/ provider unchanged.
- model_name: gemini-3.5-flash-aistudio
litellm_params:
model: gemini/gemini-3.5-flash
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
airlock_batch:
backend: aistudio
provider_model: gemini-3.5-flash
- model_name: gemini-3.1-pro-aistudio
litellm_params:
model: gemini/gemini-3.1-pro-preview
api_key: os.environ/GOOGLE_AISTUDIO_API_KEY
airlock_batch:
backend: aistudio
provider_model: gemini-3.1-pro-preview
# --- Mistral BATCH (Airlock Batch Gateway, not LiteLLM-native) ---
# LiteLLM does NOT wire the `mistral` provider into its Batch API, so these
# aliases are handled by the Airlock Batch Gateway middleware, selected via
# ?custom_llm_provider=mistral on /v1/files + /v1/batches. The `airlock_batch`
# marker is a SIBLING of litellm_params (NOT nested inside it) so it never
# leaks to the provider SDK on the sync path (§7.4). The sync completion still
# routes through litellm's native mistral/ provider unchanged.
- model_name: mistral-large-batch
litellm_params:
model: mistral/mistral-large-latest
api_key: os.environ/MISTRAL_API_KEY
airlock_batch:
backend: mistral
provider_model: mistral-large-latest
- model_name: mistral-small-batch
litellm_params:
model: mistral/mistral-small-latest
api_key: os.environ/MISTRAL_API_KEY
airlock_batch:
backend: mistral
provider_model: mistral-small-latest
# Local vLLM batch via the gateway-as-executor backend
# (?custom_llm_provider=vllm). vLLM has no async Batch server API, so Airlock
# executes the scanned rows against the live /v1/chat/completions endpoint and
# owns the job lifecycle. api_base/api_key are read per-alias from
# litellm_params (api_base must end in /v1). See
# dev/plans/prompts/vllm-batch-executor.md.
- model_name: qwen36-27b-vllm-batch
litellm_params:
model: openai/qwen3.6-27b
api_base: http://192.168.1.45:8000/v1
api_key: os.environ/VLLM_API_KEY
airlock_batch:
backend: vllm
provider_model: qwen3.6-27b
# --- Mistral AI (large-3 / medium / small-4 / magistral / codestral) ---
- model_name: mistral-large
litellm_params:
model: mistral/mistral-large-latest
api_key: os.environ/MISTRAL_API_KEY
- model_name: mistral-medium
litellm_params:
model: mistral/mistral-medium-latest
api_key: os.environ/MISTRAL_API_KEY
- model_name: mistral-small
litellm_params:
model: mistral/mistral-small-latest
api_key: os.environ/MISTRAL_API_KEY
- model_name: codestral
litellm_params:
model: mistral/codestral-latest
api_key: os.environ/MISTRAL_API_KEY
- model_name: magistral-medium
litellm_params:
model: mistral/magistral-medium-latest
api_key: os.environ/MISTRAL_API_KEY
# Provider-prefixed Mistral aliases (0.5.2) — same litellm_params as above.
# mistral/mistral-large and mistral/mistral-small carry the airlock_batch
# marker (consolidating the legacy -batch twins) so they serve sync AND batch.
- model_name: mistral/mistral-large
litellm_params:
model: mistral/mistral-large-latest
api_key: os.environ/MISTRAL_API_KEY
airlock_batch:
backend: mistral
provider_model: mistral-large-latest
- model_name: mistral/mistral-medium
litellm_params:
model: mistral/mistral-medium-latest
api_key: os.environ/MISTRAL_API_KEY
- model_name: mistral/mistral-small
litellm_params:
model: mistral/mistral-small-latest
api_key: os.environ/MISTRAL_API_KEY
airlock_batch:
backend: mistral
provider_model: mistral-small-latest
- model_name: mistral/codestral
litellm_params:
model: mistral/codestral-latest
api_key: os.environ/MISTRAL_API_KEY
- model_name: mistral/magistral-medium
litellm_params:
model: mistral/magistral-medium-latest
api_key: os.environ/MISTRAL_API_KEY
# --- Perplexity Sonar ---
- model_name: perplexity-sonar
litellm_params:
model: perplexity/sonar
api_key: os.environ/PERPLEXITY_API_KEY
- model_name: perplexity-sonar-pro
litellm_params:
model: perplexity/sonar-pro
api_key: os.environ/PERPLEXITY_API_KEY
- model_name: perplexity-sonar-reasoning-pro
litellm_params:
model: perplexity/sonar-reasoning-pro
api_key: os.environ/PERPLEXITY_API_KEY
- model_name: perplexity-sonar-deep-research
litellm_params:
model: perplexity/sonar-deep-research
api_key: os.environ/PERPLEXITY_API_KEY
# Provider-prefixed Perplexity aliases (0.5.2) — same litellm_params as above.
- model_name: perplexity/sonar
litellm_params:
model: perplexity/sonar
api_key: os.environ/PERPLEXITY_API_KEY
- model_name: perplexity/sonar-pro
litellm_params:
model: perplexity/sonar-pro
api_key: os.environ/PERPLEXITY_API_KEY
- model_name: perplexity/sonar-reasoning-pro
litellm_params:
model: perplexity/sonar-reasoning-pro
api_key: os.environ/PERPLEXITY_API_KEY
- model_name: perplexity/sonar-deep-research
litellm_params:
model: perplexity/sonar-deep-research
api_key: os.environ/PERPLEXITY_API_KEY
# --- Tavily (web search as chat completion) ---
- model_name: tavily-search
litellm_params:
model: tavily/web-search
api_key: os.environ/TAVILY_API_KEY
# Provider-prefixed Tavily alias (0.5.2) — == litellm_params.model.
- model_name: tavily/web-search
litellm_params:
model: tavily/web-search
api_key: os.environ/TAVILY_API_KEY
# --- Smart Router (auto-selects model by query complexity) ---
# Send model: "smart" and Airlock auto-classifies prompt complexity:
# simple → low tier (claude-haiku, gemini-flash, gpt-5-nano, mistral-small)
# moderate → medium tier (claude-sonnet, gemini-pro, gpt-5-mini, mistral-medium)
# complex → high tier (claude-opus, gpt-5, gpt-5-pro, magistral-medium)
# Native heuristic classifier in router.py — no ML deps, ~50μs latency.
# Thresholds configurable via AIRLOCK_SMART_THRESHOLDS env var.
# Tier membership is overridable via the `cost_tiers:` block below or
# AIRLOCK_COST_TIERS env var. Composes with session_id / prefer_provider.
# --- Local vLLM (Gemma 4 31B AWQ) ---
- model_name: gemma-4
litellm_params:
model: openai/gemma4-31b
api_base: http://192.168.1.45:8000/v1
api_key: os.environ/VLLM_API_KEY
# --- Local vLLM (Kimi-Dev 72B AWQ-4bit) ---
# Shares the same vLLM host as gemma-4 (only one local model loaded at a time).
# Reasoning blocks ◁think▷…◁/think▷ are stripped by the
# airlock-reasoning-stripper guardrail (Kimi's delimiters aren't single
# tokens, so vLLM's native --reasoning-parser cannot handle them).
- model_name: kimi-dev
litellm_params:
model: openai/kimi-dev-72b
api_base: http://192.168.1.45:8000/v1
api_key: os.environ/VLLM_API_KEY
# --- Local vLLM (Qwen3 32B AWQ) ---
# Official Qwen AWQ-4bit (~19 GB). Dense model, comfortable 32k context.
- model_name: qwen3-32b
litellm_params:
model: openai/qwen3-32b
api_base: http://192.168.1.45:8000/v1
api_key: os.environ/VLLM_API_KEY
# --- Local vLLM (Qwen3.6 27B AWQ-INT4, cyankiwi) ---
# Larger-than-typical AWQ (~41 GB; mixed-precision compressed-tensors).
# Conservative context until tuned (see start-qwen36-27b.sh).
- model_name: qwen3.6-27b
litellm_params:
model: openai/qwen3.6-27b
api_base: http://192.168.1.45:8000/v1
api_key: os.environ/VLLM_API_KEY
# Provider-prefixed local-vLLM aliases (0.5.2) — served-by `openai`
# (OpenAI-compatible endpoint). vllm/qwen3.6-27b carries the airlock_batch
# marker (consolidating qwen36-27b-vllm-batch) so it serves sync AND batch.
- model_name: vllm/gemma-4
litellm_params:
model: openai/gemma4-31b
api_base: http://192.168.1.45:8000/v1
api_key: os.environ/VLLM_API_KEY
- model_name: vllm/kimi-dev
litellm_params:
model: openai/kimi-dev-72b
api_base: http://192.168.1.45:8000/v1
api_key: os.environ/VLLM_API_KEY
- model_name: vllm/qwen3-32b
litellm_params:
model: openai/qwen3-32b
api_base: http://192.168.1.45:8000/v1
api_key: os.environ/VLLM_API_KEY
- model_name: vllm/qwen3.6-27b
litellm_params:
model: openai/qwen3.6-27b
api_base: http://192.168.1.45:8000/v1
api_key: os.environ/VLLM_API_KEY
airlock_batch:
backend: vllm
provider_model: qwen3.6-27b
# --- Internal RAG (Phase 3 — uncomment when ready) ---
# - model_name: internal-docs
# litellm_params:
# model: openai/internal-docs
# api_base: http://rag-service.internal:8000/v1
# api_key: os.environ/INTERNAL_RAG_KEY
# ---------------------------------------------------------------------------
# Proxy settings
# ---------------------------------------------------------------------------
litellm_settings:
# Drop any unsupported params instead of erroring
drop_params: true
# Inject a dummy user message when messages=[] so Anthropic doesn't reject the request
modify_params: true
# Custom LLM providers — Tavily search exposed as a chat completion model.
# get_instance_fn resolves the dotted path to a module-level instance.
custom_provider_map:
- provider: tavily
custom_handler: airlock.providers.tavily_provider.tavily_handler
- provider: enhanced
custom_handler: airlock.providers.enhanced_passthrough.enhanced_handler
# Callbacks — enterprise logging + fast subsystem metrics
# NOTE: these must point at module-level *instances*, not classes.
# LiteLLM's get_instance_fn does getattr — classes fail isinstance(cb, CustomLogger).
callbacks: ["airlock.callbacks.model_override_headers.proxy_model_override_headers"]
success_callback: ["airlock.callbacks.recorder.recorder_callback", "airlock.fast.monitor.proxy_monitor"] # recorder owns enterprise+fathom dispatch (fathom gated by AIRLOCK_ENABLE_FATHOM_LOGGER); MUST stay before proxy_monitor
failure_callback: ["airlock.callbacks.recorder.recorder_callback", "airlock.fast.monitor.proxy_monitor"] # recorder owns enterprise+fathom dispatch (fathom gated by AIRLOCK_ENABLE_FATHOM_LOGGER); MUST stay before proxy_monitor
# Budget & rate-limiting (per-user, requires virtual keys)
# max_budget: 100.0 # USD per 30-day rolling window
# budget_duration: 30d
# Request timeout (seconds)
request_timeout: 300
# ---------------------------------------------------------------------------
# Files settings — provider config for the Batch/Files API (/v1/files, /v1/batches)
# ---------------------------------------------------------------------------
# Required so /v1/files can resolve a provider for non-vertex batch. vertex_ai is
# special-cased by LiteLLM (needs no entry; uses ADC + GCS_BUCKET_NAME). Call
# /v1/files and /v1/batches with custom_llm_provider=openai (or vertex_ai).
files_settings:
- custom_llm_provider: openai
api_key: os.environ/OPENAI_API_KEY
# ---------------------------------------------------------------------------
# Router settings
# ---------------------------------------------------------------------------
router_settings:
routing_strategy: cost-based-routing
# provider_budget_config — ONE number drives THREE behaviors per provider:
# 1. LiteLLM hard block — request refused once spend hits budget_limit.
# 2. Monitor near-limit warn — emits `X-Airlock-Budget-State: near_limit` as spend
# approaches the cap (default warn at 80%).
# 3. Fast-router reroute — proactively routes away from a provider as its spend
# approaches the cap (default at 90%).
# `budget_limit: 0` means NO enforcement / unlimited / no warn / no reroute (the falsy
# short-circuit is identical across all three layers). `time_period: "1d"` is the
# rolling enforcement window. Override the whole map with the AIRLOCK_PROVIDER_BUDGETS
# env var (JSON, e.g. {"anthropic": 50}). With no provider_budget_config and no env
# override there is NO hidden default budget — nothing is blocked, warned, or rerouted.
provider_budget_config:
anthropic:
budget_limit: 0 # 0 = no enforcement (unlimited); LiteLLM skips the check on falsy max_budget
time_period: "1d"
openai:
budget_limit: 0 # 0 = no enforcement (unlimited)
time_period: "1d"
gemini:
budget_limit: 0 # 0 = no enforcement (unlimited)
time_period: "1d"
mistral:
budget_limit: 0 # 0 = no enforcement (unlimited)
time_period: "1d"
perplexity:
budget_limit: 0 # 0 = no enforcement (unlimited)
time_period: "1d"
fallbacks:
- claude-opus: [claude-sonnet, gpt-5-pro, gemini-3.1-pro]
- claude-sonnet: [claude-haiku, gpt-5-mini, gemini-pro]
- claude-haiku: [gemini-flash, gpt-5-nano, mistral-small]
- gpt-5-pro: [gpt-5, claude-opus, gemini-3.1-pro]
- gpt-5: [gpt-5.4, gpt-5-mini, claude-sonnet]
- gpt-5.4: [gpt-5, gpt-5-mini, claude-sonnet]
- gpt-5-mini: [gpt-5-nano, claude-haiku, gemini-flash]
- gpt-5-nano: [claude-haiku, gemini-flash, mistral-small]
- gpt-5-codex: [codestral, gpt-5, claude-sonnet]
# GPT-5.6 — intra-family first (identical 1.05M ctx + capability envelope,
# so the swap is genuinely transparent), then cross-provider. Chains are
# grouped by BODY so aliases sharing a body fail over identically.
# Luna deliberately does NOT fall back to claude-haiku: haiku caps at 200K
# vs 5.6's 1.05M, so a long-context failover would HARD-FAIL rather than
# degrade, turning a retryable blip into a client error.
- gpt-5.6-sol: [gpt-5.6-terra, gpt-5.6-luna, claude-opus]
- openai/gpt-5.6-sol: [gpt-5.6-terra, gpt-5.6-luna, claude-opus]
- gpt-5.6: [gpt-5.6-terra, gpt-5.6-luna, claude-opus]
- openai/gpt-5.6: [gpt-5.6-terra, gpt-5.6-luna, claude-opus]
- gpt-5.6-terra: [gpt-5.6-luna, gpt-5.6-sol, claude-sonnet]
- openai/gpt-5.6-terra: [gpt-5.6-luna, gpt-5.6-sol, claude-sonnet]
- gpt-5.6-luna: [gpt-5.6-terra, gemini-flash, claude-sonnet]
- openai/gpt-5.6-luna: [gpt-5.6-terra, gemini-flash, claude-sonnet]
- gemini-flash: [claude-haiku, gpt-5-nano, mistral-small]
- gemini-flash-lite: [gemini-flash, claude-haiku, gpt-5-nano]
- gemini-pro: [claude-sonnet, gpt-5-mini, mistral-large]
- gemini-3.5-flash: [gemini-3-flash, gemini-flash, claude-haiku]
- gemini-3.1-flash-lite: [gemini-flash-lite, gemini-flash, gpt-5-nano]
- gemini-3-flash: [gemini-flash, claude-haiku, gpt-5-nano]
- gemini-3.1-pro: [gemini-3-flash, claude-opus, gpt-5-pro]
- gemini-3.1-pro-tools: [gemini-3.1-pro, gemini-3-flash, claude-opus]
- gemini-coding: [gemini-3.1-pro-tools, claude-opus, gpt-5-codex]
- mistral-large: [claude-sonnet, gpt-5, gemini-pro]
- mistral-medium: [mistral-large, claude-sonnet, gpt-5-mini]
- mistral-small: [claude-haiku, gemini-flash, gpt-5-nano]
- codestral: [gpt-5-codex, claude-sonnet, gpt-5]
- magistral-medium: [claude-opus, gpt-5-pro, gemini-3.1-pro]
- perplexity-sonar: [perplexity-sonar-pro]
- perplexity-sonar-pro: [perplexity-sonar-reasoning-pro, perplexity-sonar]
- perplexity-sonar-reasoning-pro: [perplexity-sonar-pro]
- gemma-4: [claude-haiku, gemini-flash, mistral-small]
# ---------------------------------------------------------------------------
# Cost tiers — Airlock routing directive targets
# ---------------------------------------------------------------------------
# Used by cost_tier / smart routing directives. Each tier is an ordered list
# of `model_name` aliases from `model_list` above; the first entry is the
# default swap target for that tier. Env var AIRLOCK_COST_TIERS (JSON) takes
# precedence if set.
cost_tiers:
low:
- claude-haiku
- gemini-flash
- gemini-flash-lite
- gpt-5-nano
- mistral-small
- gemma-4
# GPT-5.6 Luna ($1/$6) — the CEILING of low, not the floor. gpt-5-nano
# ($0.20/$1.25) stays first so the default swap target is unchanged.
- gpt-5.6-luna
- openai/gpt-5.6-luna
medium:
- claude-sonnet
- gemini-pro
- gpt-5-mini
- mistral-medium
- codestral
# GPT-5.6 Terra ($2.50/$15) — just under claude-sonnet ($3/$15).
- gpt-5.6-terra
- openai/gpt-5.6-terra
high:
- claude-opus
- gpt-5
- gpt-5.4 # cheaper previous flagship ($2.50/$15) for cost-based routing
- gpt-5-pro
- mistral-large
- magistral-medium
- gemini-3.1-pro
# GPT-5.6 Sol ($5/$30) — with claude-opus ($5/$25). EVERY callable 5.6
# alias must appear in exactly one tier: _apply_cost_tier is a membership
# test that FORCE-SWAPS to tier_models[0] on a miss, so an untiered alias
# is silently rerouted to a different model.
- gpt-5.6-sol
- openai/gpt-5.6-sol
- gpt-5.6
- openai/gpt-5.6
# ---------------------------------------------------------------------------
# Model alias disclosure — client-visible, advisory only
# ---------------------------------------------------------------------------
# A configured key opts the alias into X-Airlock-Model-Alias. The header always
# reports the actual LiteLLM model id that served the request; the value is a
# directly callable successor/current-generation alias for clients that want to
# pin their choice. Both sides are checked against model_list by the test suite.
# AIRLOCK_MODEL_SUCCESSORS accepts the same mapping as JSON and takes precedence.
model_successors:
gpt-5: openai/gpt-5.6-sol
gemini-flash: gemini-3-flash
gemini-flash-lite: gemini-3.1-flash-lite
gemini-pro: gemini-3.1-pro
# ---------------------------------------------------------------------------
# Transparency — mutation ledger + served-backend attribution (0.5.0)
# ---------------------------------------------------------------------------
# All additive and backward-compatible — absent → body and behavior unchanged
# (CC-T7). Defaults shown; this block is read by airlock/transparency.py.
transparency:
mutation_headers: compact # off | compact | full
served_headers: true # X-Airlock-Served-By / -Region
explain_body_optin_header: X-Airlock-Explain # request header that adds body envelope
attribute_accounting_to_served: true # key spend/quarantine off the served provider
mutation_header_budget_bytes: 256
# ---------------------------------------------------------------------------
# Circuit breaker — provider-protection tuning (A1 + E)
# ---------------------------------------------------------------------------
# Optional. Omit this block entirely to keep the historical one-strike / 300 s
# behaviour (defaults shown). Per-client keys (key:<last8> of the virtual key, or
# the airlock_client identity) override the global defaults. The env var
# AIRLOCK_BREAKER_OVERRIDES (JSON: {"defaults":{...},"clients":{...}}) takes
# precedence if set. Read once at startup — a config change needs a proxy restart.
# airlock_settings:
# circuit_breaker:
# rate_limit_threshold: 1 # 429s within the window before quarantine
# rate_limit_window_seconds: 300
# client_cooldown_seconds: 300
# provider_cooldown_seconds: 300
# provider_escalation_client_threshold: 2
# clients:
# "key:b35cf679": # a trusted first-party batch client
# rate_limit_threshold: 8 # tolerate bursts before tripping
# client_cooldown_seconds: 30 # short cooldown
# escalation_exempt: true # never quarantine the provider for others
# # disabled: true # turn the breaker off for this client
# ---------------------------------------------------------------------------
# Admin control plane (off by default) — see docs/guide/admin-api.md
# ---------------------------------------------------------------------------
# When admin.enabled is false (default), /airlock/admin/* returns 404. Auth is
# either loopback (operator) or a bearer credential (master key or a capability
# JWT). Bearer admin over a non-loopback bind without TLS is refused at startup
# unless behind_tls_proxy or allow_insecure_tokens is set (set AIRLOCK_JWT_SECRET
# for token signing; it falls back to deriving from AIRLOCK_MASTER_KEY).
# admin:
# enabled: false
# trust_loopback: true # a request on the loopback interface is the operator
# behind_tls_proxy: false # assert TLS is terminated upstream (skips fail-closed)
# allow_insecure_tokens: false # last resort: permit bearer admin over plaintext
# ---------------------------------------------------------------------------
# Guardrail overrides — per-request capability skip (off by default)
# ---------------------------------------------------------------------------
# A trusted client may present X-Airlock-Capability (a guardrail:skip:<name> JWT)
# to downgrade a CONTENT guard for that request. "Skip" = downgrade to observe
# (still scans + logs) unless downgrade_to: off. PII is non-skippable by default.
# The token sub must be the client's authenticated key id (key:<last8>).
# guardrail_overrides:
# allow_capability_skip: false
# capability_header: X-Airlock-Capability
# skippable:
# pii_redact: { skippable: false }
# keyword: { skippable: true, downgrade_to: observe }
# response_scan: { skippable: true, downgrade_to: observe }
# reasoning_strip: { skippable: true, downgrade_to: off }
# ---------------------------------------------------------------------------
# Batch profile — Airlock Batch Gateway posture (design §3.3 / §4.2)
# ---------------------------------------------------------------------------
# One place to declare the batch posture, distinct from the interactive chat
# default. The batch path ignores caller-supplied guardrail/metadata disables
# (operator/config-controlled trust boundary). `scan_at_upload` runs the async
# content scan at upload (keyword reject + PII redaction), gated at batch
# create — see dev/design-batch-content-scan.md.
batch_profile:
default:
scan_at_upload: true # async upload scan: keyword reject + PII redaction, gated at create
keyword_block: true
pii_redact: false # TEMP: off for the LME synthetic/public real run; RE-ENABLE (true) after — see dev/notes/handoff-fathom-vllm-batch-throughput.md
pii_hydrate_output: false # terminal redaction only (A2); no PII reverse-map
output_scan_mode: observe
max_rows: 50000
max_bytes: 2147483648
max_concurrent_jobs: 5
# ---------------------------------------------------------------------------
# Guardrails
# ---------------------------------------------------------------------------
guardrails:
- guardrail_name: airlock-oom-diagnostics
litellm_params:
guardrail: airlock.callbacks.oom_diagnostics.AirlockOOMDiagnosticsGuard
mode: [pre_call, post_call]
default_on: true
- guardrail_name: airlock-pii-guard
litellm_params:
guardrail: airlock.guardrails.pii_guard.AirlockPIIGuard
mode: [pre_call, pre_mcp_call] # redaction — must run first
default_on: true
- guardrail_name: airlock-keyword-guard
litellm_params:
guardrail: airlock.guardrails.keyword_guard.AirlockKeywordGuard
mode: [pre_call, pre_mcp_call]
default_on: true
- guardrail_name: airlock-enhanced-interceptor
litellm_params:
guardrail: airlock.guardrails.enhanced_interceptor.EnhancedModelInterceptor
mode: [pre_call, pre_mcp_call]
default_on: true
- guardrail_name: airlock-fast-guardian
litellm_params:
guardrail: airlock.fast.guardian.AirlockFastGuardian
mode: [pre_call, pre_mcp_call]
default_on: true
- guardrail_name: airlock-enforcer
litellm_params:
guardrail: airlock.guardrails.enforcer.AirlockEnforcer
mode: [pre_call, pre_mcp_call]
default_on: true
- guardrail_name: airlock-semantic-guard
litellm_params:
guardrail: airlock.guardrails.semantic.AirlockSemanticGuard
mode: [during_call, during_mcp_call]
default_on: true
- guardrail_name: airlock-orchestrator
litellm_params:
guardrail: airlock.guardrails.orchestrator.AirlockOrchestrator
mode: [during_call, during_mcp_call]
default_on: true
- guardrail_name: airlock-mcp-tool-guard
litellm_params:
guardrail: airlock.guardrails.mcp_tool_guard.AirlockMCPToolGuard
mode: pre_mcp_call # MCP-only: tool allowlist/blocklist + arg sanitization
default_on: true
- guardrail_name: airlock-response-scanner
litellm_params:
guardrail: airlock.guardrails.response_scanner.AirlockResponseScanner
mode: post_call
default_on: true
- guardrail_name: airlock-reasoning-stripper
litellm_params:
guardrail: airlock.guardrails.reasoning_stripper.AirlockReasoningStripper
mode: post_call # model-scoped via AIRLOCK_REASONING_STRIP_MODELS (default: kimi-dev)
default_on: true
- guardrail_name: airlock-local-vllm-router
litellm_params:
guardrail: airlock.guardrails.local_vllm_router.AirlockLocalVLLMRouter
mode: pre_call # fails fast when requested local alias isn't the loaded vLLM model
default_on: true
- guardrail_name: airlock-pii-hydrator
litellm_params:
guardrail: airlock.guardrails.pii_guard.AirlockPIIGuard
mode: post_call # hydration — must run after response scanner
default_on: true
# ---------------------------------------------------------------------------
# MCP Servers
# ---------------------------------------------------------------------------
mcp_servers:
# NOTE: machine-specific MCP servers with absolute local command paths (e.g. ones
# under a developer's home directory) do NOT belong in this tracked config. Keep
# them in `config.local.yaml` (see `config.local.yaml.example`). The repository
# ships an empty fallback for the tracked include; keep machine-specific additions
# uncommitted. LiteLLM merges it, and a local `mcp_servers` mapping replaces this
# entire mapping, so it must repeat the servers that should remain enabled.
# --- NewsCatcher CatchAll (stdio) ---
# Web search with NLP enrichment via CatchAll API (jobs-based, async).
# Jobs take 2-10+ minutes. Best for batch research, not real-time.
newscatcher:
transport: stdio
command: python3
args: ["-m", "airlock.mcp_servers.newscatcher_server"]
env:
NEWS_CATCHER_API_KEY: os.environ/NEWS_CATCHER_API_KEY
# ---------------------------------------------------------------------------
# General proxy behaviour
# ---------------------------------------------------------------------------
general_settings:
# Master key protects the /key/generate endpoint (set via env)
master_key: os.environ/AIRLOCK_MASTER_KEY