Repository navigation
Expand file tree
/
Copy pathpyproject.toml
More file actions
307 lines (292 loc) · 9.83 KB
/
Copy pathpyproject.toml
File metadata and controls
307 lines (292 loc) · 9.83 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
[build-system]
requires = ["uv_build==0.7.6"]
build-backend = "uv_build"
[tool.uv]
environments = [
"sys_platform == 'linux' and platform_machine == 'x86_64'",
"sys_platform == 'linux' and platform_machine == 'aarch64'",
"sys_platform == 'darwin' and platform_machine == 'x86_64'",
"sys_platform == 'darwin' and platform_machine == 'arm64'",
]
# bfcl-eval hard-pins an old dependency set (numpy==1.26.4, filelock, etc.).
# Mark bfcl as conflicting with the tooling extras so uv resolves it in its own
# fork; otherwise those old pins drag shared dev/test/performance deps (filelock,
# virtualenv) down to versions with known CVEs. CI installs the tooling extras
# without bfcl; bfcl is installed standalone (pip install -e ".[bfcl]").
conflicts = [
[{ extra = "bfcl" }, { extra = "dev" }],
[{ extra = "bfcl" }, { extra = "test" }],
[{ extra = "bfcl" }, { extra = "performance" }],
]
# CVE floors for transitive deps we don't declare directly. click is pulled in
# via transformers -> typer; 8.3.2 carries PYSEC-2026-2132 (fixed in 8.3.3).
# setuptools is pulled in via pbr/torch; 82.0.1 carries PYSEC-2026-3447 (fixed
# in 83.0.0).
constraint-dependencies = ["click>=8.3.3", "setuptools>=83.0.0"]
[tool.uv.build-backend]
module-root = "src"
source-exclude = [
"inference_endpoint/evaluation/livecodebench/_server.py",
# Isolated uv subproject (own pyproject.toml/uv.lock; pinned transformers /
# numpy<2 / prm800k that conflict with the parent env). Invoked only via
# `uv run --project`, never imported - keep it out of the parent wheel.
"inference_endpoint/evaluation/legacy_mlperf_deepseek_r1/**",
# Isolated native uv service that owns SWE-bench Docker execution.
"inference_endpoint/evaluation/swebench_service/**",
]
[project]
name = "inference-endpoint"
version = "0.1.0"
description = "High-performance LLM endpoint benchmarking system with 50k QPS capability"
readme = "README.md"
license = {text = "Apache-2.0"}
authors = [
{name = "MLPerf Inference Endpoint Team"}
]
keywords = ["mlperf", "benchmarking", "llm", "endpoint", "performance"]
classifiers = [
"Development Status :: 3 - Alpha",
"Intended Audience :: Developers",
"Intended Audience :: Science/Research",
"License :: OSI Approved :: Apache Software License",
"Operating System :: OS Independent",
"Programming Language :: Python :: 3",
"Programming Language :: Python :: 3.12",
"Topic :: Scientific/Engineering :: Artificial Intelligence",
"Topic :: Software Development :: Testing",
"Topic :: System :: Benchmark",
]
requires-python = ">=3.12"
dependencies = [
# Core Python dependencies
"typing-extensions==4.15.0",
# System utilities
"psutil==7.2.2",
# Async and networking
"pyzmq==27.1.0",
"uvloop==0.22.1",
"httptools==0.7.1",
"websocket-client==1.9.0",
# Data handling
"duckdb==1.5.1",
"msgspec==0.20.0",
"pydantic==2.12.5",
"pydantic_core==2.41.5",
# YAML/CLI from Pydantic Schema
"cyclopts==4.10.0",
"rich==14.3.3",
# Needed for tokenization and OSL reporting
"transformers==5.16.1",
"tiktoken==0.13.0",
# Required by transformers' apply_chat_template
"jinja2==3.1.6",
"numpy>=1.26.4",
"datasets==5.0.1",
"Pillow==12.3.0",
"sentencepiece==0.2.1",
"protobuf==7.34.1",
"openai_harmony==0.0.8",
# HDR Histogram for live percentile/histogram approximations in the
# metrics aggregator (PyPI: hdrhistogram, importable as hdrh.histogram).
"hdrhistogram==0.10.3",
# Color support for cross-platform terminals
"colorama==0.4.6",
# Fix pytz-2024 import warning
"pytz==2026.1.post1",
"urllib3==2.8.0",
"pyyaml==6.0.3",
# anyio is pulled in transitively (openai -> httpx -> anyio); pinned here to
# force past CVE-2026-63374 / CVE-2026-64847 (both fixed in 4.14.2), which
# otherwise fail `uv run pip-audit`.
"anyio==4.14.2",
]
[project.optional-dependencies]
sql = [
# SQL event logger (swappable backends, default sqlite)
"sqlalchemy==2.0.48",
]
dev = [
# Code quality
"ruff==0.15.8",
# Development tools
"pre-commit==4.5.1",
# Profiling tools
"line-profiler==5.0.2",
"Pympler==1.1",
# Documentation
"sphinx==9.1.0",
"sphinx-rtd-theme==3.1.0",
"sphinx-autodoc-typehints==3.9.11",
"myst-parser==5.0.0",
# Security auditing
"pip-audit==2.10.0",
# bfcl-eval hard-pins filelock==3.20.0, which uv would otherwise share across
# every resolution fork. Because bfcl conflicts with this extra (see
# [tool.uv].conflicts), these floors force uv to fork filelock/virtualenv:
# patched here, pinned only inside the bfcl fork. Closes CVE-2025-68146 /
# CVE-2026-22701 (filelock) and CVE-2026-22702 (virtualenv).
"filelock>=3.20.3",
"virtualenv==21.7.13",
# pip is pulled in transitively (pip-audit -> pip-api -> pip); force past
# PYSEC-2026-3721 (fixed in 26.2), which otherwise fails `uv run pip-audit`.
"pip==26.2",
]
test = [
# Includes optional dependencies for full test coverage
"inference-endpoint[sql]",
"inference-endpoint[rouge]",
# Testing framework
"pytest==9.0.3",
"pytest-asyncio==1.3.0",
"pytest-cov==7.1.0",
"pytest-benchmark==5.2.3",
"pytest-timeout==2.4.0",
"pytest-xdist==3.8.0",
# Test utilities
"coverage==7.13.4",
"line-profiler==5.0.2",
"Pympler==1.1",
"scipy==1.17.1",
# HTTP server and client for mock server fixture
"aiohttp==3.14.3",
# Plotting for benchmark sweep mode
"matplotlib==3.10.8",
# Property-based testing (CLI fuzz)
"hypothesis==6.151.10",
]
performance = [
"pytest-benchmark==5.2.3",
"memory-profiler==0.61.0",
]
rouge = [
# ROUGE text-generation scoring (eval_method: "rouge"), used by the Llama
# CNN-DailyMail / OpenOrca accuracy paths. Imported lazily in
# evaluation/scoring.py, so the base install stays lean. Resolves cleanly
# against the pinned datasets==5.0.1 - no conflicts fork needed, unlike bfcl.
# NOTE: not self-contained at runtime - nltk needs the punkt/punkt_tab
# corpora and evaluate.load("rouge") fetches its metric script from the HF
# Hub, so an offline image must prefetch both.
# nltk advisory PYSEC-2026-3740 has no fixed release and is ignored in CI's
# pip-audit (affected APIs unused); tracked in
# https://github.com/mlcommons/endpoints/issues/520
"nltk==3.10.3",
"evaluate==0.4.6",
"rouge-score==0.1.2",
]
bfcl = [
# BFCL v4 function-calling evaluation. Pins numpy==1.26.4, which is why
# the top-level numpy requirement is a lower bound (>=1.26.4).
"bfcl-eval==2026.3.23",
# bfcl-eval's qwen model handler transitively imports qwen_agent → soundfile;
# soundfile is not used by our scorer but must be present for the import to succeed.
"soundfile==0.13.1",
# Used directly by the multi-turn runner (bfcl_v4_multi_turn_runner.py).
# Present transitively via openai, but declared explicitly so the import
# does not depend on another package keeping it.
"httpx==0.28.1",
]
[project.scripts]
inference-endpoint = "inference_endpoint.main:run"
[project.urls]
Homepage = "https://github.com/mlperf/inference-endpoint"
Documentation = "https://github.com/mlperf/inference-endpoint#readme"
Repository = "https://github.com/mlperf/inference-endpoint.git"
Issues = "https://github.com/mlperf/inference-endpoint/issues"
[tool.ruff]
target-version = "py312"
line-length = 88
exclude = [
"tests/assets/datasets/*",
"src/inference_endpoint/openai/openai_types_gen.py",
"src/inference_endpoint/openai/openapi.yaml",
# Vendored third-party MLCommons evaluator - keep verbatim.
"src/inference_endpoint/evaluation/legacy_mlperf_deepseek_r1/mlperf_eval/",
"datasets/*",
"datasets/",
]
[tool.ruff.lint]
select = [
"E", # pycodestyle errors
"W", # pycodestyle warnings
"F", # pyflakes
"I", # isort
"B", # flake8-bugbear
"C4", # flake8-comprehensions
"UP", # pyupgrade
]
ignore = [
"E501", # line too long, handled by ruff-format
"B008", # do not perform function calls in argument defaults
"C901", # too complex
]
[tool.ruff.format]
quote-style = "double"
indent-style = "space"
skip-magic-trailing-comma = false
line-ending = "auto"
[tool.pytest.ini_options]
testpaths = ["tests"]
python_files = ["test_*.py", "*_test.py"]
python_classes = ["Test*"]
python_functions = ["test_*"]
asyncio_mode = "strict"
timeout = 300
timeout_func_only = true
addopts = [
"--strict-markers",
"--strict-config",
"--cov=src",
"--cov-report=term-missing",
"--cov-report=html",
"--cov-report=xml",
"-m not run_explicitly",
]
markers = [
"slow: marks tests as slow (skip in CI)",
"performance: marks tests as performance tests (skip in CI) - no timeout",
"integration: marks tests as integration tests",
"unit: marks tests as unit tests",
"run_explicitly: mark test to only run explicitly",
"schema_fuzz: hypothesis CLI fuzz tests (run in CI on schema changes)",
]
filterwarnings = [
"ignore:Session timeout reached:RuntimeWarning",
]
[tool.coverage.run]
source = ["src"]
concurrency = ["multiprocessing", "thread"]
parallel = true
sigterm = true
omit = [
"*/tests/*",
"*/test_*",
"*/__pycache__/*",
"*/venv/*",
"*/\\.venv/*",
]
[tool.coverage.report]
exclude_lines = [
"pragma: no cover",
"def __repr__",
"if self.debug:",
"if settings.DEBUG",
"raise AssertionError",
"raise NotImplementedError",
"if 0:",
"if __name__ == .__main__.:",
"class .*\\\\bProtocol\\):",
"@(abc\\\\.)?abstractmethod",
]
[tool.mypy]
ignore_missing_imports = true
exclude = [
"openai_types_gen\\.py$",
"metrics/reporter\\.py$",
"legacy_mlperf_deepseek_r1/mlperf_eval/",
]
[[tool.mypy.overrides]]
module = [
"inference_endpoint.openai.openai_types_gen",
"inference_endpoint.metrics.reporter",
]
ignore_errors = true