Repository navigation
Expand file tree
/
Copy pathpyproject.toml
More file actions
executable file
·202 lines (185 loc) · 7.44 KB
/
Copy pathpyproject.toml
File metadata and controls
executable file
·202 lines (185 loc) · 7.44 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
[build-system]
requires = ["setuptools>=42", "wheel"]
build-backend = "setuptools.build_meta"
[project]
name = "mlpstorage"
version = "4.0.1"
description = "MLPerf Storage Benchmark Suite"
readme = "README.md"
license = {text = "Apache-2.0"}
authors = [
{name = "MLCommons Storage Working Group"}
]
requires-python = ">=3.12, <3.13"
dependencies = [
"psutil>=5.9",
"pyarrow",
"pyyaml>=6.0",
"packaging>=21.0",
"rich>=13.0",
"dlio-benchmark", # Required dependency
"pydantic>=2.13.4",
"minio>=7.2.20",
"s3torchconnector>=1.5.0",
"python-dotenv>=1.0.0",
"s3dlio>=0.9.112",
]
[project.optional-dependencies]
test = [
"pytest>=7.0",
"pytest-cov>=4.0",
"pytest-mock>=3.0",
"pytest-xdist>=3.0",
]
full = [
"dlio-benchmark",
]
vectordb = [
"pymilvus>=2.4.0",
"psycopg2-binary>=2.9",
"pgvector>=0.2",
"elasticsearch>=8.0",
"numpy>=1.24",
"pandas>=2.0",
"pyyaml>=6.0",
"tabulate>=0.9",
]
vectordb-milvus = [
"pymilvus>=2.4.0",
"numpy>=1.24",
"pandas>=2.0",
"pyyaml>=6.0",
"tabulate>=0.9",
]
vectordb-pgvector = [
"psycopg2-binary>=2.9",
"pgvector>=0.2",
"numpy>=1.24",
"pandas>=2.0",
"pyyaml>=6.0",
"tabulate>=0.9",
]
vectordb-elasticsearch = [
"elasticsearch>=8.0",
"numpy>=1.24",
"pandas>=2.0",
"pyyaml>=6.0",
"tabulate>=0.9",
]
[project.urls]
"Homepage" = "https://github.com/mlcommons/storage"
"Bug Tracker" = "https://github.com/mlcommons/storage/issues"
[tool.setuptools]
packages = {find = {include = ["mlpstorage_py*", "vdbbench*"]}}
[tool.setuptools.package-data]
"mlpstorage_py" = [
"../configs/dlio/workload/*.yaml",
"system_description/sysctl_allowlist.txt",
]
[project.scripts]
mlpstorage = 'mlpstorage_py.main:main'
[tool.pytest.ini_options]
testpaths = ["tests"]
python_files = ["test_*.py"]
python_functions = ["test_*"]
addopts = "-v --tb=short --setup-show -m 'not slow'"
filterwarnings = [
"ignore::DeprecationWarning",
]
markers = [
"slow: tests that invoke the real mlpstorage CLI / MPI and take >10s. Excluded from the default suite; opt in with `pytest -m slow` (or `pytest -m ''` to run everything).",
"integration: subprocess + end-to-end tests (Phase 3 DoD per CONTEXT.md D-E1)",
]
[tool.coverage.run]
source = ["mlpstorage_py"]
omit = ["*/tests/*", "*/__pycache__/*"]
[tool.coverage.report]
exclude_lines = [
"pragma: no cover",
"def __repr__",
"raise NotImplementedError",
"if __name__ == .__main__.:",
]
# =============================================================================
# UV Configuration
# =============================================================================
[[tool.uv.index]]
name = "pytorch-cpu"
url = "https://download.pytorch.org/whl/cpu"
explicit = true
[tool.uv]
# s3dlio only ships Linux wheels — restrict resolution to Linux.
environments = ["sys_platform == 'linux'"]
[tool.uv.sources]
torch = [{ index = "pytorch-cpu" }]
torchvision = [{ index = "pytorch-cpu" }]
torchaudio = [{ index = "pytorch-cpu" }]
# Pinned to a specific commit for reproducibility; bump by updating `rev` and
# re-running `uv lock --upgrade-package dlio-benchmark`.
# Commit history (most recent first):
# - DLIO PR #51 v3.0.5: dataset.num_files_generated — under skip_listing
# the "_of_{total}" name suffix (and its zero-pad width) comes from the
# GENERATED count while num_files_train stays the read count, so a run
# can read the first N files of a larger generated set (storage #571
# Q4 / #795 follow-up). mlpstorage fills it from the datagen manifest.
# Also aligns two generator conventions in the reconstruction
# (subfolder pad width for power-of-ten counts; num_subfolders==1 flat).
# - DLIO PR #50 v3.0.4: three fixes (storage #780, #756, #755) plus
# s3dlio pin bump to >=0.9.112.
# - DLIO PR #49 fix for storage #754: race-tolerant Hydra job-dir
# bootstrap on multi-node runs — Hydra's own run_job/_save_config
# internals call pathlib.Path(...).mkdir(exist_ok=True) on the shared
# hydra.run.dir / output_subdir from every rank on every node, and
# pathlib's post-FileExistsError is_dir() recheck can observe stale
# metadata from a peer's just-completed mkdir on shared/networked FS
# and re-raise. Fix monkeypatches Path.mkdir for the duration of the
# run_benchmark() call so the race is absorbed instead of aborting.
# - DLIO PR #48 fix for storage #741: drop page cache before the
# read_threads memory guard — prevents run 2..N of a 5-run
# submission from tripping the guard on reclaimable Lustre page
# cache pinned by the prior run. Gated on local_rank() == 0 (not
# MPI.node(), which is unreliable under --map-by node). Reuses
# the existing DLIO_DROP_CACHES_TIMEOUT knob.
# - DLIO PR #47 fix for storage #626 (bucket 2): shared 64-way concurrency
# ceiling across object-storage libraries — caps s3dlio /
# ObjStoreLibStorage / SimpleStreamingCheckpointing on the same
# pool so multi-adapter runs don't blow past the tuned limit
# - DLIO PR #46 fix for storage #626 (bucket 1): fork-safe
# _s3_iterable_mixin prefetch pool — reset pool state across fork()
# so DataLoader worker processes don't inherit a dead pool
# - DLIO PR #45 fix for storage #699: os.makedirs(exist_ok=True) race on
# multi-host checkpoint runs — exponential-backoff retry in
# _makedirs_race_safe(); applied to FileStorage, ObjStoreLibStorage,
# and SimpleStreamingCheckpointing
# - DLIO PR #44 fix for storage #690: checkpoint read double-prepends bucket
# for object storage — affects llama3-* checkpointing on S3/GCS/Azure
# - DLIO PR #43 fix for storage #671: env-var fallback for OpenMPI
# COMM_TYPE_SHARED collapse — prevents num_hosts = MPI-rank-count on
# Ubuntu 24.04 / kernel 6.8+ / OpenMPI 4.1.6
# - DLIO PR #42 fix for storage #667: _checkpoint LIST once on rank 0
# and broadcast the count — prevents SlowDown/503 storms and spurious
# "0 checkpoints available" aborts at scale (llama3-405b, 512 ranks
# on S3). Also fixes a latent missing-f-string on the failure Exception.
# - DLIO PR #41 fix for storage #593: write integrity for obj_store_lib —
# opt-in verification, s3dlio 0.9.106
# - DLIO PR #39 fix for storage #567: O_DIRECT path triggers on
# uri_scheme=direct so direct_fs (--o-direct) reaches s3dlio instead of
# crashing buffered open() with a direct:// URI
# - DLIO PR #38 fix for storage #538: wire DIRECT_FS for --o-direct local
# O_DIRECT I/O (merged form supersedes the pre-merge pin from 3.0.23)
# - DLIO PR #37: restrict default `pytest` to checkin suite; full suite must
# be invoked explicitly with `pytest tests/ -m full_suite`
# - DLIO PR #28 fix for storage #487: expose DLIO_DROP_CACHES_TIMEOUT,
# distinguish slow flush from sudo refusal
# - DLIO PR #27 fix for storage #466 #472: skip_listing, S3 object storage,
# epoch-2+ AU, TFRecord via s3dlio, large-scale listing OOM prevention
# - DLIO PR #25 fix for storage #455 (sampler floor division — equal per-rank shards)
# - DLIO PR #24 fix for storage #448 (per-node worker-memory budget)
# - DLIO PR #23 fix for storage #391 part 2 (sudo -n + warn-once for drop_caches)
# - DLIO PR #22 fix for storage #415 (mpi4py auto-init in spawn workers)
# - DLIO PR #21 fix for storage #391 part 1 (TorchIterableDatasetSimple gating)
dlio-benchmark = { git = "https://github.com/mlcommons/DLIO_local_changes.git", rev = "dfaa17c11a0a2f229997a3a5041b621b1fd2f8c5" }
[dependency-groups]
dev = [
"pytest>=9.0.2",
]