Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
186 changes: 186 additions & 0 deletions flutter/unit_test/benchmark/performance_result_validity_test.dart
Original file line number Diff line number Diff line change
@@ -0,0 +1,186 @@
import 'package:flutter_test/flutter_test.dart';

import 'package:mlperfbench/app_constants.dart';
import 'package:mlperfbench/backend/loadgen_info.dart';
import 'package:mlperfbench/benchmark/performance_result_validity.dart';
import 'package:mlperfbench/data/generation_helpers/sample_generator.dart';
import 'package:mlperfbench/data/results/benchmark_result.dart';
import 'package:mlperfbench/ui/app_styles.dart';

LoadgenInfo _loadgenInfo({
required bool minDurationMet,
required bool minQueryMet,
required bool earlyStoppingMet,
}) => LoadgenInfo(
queryCount: 1,
latencyMean: 0.1,
latency90: 0.1,
latencyFirstTokenMean: 0.1,
latencyFirstToken90: 0.1,
tokenThroughput: 1,
isMinDurationMet: minDurationMet,
isMinQueryMet: minQueryMet,
isEarlyStoppingMet: earlyStoppingMet,
isTokenBased: false,
isResultValid: true,
);

BenchmarkExportResult _exportResult({
required String benchmarkId,
required LoadgenInfo? loadgenInfo,
bool hasPerformanceRun = true,
}) {
final sample = SampleGenerator();
final run = sample.runResult;
final base = sample.exportResult;
return BenchmarkExportResult(
benchmarkId: benchmarkId,
benchmarkName: base.benchmarkName,
loadgenScenario: base.loadgenScenario,
backendSettings: base.backendSettings,
backendInfo: base.backendInfo,
performanceRun: hasPerformanceRun
? BenchmarkRunResult(
throughput: run.throughput,
accuracy: run.accuracy,
accuracy2: run.accuracy2,
dataset: run.dataset,
measuredDuration: run.measuredDuration,
measuredSamples: run.measuredSamples,
startDatetime: run.startDatetime,
loadgenInfo: loadgenInfo,
)
: null,
accuracyRun: base.accuracyRun,
minDuration: base.minDuration,
minSamples: base.minSamples,
);
}

PerformanceResultValidityEnum _validity({
required String benchmarkId,
required bool minDurationMet,
required bool minQueryMet,
required bool earlyStoppingMet,
}) {
return PerformanceResultValidityEnum.forBenchmarkExportResult(
benchmarkExportResult: _exportResult(
benchmarkId: benchmarkId,
loadgenInfo: _loadgenInfo(
minDurationMet: minDurationMet,
minQueryMet: minQueryMet,
earlyStoppingMet: earlyStoppingMet,
),
),
);
}

void main() {
const valid = PerformanceResultValidityEnum.valid;
const semivalid = PerformanceResultValidityEnum.semivalid;
const invalid = PerformanceResultValidityEnum.invalid;

group('PerformanceResultValidityEnum.forBenchmarkExportResult', () {
test('missing result is invalid', () {
expect(
PerformanceResultValidityEnum.forBenchmarkExportResult(
benchmarkExportResult: null,
),
invalid,
);
});
test('missing performance run is invalid', () {
final result = _exportResult(
benchmarkId: BenchmarkId.imageClassificationV2,
loadgenInfo: null,
hasPerformanceRun: false,
);
expect(
PerformanceResultValidityEnum.forBenchmarkExportResult(
benchmarkExportResult: result,
),
invalid,
);
});
test('missing loadgen info is invalid', () {
final result = _exportResult(
benchmarkId: BenchmarkId.imageClassificationV2,
loadgenInfo: null,
);
expect(
PerformanceResultValidityEnum.forBenchmarkExportResult(
benchmarkExportResult: result,
),
invalid,
);
});

// (minDurationMet, minQueryMet, earlyStoppingMet) -> expected validity
final regularCases = <(bool, bool, bool, PerformanceResultValidityEnum)>[
(true, true, true, valid),
(true, false, true, semivalid),
(true, true, false, invalid),
(true, false, false, invalid),
(false, true, true, invalid),
(false, false, true, invalid),
(false, true, false, invalid),
(false, false, false, invalid),
];
for (final (duration, query, early, expected) in regularCases) {
test(
'duration=$duration query=$query early=$early is ${expected.name}',
() {
expect(
_validity(
benchmarkId: BenchmarkId.imageClassificationV2,
minDurationMet: duration,
minQueryMet: query,
earlyStoppingMet: early,
),
expected,
);
},
);
}

group('stable diffusion ignores the early stopping condition', () {
final cases = <(bool, bool, bool, PerformanceResultValidityEnum)>[
(true, true, false, valid),
(true, false, false, semivalid),
(false, true, false, invalid),
(false, false, false, invalid),
];
for (final (duration, query, early, expected) in cases) {
test(
'duration=$duration query=$query early=$early is ${expected.name}',
() {
expect(
_validity(
benchmarkId: BenchmarkId.stableDiffusion,
minDurationMet: duration,
minQueryMet: query,
earlyStoppingMet: early,
),
expected,
);
},
);
}
});
});

group('PerformanceResultValidityExtension.color', () {
test('maps each validity to its app color', () {
expect(valid.color, AppColors.resultValidText);
expect(semivalid.color, AppColors.resultSemiValidText);
expect(invalid.color, AppColors.resultInvalidText);
});
test('colors are distinct', () {
final colors = PerformanceResultValidityEnum.values.map((e) => e.color);
expect(
colors.toSet().length,
PerformanceResultValidityEnum.values.length,
);
});
});
}
31 changes: 31 additions & 0 deletions flutter/unit_test/board_decoder_test.dart
Original file line number Diff line number Diff line change
@@ -0,0 +1,31 @@
import 'package:flutter_test/flutter_test.dart';

import 'package:mlperfbench/board_decoder.dart';

void main() {
TestWidgetsFlutterBinding.ensureInitialized();

group('BoardDecoder', () {
test('has no boards before init', () {
expect(BoardDecoder().boards, isEmpty);
});

test('init loads the bundled board database', () async {
final decoder = BoardDecoder();
await decoder.init();

expect(decoder.boards, isNotEmpty);
expect(decoder.boards.keys, everyElement(isNotEmpty));
expect(decoder.boards.values, everyElement(isNotEmpty));
});

test('init is idempotent', () async {
final decoder = BoardDecoder();
await decoder.init();
final firstLoad = Map<String, String>.of(decoder.boards);
await decoder.init();

expect(decoder.boards, firstLoad);
});
});
}
92 changes: 92 additions & 0 deletions flutter/unit_test/resources/utils_test.dart
Original file line number Diff line number Diff line change
@@ -0,0 +1,92 @@
import 'package:flutter_test/flutter_test.dart';

import 'package:mlperfbench/resources/resource.dart';
import 'package:mlperfbench/resources/utils.dart';

Resource _resource(String path) =>
Resource(type: ResourceTypeEnum.model, path: path, md5Checksum: '');

void main() {
group('isInternetResource', () {
test('accepts http and https', () {
expect(isInternetResource('http://example.com/model.tflite'), isTrue);
expect(isInternetResource('https://example.com/model.tflite'), isTrue);
});
test('rejects other schemes and local paths', () {
expect(isInternetResource('ftp://example.com/model.tflite'), isFalse);
expect(isInternetResource('asset://local/model.tflite'), isFalse);

Check failure on line 17 in flutter/unit_test/resources/utils_test.dart

View check run for this annotation

SonarQubeCloud / SonarCloud Code Analysis

Define a constant instead of duplicating this literal "asset://local/model.tflite" 3 times.

See more on https://sonarcloud.io/project/issues?id=mlcommons_mobile_app_open&issues=AaEROsVUK7f85vLbOMTR&open=AaEROsVUK7f85vLbOMTR&pullRequest=1187
expect(isInternetResource('/data/local/model.tflite'), isFalse);
expect(isInternetResource(''), isFalse);
});
});

group('asset uri', () {
test('isAsset', () {
expect(isAsset('asset://local/model.tflite'), isTrue);
expect(isAsset('https://example.com/model.tflite'), isFalse);
expect(isAsset('local/model.tflite'), isFalse);
});
test('stripAssetPrefix', () {
expect(
stripAssetPrefix('asset://local/model.tflite'),
'local/model.tflite',
);
expect(stripAssetPrefix('asset://'), '');
});
});

group('filterInternetResources', () {
test('keeps only internet resources, in order', () {
final resources = [
_resource('https://example.com/a.tflite'),
_resource('asset://local/b.tflite'),
_resource('/data/c.tflite'),
_resource('http://example.com/d.tflite'),
];
expect(filterInternetResources(resources), [
'https://example.com/a.tflite',
'http://example.com/d.tflite',
]);
});
test('returns empty list when there is nothing to download', () {
expect(filterInternetResources([]), isEmpty);
expect(filterInternetResources([_resource('asset://local/b')]), isEmpty);
});
});

group('jsonToStringIndented', () {
test('indents with two spaces', () {
final json = {
'name': 'mlperf',
'values': [1, 2],
};
expect(
jsonToStringIndented(json),
'{\n'
' "name": "mlperf",\n'
' "values": [\n'
' 1,\n'
' 2\n'
' ]\n'
'}',
);
});
});

group('lerpRange', () {
test('returns start and end at the bounds of the value range', () {
expect(lerpRange(0, 100, 10, 20, 10), 0);
expect(lerpRange(0, 100, 10, 20, 20), 100);
});
test('interpolates inside the value range', () {
expect(lerpRange(0, 100, 10, 20, 15), 50);
expect(lerpRange(200, 100, 0, 4, 1), 175);
});
test('extrapolates outside the value range', () {
expect(lerpRange(0, 100, 10, 20, 25), 150);
});
test('returns null when both ends are null', () {
expect(lerpRange(null, null, 0, 1, 0.5), isNull);
});
});
}
9 changes: 8 additions & 1 deletion tools/scanner/scan.mk
Original file line number Diff line number Diff line change
Expand Up @@ -51,8 +51,15 @@ scanner/build-app:
build-wrapper-linux-x86-64 --out-dir "${SONAR_OUT_DIR}" \
make flutter/android

# The Dart analyzer needs resolved dependencies and generated sources (protos,
# l10n, *.g.dart, ...), otherwise every import is reported as unresolved.
# https://docs.sonarsource.com/sonarqube-cloud/analyzing-source-code/languages/dart
.PHONY: scanner/prepare-dart
scanner/prepare-dart:
make flutter/prepare

.PHONY: scanner/scan
scanner/scan: scanner/build-app
scanner/scan: scanner/build-app scanner/prepare-dart
sonar-scanner \
-Dsonar.organization=mlcommons \
-Dsonar.projectKey=mlcommons_${REPO_NAME} \
Expand Down
Loading