diff --git a/src/main/python/tests/scuro/test_fusion_orders.py b/src/main/python/tests/scuro/test_fusion_orders.py index 22d64bcc0bf..058c6b25f80 100644 --- a/src/main/python/tests/scuro/test_fusion_orders.py +++ b/src/main/python/tests/scuro/test_fusion_orders.py @@ -19,77 +19,80 @@ # # ------------------------------------------------------------- -import os -import shutil import unittest import numpy as np from systemds.scuro import Concatenation, RowMax, Hadamard -from systemds.scuro.modality.unimodal_modality import UnimodalModality -from systemds.scuro.representations.bert import Bert -from systemds.scuro.representations.mel_spectrogram import MelSpectrogram from systemds.scuro.representations.average import Average from tests.scuro.data_generator import ModalityRandomDataGenerator from systemds.scuro.modality.type import ModalityType class TestFusionOrders(unittest.TestCase): + """ + Tests the order properties of the fusion operators: commutativity, whether + the order of a pairwise chain matters, and whether a pairwise chain gives + the same result as the n-ary call. + """ + + # (operator, chain_order_independent, chain_equals_nary) + # + # Commutativity is not in the table. Every Fusion operator has a + # "commutative" attribute. It is False on the base class and a subclass + # can override it. The test compares the measured result with that + # attribute, so an operator whose attribute does not match its + # implementation fails here. + # + # Combining a pair is never the same as combining all three. That is + # checked for every operator, so it is not in the table either. + FUSION_PROPERTIES = [ + (Average, True, False), + (Concatenation, False, True), + (RowMax, True, True), + (Hadamard, True, True), + ] + @classmethod def setUpClass(cls): - cls.num_instances = 40 + # These properties do not depend on the input shape, so the data can + # be small. + cls.num_instances = 4 + cls.num_features = 8 cls.data_generator = ModalityRandomDataGenerator() - cls.r_1 = cls.data_generator.create1DModality(40, 100, ModalityType.AUDIO) - cls.r_2 = cls.data_generator.create1DModality(40, 100, ModalityType.TEXT) - cls.r_3 = cls.data_generator.create1DModality(40, 100, ModalityType.TEXT) - - def test_fusion_order_avg(self): - r_1_r_2 = self.r_1.combine(self.r_2, Average()) - r_2_r_1 = self.r_2.combine(self.r_1, Average()) - r_1_r_2_r_3 = r_1_r_2.combine(self.r_3, Average()) - r_2_r_1_r_3 = r_2_r_1.combine(self.r_3, Average()) - - r1_r2_r3 = self.r_1.combine([self.r_2, self.r_3], Average()) - - self.assertTrue(np.array_equal(r_1_r_2.data, r_2_r_1.data)) - self.assertTrue(np.array_equal(r_1_r_2_r_3.data, r_2_r_1_r_3.data)) - self.assertFalse(np.array_equal(r_1_r_2_r_3.data, r1_r2_r3.data)) - self.assertFalse(np.array_equal(r_1_r_2.data, r1_r2_r3.data)) - - def test_fusion_order_concat(self): - r_1_r_2 = self.r_1.combine(self.r_2, Concatenation()) - r_2_r_1 = self.r_2.combine(self.r_1, Concatenation()) - r_1_r_2_r_3 = r_1_r_2.combine(self.r_3, Concatenation()) - r_2_r_1_r_3 = r_2_r_1.combine(self.r_3, Concatenation()) - - r1_r2_r3 = self.r_1.combine([self.r_2, self.r_3], Concatenation()) - - self.assertFalse(np.array_equal(r_1_r_2.data, r_2_r_1.data)) - self.assertFalse(np.array_equal(r_1_r_2_r_3.data, r_2_r_1_r_3.data)) - self.assertFalse(np.array_equal(r_2_r_1.data, r1_r2_r3.data)) - self.assertFalse(np.array_equal(r_1_r_2.data, r1_r2_r3.data)) - - def test_fusion_order_max(self): - r_1_r_2 = self.r_1.combine(self.r_2, RowMax()) - r_2_r_1 = self.r_2.combine(self.r_1, RowMax()) - r_1_r_2_r_3 = r_1_r_2.combine(self.r_3, RowMax()) - r_2_r_1_r_3 = r_2_r_1.combine(self.r_3, RowMax()) - - r1_r2_r3 = self.r_1.combine([self.r_2, self.r_3], RowMax()) - - self.assertTrue(np.array_equal(r_1_r_2.data, r_2_r_1.data)) - self.assertTrue(np.array_equal(r_1_r_2_r_3.data, r_2_r_1_r_3.data)) - self.assertTrue(np.array_equal(r_1_r_2_r_3.data, r1_r2_r3.data)) - self.assertFalse(np.array_equal(r_1_r_2.data, r1_r2_r3.data)) - - def test_fusion_order_hadamard(self): - r_1_r_2 = self.r_1.combine(self.r_2, Hadamard()) - r_2_r_1 = self.r_2.combine(self.r_1, Hadamard()) - r_1_r_2_r_3 = r_1_r_2.combine(self.r_3, Hadamard()) - r_2_r_1_r_3 = r_2_r_1.combine(self.r_3, Hadamard()) - - r1_r2_r3 = self.r_1.combine([self.r_2, self.r_3], Hadamard()) - self.assertTrue(np.array_equal(r_1_r_2.data, r_2_r_1.data)) - self.assertTrue(np.array_equal(r_1_r_2_r_3.data, r_2_r_1_r_3.data)) - self.assertTrue(np.array_equal(r_1_r_2_r_3.data, r1_r2_r3.data)) - self.assertFalse(np.array_equal(r_1_r_2.data, r1_r2_r3.data)) + def setUp(self): + self.r_1 = self.data_generator.create1DModality( + self.num_instances, self.num_features, ModalityType.AUDIO + ) + self.r_2 = self.data_generator.create1DModality( + self.num_instances, self.num_features, ModalityType.TEXT + ) + self.r_3 = self.data_generator.create1DModality( + self.num_instances, self.num_features, ModalityType.TEXT + ) + + @staticmethod + def _equal(left, right): + return np.array_equal(np.asarray(left.data), np.asarray(right.data)) + + def test_fusion_order_properties(self): + for ( + fusion_operator, + chain_order_independent, + chain_equals_nary, + ) in self.FUSION_PROPERTIES: + with self.subTest(fusion=fusion_operator.__name__): + r_1_r_2 = self.r_1.combine(self.r_2, fusion_operator()) + r_2_r_1 = self.r_2.combine(self.r_1, fusion_operator()) + r_1_r_2_r_3 = r_1_r_2.combine(self.r_3, fusion_operator()) + r_2_r_1_r_3 = r_2_r_1.combine(self.r_3, fusion_operator()) + r1_r2_r3 = self.r_1.combine([self.r_2, self.r_3], fusion_operator()) + + self.assertEqual( + self._equal(r_1_r_2, r_2_r_1), fusion_operator().commutative + ) + self.assertEqual( + self._equal(r_1_r_2_r_3, r_2_r_1_r_3), chain_order_independent + ) + self.assertEqual(self._equal(r_1_r_2_r_3, r1_r2_r3), chain_equals_nary) + self.assertFalse(self._equal(r_1_r_2, r1_r2_r3)) diff --git a/src/main/python/tests/scuro/test_hp_tuner.py b/src/main/python/tests/scuro/test_hp_tuner.py index 8f7aa0b1284..2235ffdba32 100644 --- a/src/main/python/tests/scuro/test_hp_tuner.py +++ b/src/main/python/tests/scuro/test_hp_tuner.py @@ -19,6 +19,7 @@ # # ------------------------------------------------------------- +import sys import unittest from types import SimpleNamespace @@ -50,7 +51,29 @@ ) from systemds.scuro.modality.type import ModalityType -from systemds.scuro.drsearch.hyperparameter_tuner import HyperparameterTuner +from systemds.scuro.drsearch.hyperparameter_tuner import ( + HyperparamResult, + HyperparamResults, + HyperparameterTuner, + _apply_pushdown_trial_params, + _apply_trial_params_to_node, + _expand_aggregation_param_specs, + _is_aggregated_representation_operation, + _is_window_operation, + _materialize_node_params, + _param_values_to_spec, + _window_input_stats, +) +from systemds.scuro.drsearch.representation_dag import ( + RepresentationDAGBuilder, + RepresentationDag, + RepresentationNode, +) +from systemds.scuro.representations.aggregate import Aggregation +from systemds.scuro.representations.aggregated_representation import ( + AggregatedRepresentation, +) +from systemds.scuro.representations.window_aggregation import WindowAggregation from unittest.mock import patch @@ -247,5 +270,354 @@ def fake_evaluate( self.assertEqual([r[0]["x"] for r in results], [1, 2]) +class _NarrowingAggregation: + """An aggregation that drops window sizes larger than the input. It stands + in for operators that adapt their domain to the data they receive.""" + + def __init__(self): + self.parameters = {"window_size": [2, 4, 8]} + + def filter_parameter_domain(self, name, values, input_stats): + return [value for value in values if value <= input_stats.output_shape[0]] + + +class TestSearchSpaceConstruction(unittest.TestCase): + """Every operator declares its parameters in its own shape. These helpers + translate them into the specifications optuna samples from.""" + + def test_param_values_to_spec_maps_a_list_to_a_categorical_domain(self): + spec = _param_values_to_spec("node0-window_size", [2, 4, 8]) + self.assertEqual( + spec, + {"name": "node0-window_size", "type": "categorical", "domain": [2, 4, 8]}, + ) + + def test_param_values_to_spec_maps_an_integer_pair_to_an_integer_domain(self): + # a pair describes a range and is sorted, so the declared order does + # not matter + spec = _param_values_to_spec("node0-n_mfcc", (20, 13)) + self.assertEqual(spec["type"], "integer") + self.assertEqual(spec["domain"], (13, 20)) + + def test_param_values_to_spec_maps_a_float_pair_to_a_real_domain(self): + spec = _param_values_to_spec("node0-ratio", (0.9, 0.1)) + self.assertEqual(spec["type"], "real") + self.assertEqual(spec["domain"], (0.1, 0.9)) + + def test_param_values_to_spec_wraps_a_single_value_in_a_categorical_domain(self): + # a fixed value becomes a domain too, so every parameter reaches optuna + # in the same shape + spec = _param_values_to_spec("node0-aggregation", "mean") + self.assertEqual(spec["type"], "categorical") + self.assertEqual(spec["domain"], ["mean"]) + + def test_param_values_to_spec_accepts_any_other_iterable(self): + spec = _param_values_to_spec("node0-window_size", range(3)) + self.assertEqual(spec["type"], "categorical") + self.assertEqual(spec["domain"], [0, 1, 2]) + + def test_param_values_to_spec_returns_none_for_values_without_a_domain(self): + for param_values in (None, {"window_size": 4}, range(0)): + with self.subTest(param_values=param_values): + self.assertIsNone(_param_values_to_spec("node0-p", param_values)) + + def test_param_values_to_spec_returns_none_when_the_value_cannot_be_listed(self): + # a zero dimensional array offers __iter__ but raises on list() + self.assertIsNone(_param_values_to_spec("node0-p", np.array(5))) + + def test_window_input_stats_reports_the_window_size_as_the_input_shape(self): + stats = _window_input_stats({"window_size": 8}) + self.assertEqual(stats.num_instances, 1) + self.assertEqual(stats.output_shape, (8,)) + + def test_window_input_stats_returns_none_without_a_window_size(self): + for node_parameters in (None, {}, {"aggregation_function": Aggregation}): + with self.subTest(node_parameters=node_parameters): + self.assertIsNone(_window_input_stats(node_parameters)) + + def test_expand_aggregation_param_specs_expands_the_nested_parameters(self): + specs = _expand_aggregation_param_specs("node0", Aggregation) + # Aggregation offers two nested names, but pad_modality carries no + # domain and is skipped + self.assertEqual(len(specs), 1) + self.assertEqual( + specs[0]["name"], "node0-aggregation_function_aggregation_function" + ) + self.assertIn("mean", specs[0]["domain"]) + + def test_expand_aggregation_param_specs_narrows_the_domain_to_the_input(self): + specs = _expand_aggregation_param_specs( + "node0", _NarrowingAggregation, _window_input_stats({"window_size": 4}) + ) + self.assertEqual([spec["domain"] for spec in specs], [[2, 4]]) + + def test_expand_aggregation_param_specs_returns_nothing_for_an_instance(self): + self.assertEqual(_expand_aggregation_param_specs("node0", Aggregation()), []) + + def test_expand_aggregation_param_specs_returns_nothing_without_nested_names(self): + class _PlainAggregation: + pass + + self.assertEqual( + _expand_aggregation_param_specs("node0", _PlainAggregation), [] + ) + + def test_expand_aggregation_param_specs_returns_nothing_for_a_failing_class(self): + # nested names are looked up by class name, so a class called + # Aggregation reports them before any instance exists -- only then can + # building the instance still fail + class Aggregation: + def __init__(self): + raise ValueError("cannot be built") + + self.assertEqual(_expand_aggregation_param_specs("node0", Aggregation), []) + + +class TestApplyTrialParams(unittest.TestCase): + """A trial hands back one flat value per parameter. Putting a value back is + not always a top level assignment: a pushed down aggregation keeps its + parameters in a nested key.""" + + def _pushdown_node_parameters(self): + # pushdown_aggregation moves the parameters of an aggregation node into + # this nested key and leaves the representation parameters on top + return {"layer": "avgpool", "_pushdown_aggregation": {"aggregation": "mean"}} + + def test_apply_pushdown_trial_params_moves_nested_values_into_the_pushdown(self): + result = _apply_pushdown_trial_params( + self._pushdown_node_parameters(), + {"aggregation_function_pad_modality": False}, + ) + self.assertEqual( + result["_pushdown_aggregation"], + {"aggregation": "mean", "aggregation_function_pad_modality": False}, + ) + + def test_apply_pushdown_trial_params_renames_the_aggregation_value(self): + # AggregatedRepresentation reads aggregation_function_aggregation_function + # before the plain aggregation key, so the sampled value wins + result = _apply_pushdown_trial_params( + self._pushdown_node_parameters(), {"aggregation": "max"} + ) + self.assertEqual( + result["_pushdown_aggregation"][ + "aggregation_function_aggregation_function" + ], + "max", + ) + + def test_apply_pushdown_trial_params_keeps_other_values_at_the_top_level(self): + result = _apply_pushdown_trial_params( + self._pushdown_node_parameters(), + {"layer": "fc", "_pushdown_aggregation": "ignored"}, + ) + self.assertEqual(result["layer"], "fc") + self.assertEqual(result["_pushdown_aggregation"], {"aggregation": "mean"}) + + def test_apply_pushdown_trial_params_leaves_the_base_parameters_unchanged(self): + base_params = self._pushdown_node_parameters() + _apply_pushdown_trial_params(base_params, {"aggregation": "max", "layer": "fc"}) + self.assertEqual(base_params, self._pushdown_node_parameters()) + + def test_apply_pushdown_trial_params_removes_nested_keys_from_the_top_level(self): + # a nested key left on top would reach the representation instead of + # the aggregation + base_params = { + "aggregation_function_pad_modality": True, + "_pushdown_aggregation": {}, + } + result = _apply_pushdown_trial_params(base_params, {"n_components": 4}) + self.assertNotIn("aggregation_function_pad_modality", result) + self.assertEqual(result["n_components"], 4) + + def test_apply_trial_params_to_node_routes_a_pushdown_node_to_the_merge(self): + node = RepresentationNode( + node_id="n0", + operation=WindowAggregation, + inputs=[], + parameters=self._pushdown_node_parameters(), + ) + result = _apply_trial_params_to_node(node, {"n0-aggregation": "max"}) + self.assertEqual(result["layer"], "avgpool") + self.assertEqual( + result["_pushdown_aggregation"][ + "aggregation_function_aggregation_function" + ], + "max", + ) + + def test_apply_trial_params_to_node_mirrors_the_aggregation(self): + node = RepresentationNode( + node_id="n1", + operation=AggregatedRepresentation, + inputs=[], + parameters={ + "aggregation_function_aggregation_function": "mean", + "aggregation_function_pad_modality": True, + "target_dimensions": 8, + }, + ) + result = _apply_trial_params_to_node(node, {"n1-aggregation": "max"}) + self.assertEqual(result["aggregation_function_aggregation_function"], "max") + self.assertNotIn("aggregation_function_pad_modality", result) + self.assertEqual(result["target_dimensions"], 8) + + def test_is_window_operation_recognises_only_window_classes(self): + for operation, expected in ( + (WindowAggregation, True), + (AggregatedRepresentation, False), + ("WindowAggregation", False), + ): + with self.subTest(operation=operation): + self.assertEqual(_is_window_operation(operation), expected) + + def test_is_aggregated_representation_operation_recognises_only_its_classes(self): + for operation, expected in ( + (AggregatedRepresentation, True), + (WindowAggregation, False), + (None, False), + ): + with self.subTest(operation=operation): + self.assertEqual( + _is_aggregated_representation_operation(operation), expected + ) + + def test_operation_checks_fall_back_when_the_module_is_unavailable(self): + # both checks import their base class inside the function + checks = ( + ( + _is_window_operation, + "systemds.scuro.representations.window_aggregation", + WindowAggregation, + ), + ( + _is_aggregated_representation_operation, + "systemds.scuro.representations.aggregated_representation", + AggregatedRepresentation, + ), + ) + for check, module_name, operation in checks: + with self.subTest(check=check.__name__): + with patch.dict(sys.modules, {module_name: None}): + self.assertFalse(check(operation)) + + def test_materialize_node_params_returns_the_input_outside_a_window(self): + window_node = RepresentationNode( + node_id="n0", operation=WindowAggregation, inputs=[] + ) + leaf_node = RepresentationNode(node_id="n1", operation=None, inputs=[]) + for node, flat_params in ((leaf_node, {"window_size": 4}), (window_node, {})): + with self.subTest(operation=node.operation): + self.assertEqual( + _materialize_node_params(node, flat_params), flat_params + ) + + +class TestHyperparamResults(unittest.TestCase): + """HyperparamResults stores what the tuner found. Reading it back rebuilds + the DAG of a result with the parameters that scored best.""" + + def setUp(self): + self.task = SimpleNamespace(model=SimpleNamespace(name="TuningTask")) + self.modality = SimpleNamespace(modality_id="audio_0") + self.results = HyperparamResults([self.task], [self.modality]) + + def _dag_with_one_operation(self): + # the unimodal optimizer builds its DAGs through the same builder + builder = RepresentationDAGBuilder() + leaf_id = builder.create_leaf_node(self.modality.modality_id) + operation_id = builder.create_operation_node( + WindowAggregation, [leaf_id], {"window_size": 4} + ) + return builder.build(operation_id), operation_id + + def _result(self, dag, best_params, mm_opt=False): + return HyperparamResult( + representation_name="WindowAggregation", + best_params=best_params, + best_score=0.8, + all_results=[], + tuning_time=0.1, + modality_id=self.modality.modality_id, + task_name=self.task.model.name, + dag=dag, + mm_opt=mm_opt, + ) + + def _store(self, results): + self.results.results[self.task.model.name][self.modality.modality_id] = results + + def test_add_result_skips_a_missing_result(self): + # a representation that could not be tuned arrives as None + self.results.add_result([None]) + self.assertEqual( + self.results.results[self.task.model.name][self.modality.modality_id], [] + ) + + def test_add_result_stores_a_multimodal_result_under_its_own_key(self): + dag, _ = self._dag_with_one_operation() + self.results.setup_mm(optimize_unimodal=False) + self.results.add_result([self._result(dag, {}, mm_opt=True)]) + self.assertEqual( + len(self.results.results[self.task.model.name]["mm_results"]), 1 + ) + + def test_setup_mm_replaces_the_results_with_a_multimodal_slot(self): + self.results.setup_mm(optimize_unimodal=False) + self.assertEqual( + self.results.results, {self.task.model.name: {"mm_results": []}} + ) + + def test_setup_mm_keeps_the_results_when_the_unimodal_step_runs(self): + self.results.setup_mm(optimize_unimodal=True) + self.assertEqual( + self.results.results, + {self.task.model.name: {self.modality.modality_id: []}}, + ) + + def test_get_k_best_dags_rebuilds_the_dag_with_the_best_parameters(self): + dag, operation_id = self._dag_with_one_operation() + self._store([self._result(dag, {f"{operation_id}-window_size": 8})]) + + _, dags = self.results.get_k_best_dags(self.modality, self.task) + + operation_nodes = [node for node in dags[0].nodes if node.operation is not None] + leaf_nodes = [node for node in dags[0].nodes if node.operation is None] + self.assertEqual(operation_nodes[0].parameters["window_size"], 8) + self.assertEqual(leaf_nodes[0].modality_id, self.modality.modality_id) + self.assertEqual(dags[0].root_node_id, operation_nodes[0].node_id) + # the stored result keeps the parameters it was evaluated with + self.assertEqual(dag.get_node_by_id(operation_id).parameters["window_size"], 4) + + def test_get_k_best_dags_returns_one_dag_per_stored_result(self): + first_dag, _ = self._dag_with_one_operation() + second_dag, _ = self._dag_with_one_operation() + stored = [self._result(first_dag, {}), self._result(second_dag, {})] + self._store(stored) + + results, dags = self.results.get_k_best_dags(self.modality, self.task) + + self.assertIs(results, stored) + self.assertEqual(len(dags), 2) + + def test_get_k_best_dags_returns_nothing_without_stored_results(self): + self.assertEqual( + self.results.get_k_best_dags(self.modality, self.task), ([], []) + ) + + def test_get_k_best_results_returns_the_last_output_of_every_dag(self): + # executing a DAG returns one entry per node + dag, _ = self._dag_with_one_operation() + self._store([self._result(dag, {})]) + outputs = {"leaf": "raw modality", "operation": "representation"} + + with patch.object(RepresentationDag, "execute", return_value=outputs): + _, representations = self.results.get_k_best_results( + self.modality, self.task, "accuracy" + ) + + self.assertEqual(representations, ["representation"]) + + if __name__ == "__main__": unittest.main() diff --git a/src/main/python/tests/scuro/test_unimodal_optimizer.py b/src/main/python/tests/scuro/test_unimodal_optimizer.py index f27c721aa25..09b85fad09a 100644 --- a/src/main/python/tests/scuro/test_unimodal_optimizer.py +++ b/src/main/python/tests/scuro/test_unimodal_optimizer.py @@ -20,6 +20,10 @@ # ------------------------------------------------------------- +import os +import pickle +import shutil +import tempfile import unittest from types import SimpleNamespace @@ -30,6 +34,7 @@ from systemds.scuro.drsearch.unimodal_optimizer import ( UnimodalOptimizer, UnimodalResults, + get_dag_by_id, ) from systemds.scuro.representations.covarep_audio_features import ZeroCrossing @@ -127,16 +132,51 @@ def setUpClass(cls): TestTask("UnimodalRepresentationTask1", "Test1", cls.num_instances), ] - def test_unimodal_optimizer_for_text_modality(self): - text_data, text_md = ModalityRandomDataGenerator().create_text_data( - self.num_instances, 10 - ) - text = UnimodalModality( - TestDataLoader( - self.indices, None, ModalityType.TEXT, text_data, str, text_md + # (label, [(modality type, generator keyword arguments)]). A set with two + # entries is passed to the optimizer as one multi-modality search, because + # optimize_unimodal_representation_for_modality loops over the list. + MODALITY_SETS = [ + ("text", [(ModalityType.TEXT, {})]), + ("image", [(ModalityType.IMAGE, {})]), + ("audio", [(ModalityType.AUDIO, {})]), + ("video", [(ModalityType.VIDEO, {"num_frames": 10})]), + ( + "text+image", + [(ModalityType.TEXT, {"num_sentences": 1}), (ModalityType.IMAGE, {})], + ), + ] + + def _create_modality(self, modality_type, num_sentences=10, num_frames=1): + generator = ModalityRandomDataGenerator() + if modality_type is ModalityType.TEXT: + data, metadata = generator.create_text_data( + self.num_instances, num_sentences ) + data_type = str + elif modality_type is ModalityType.AUDIO: + data, metadata = generator.create_audio_data(self.num_instances, 3000) + data_type = np.float32 + else: + # IMAGE and VIDEO use the same generator. The number of frames is + # the difference between them. + data, metadata = generator.create_visual_modality( + self.num_instances, num_frames, 10, 10 + ) + data_type = np.float32 + + return UnimodalModality( + TestDataLoader(self.indices, None, modality_type, data, data_type, metadata) ) - self.optimize_unimodal_representation_for_modality([text]) + + def test_unimodal_optimizer_per_modality_set(self): + for label, modality_specs in self.MODALITY_SETS: + with self.subTest(modalities=label): + self.optimize_unimodal_representation_for_modality( + [ + self._create_modality(modality_type, **kwargs) + for modality_type, kwargs in modality_specs + ] + ) def test_robust_results_ignore_non_finite_scores(self): modality = SimpleNamespace(modality_id="modality") @@ -194,59 +234,6 @@ def test_bow_and_tfidf_require_dimensionality_reduction_before_task(self): task_input = dag.get_node_by_id(task_node.inputs[0]) self.assertIs(task_input.operation, MLPAveraging) - def test_unimodal_optimizer_for_image_modality(self): - image_data, image_md = ModalityRandomDataGenerator().create_visual_modality( - self.num_instances, 1, 10, 10 - ) - image = UnimodalModality( - TestDataLoader( - self.indices, None, ModalityType.IMAGE, image_data, np.float32, image_md - ) - ) - self.optimize_unimodal_representation_for_modality([image]) - - def test_unimodal_optimizer_for_multiple_modalities(self): - image_data, image_md = ModalityRandomDataGenerator().create_visual_modality( - self.num_instances, 1, 10, 10 - ) - image = UnimodalModality( - TestDataLoader( - self.indices, None, ModalityType.IMAGE, image_data, np.float32, image_md - ) - ) - text_data, text_md = ModalityRandomDataGenerator().create_text_data( - self.num_instances - ) - text = UnimodalModality( - TestDataLoader( - self.indices, None, ModalityType.TEXT, text_data, str, text_md - ) - ) - self.optimize_unimodal_representation_for_modality([text, image]) - - def test_unimodal_optimizer_for_audio_modality(self): - audio_data, audio_md = ModalityRandomDataGenerator().create_audio_data( - self.num_instances, 3000 - ) - audio = UnimodalModality( - TestDataLoader( - self.indices, None, ModalityType.AUDIO, audio_data, np.float32, audio_md - ) - ) - - self.optimize_unimodal_representation_for_modality([audio]) - - def test_unimodal_optimizer_for_video_modality(self): - video_data, video_md = ModalityRandomDataGenerator().create_visual_modality( - self.num_instances, 10, 10, 10 - ) - video = UnimodalModality( - TestDataLoader( - self.indices, None, ModalityType.VIDEO, video_data, np.float32, video_md - ) - ) - self.optimize_unimodal_representation_for_modality([video]) - # ------------------------------------------------------------------ # Every registered representation, run through the optimizer # ------------------------------------------------------------------ @@ -453,3 +440,225 @@ def optimize_unimodal_representation_for_modality(self, modalities): modalities[0], self.tasks[0], "accuracy" ) assert len(result) == 1 + + +class TestUnimodalOptimizerPersistence(unittest.TestCase): + """store_results writes the search results to disk so that a long search + can be resumed or inspected afterwards. The files are written from the + result container, so no search has to run.""" + + # TestTask stratifies its train and validation split, so it needs at least + # one instance per class in each part. + num_instances = 10 + + def setUp(self): + self.result_path = tempfile.mkdtemp() + self.addCleanup(shutil.rmtree, self.result_path) + indices = np.array(range(self.num_instances)) + data, metadata = ModalityRandomDataGenerator().create_audio_data( + self.num_instances, 200 + ) + self.modality = UnimodalModality( + TestDataLoader( + indices, None, ModalityType.AUDIO, data, np.float32, metadata + ) + ) + self.task = TestTask("PersistenceTask", "Test1", self.num_instances) + self.optimizer = UnimodalOptimizer( + [self.modality], + [self.task], + result_path=self.result_path, + enable_checkpointing=False, + ) + + def _read(self, file_name): + with open(os.path.join(self.result_path, file_name), "rb") as f: + return pickle.load(f) + + def test_store_results_writes_the_results_and_the_execution_statistics(self): + self.optimizer.store_results("results.pkl") + + self.assertEqual( + sorted(os.listdir(self.result_path)), + ["results.pkl", "results_exec_stats.pkl"], + ) + self.assertEqual( + self._read("results.pkl"), self.optimizer.operator_performance.results + ) + self.assertEqual( + set(self._read("results_exec_stats.pkl")), + { + "worker_stats", + "node_stats", + "reuse_stats", + "wall_clock_s", + "search_start_unix", + "max_num_workers", + }, + ) + + def test_store_results_names_the_file_after_the_optimizer_and_the_time(self): + self.optimizer.store_results() + + written = sorted(os.listdir(self.result_path)) + self.assertEqual(len(written), 2) + for name in written: + self.assertTrue(name.startswith("unimodal_optimizer")) + self.assertTrue(name.endswith(".pkl")) + + def test_store_results_appends_the_statistics_suffix_without_a_pkl_ending(self): + # The statistics file name is derived by replacing ".pkl". Without that + # ending the name stays unchanged, so the suffix is appended. + self.optimizer.store_results("results") + + self.assertEqual( + sorted(os.listdir(self.result_path)), + ["results", "results_exec_stats.pkl"], + ) + + def test_results_survive_a_store_and_load_round_trip(self): + modality_id = self.modality.modality_id + task_name = self.task.model.name + entry = ResultEntry( + val_score={"accuracy": 0.75}, representation_time=1.0, task_time=2.0 + ) + self.optimizer.operator_performance.results[modality_id][task_name] = [entry] + + self.optimizer.store_results("results.pkl") + self.optimizer.operator_performance.results[modality_id][task_name] = [] + self.optimizer.load_results(os.path.join(self.result_path, "results.pkl")) + + restored = self.optimizer.operator_performance.results[modality_id][task_name] + self.assertEqual(len(restored), 1) + self.assertEqual(restored[0].val_score, {"accuracy": 0.75}) + + def test_count_results_by_modality_counts_the_first_task_of_every_modality(self): + # The count drives the checkpoint progress, where every task evaluates + # the same representations, so one task stands for all of them. + counts = self.optimizer._count_results_by_modality( + {"audio": {"t1": [1, 2], "t2": [3]}, "text": {"t1": []}} + ) + + self.assertEqual(counts, {"audio": 2, "text": 0}) + + def test_resume_from_checkpoint_restores_a_stored_result_set(self): + restored = {"audio": {"t1": ["entry"]}} + with patch.object( + self.optimizer._checkpoint_manager, + "resume_from_checkpoint", + return_value=(restored, None, None), + ): + self.optimizer.resume_from_checkpoint() + + self.assertEqual(self.optimizer.operator_performance.results, restored) + + def test_resume_from_checkpoint_keeps_the_results_without_a_checkpoint(self): + current = self.optimizer.operator_performance.results + with patch.object( + self.optimizer._checkpoint_manager, + "resume_from_checkpoint", + return_value=None, + ): + self.optimizer.resume_from_checkpoint() + + self.assertIs(self.optimizer.operator_performance.results, current) + + +class TestUnimodalResultsReadout(unittest.TestCase): + """UnimodalResults holds the scores of every evaluated representation and a + cache of the data the best ones produced. get_k_best_results returns both. + The entries are written straight into the container, so no search has to + run.""" + + num_instances = 10 + + def setUp(self): + indices = np.array(range(self.num_instances)) + data, metadata = ModalityRandomDataGenerator().create_audio_data( + self.num_instances, 200 + ) + self.modality = UnimodalModality( + TestDataLoader( + indices, None, ModalityType.AUDIO, data, np.float32, metadata + ) + ) + self.task = TestTask("ReadoutTask", "Test1", self.num_instances) + self.results = UnimodalResults( + [self.modality], [self.task], k=2, metric_name="accuracy" + ) + self.modality_id = self.modality.modality_id + self.task_name = self.task.model.name + + def _entry(self, accuracy, dag=None): + return ResultEntry( + val_score={"accuracy": accuracy}, + representation_time=1.0, + task_time=1.0, + dag=dag, + ) + + def _fill(self, accuracies): + entries = [self._entry(a) for a in accuracies] + self.results.results[self.modality_id][self.task_name] = entries + return entries + + def test_get_k_best_results_returns_the_k_highest_scores(self): + self._fill([0.5, 0.9, 0.7, 0.3]) + + best, _ = self.results.get_k_best_results( + self.modality, self.task, "accuracy", cache_needed=False + ) + + self.assertEqual([entry.val_score["accuracy"] for entry in best], [0.9, 0.7]) + + def test_get_k_best_results_uses_the_cache_when_it_holds_entries(self): + self._fill([0.9, 0.7]) + self.results.cache[self.modality_id][self.task_name] = ["first", "second"] + + _, cache = self.results.get_k_best_results(self.modality, self.task, "accuracy") + + self.assertEqual(cache, ["first", "second"]) + + def test_get_k_best_results_executes_the_dag_when_the_cache_is_empty(self): + # load_results restores the scores but not the cached data, so the + # cache is empty after reading a file. The dag of every returned entry + # is executed to rebuild it. + executed = [] + + class RecordingDag: + def __init__(self, label): + self.label = label + + def execute(self, modalities): + executed.append(self.label) + return self.label + + entries = [ + self._entry(0.9, dag=RecordingDag("best")), + self._entry(0.7, dag=RecordingDag("second")), + ] + self.results.results[self.modality_id][self.task_name] = entries + self.results.cache[self.modality_id][self.task_name] = [] + + _, cache = self.results.get_k_best_results(self.modality, self.task, "accuracy") + + self.assertEqual(executed, ["best", "second"]) + self.assertEqual(cache, ["best", "second"]) + + def test_get_k_best_results_returns_nothing_without_entries(self): + best, cache = self.results.get_k_best_results( + self.modality, self.task, "accuracy", cache_needed=False + ) + + self.assertEqual(best, []) + self.assertEqual(cache, []) + + def test_get_dag_by_id_finds_the_matching_dag(self): + dags = [SimpleNamespace(dag_id=1), SimpleNamespace(dag_id=7)] + + self.assertIs(get_dag_by_id(dags, 7), dags[1]) + + def test_get_dag_by_id_returns_none_for_an_unknown_id(self): + dags = [SimpleNamespace(dag_id=1)] + + self.assertIsNone(get_dag_by_id(dags, 99)) diff --git a/src/main/python/tests/scuro/test_unimodal_representations.py b/src/main/python/tests/scuro/test_unimodal_representations.py index 27e09d48711..3816b189f3f 100644 --- a/src/main/python/tests/scuro/test_unimodal_representations.py +++ b/src/main/python/tests/scuro/test_unimodal_representations.py @@ -205,17 +205,7 @@ def test_audio_representations(self): RMSE(), Pitch(), ] - audio_data, audio_md = ModalityRandomDataGenerator().create_audio_data( - self.num_instances, 200 - ) - - audio = UnimodalModality( - TestDataLoader( - self.indices, None, ModalityType.AUDIO, audio_data, np.float32, audio_md - ) - ) - - audio.extract_raw_data() + audio = self._create_audio_modality(signal_length=200) original_data = copy.deepcopy(audio.data) for representation in audio_representations: diff --git a/src/main/python/tests/scuro/test_window_operations.py b/src/main/python/tests/scuro/test_window_operations.py index c6a258fb465..00176c288b3 100644 --- a/src/main/python/tests/scuro/test_window_operations.py +++ b/src/main/python/tests/scuro/test_window_operations.py @@ -111,67 +111,60 @@ def test_dynamic_window(self): for i in range(0, self.num_instances): assert len(aggregated_window.data[i]) == num_windows - def test_window_aggregation_on_audio_representations(self): + def test_window_aggregation_on_1d_modalities(self): + # create1DModality returns the same shape and dtype for all three + # modality types. window_aggregation looks at the data layout and not + # at the modality type, so the result should be the same for all of + # them. window_size = 10 - self.run_window_aggregation_for_modality(ModalityType.AUDIO, window_size) - def test_window_operations_on_video_representations(self): - window_size = 10 - self.run_window_aggregation_for_modality(ModalityType.VIDEO, window_size) - - def test_window_operations_on_text_representations(self): - window_size = 10 - - self.run_window_aggregation_for_modality(ModalityType.TEXT, window_size) - - def run_window_aggregation_for_modality(self, modality_type, window_size): - r = self.data_generator.create1DModality(self.num_instances, 200, modality_type) - for aggregation in self.aggregations: - windowed_modality = r.window_aggregation(window_size, aggregation) - - self.verify_window_operation(aggregation, r, windowed_modality, window_size) - - def test_window_aggregation_on_3d_modality(self): - data, _ = self.data_generator.create_3d_modality( - self.num_instances, (100, 8, 8) - ) - embedding_modality = TransformedModality( - self.data_generator, "test_transformation" - ) - embedding_modality.data = data - embedding_modality.stats = RepresentationStats(self.num_instances, (100, 8, 8)) - num_windows = 10 - - for window_operator in [ - StaticWindow(num_windows=num_windows), - DynamicWindow(num_windows=num_windows), - WindowAggregation(window_size=10), + for modality_type in [ + ModalityType.AUDIO, + ModalityType.VIDEO, + ModalityType.TEXT, ]: - stats = window_operator.get_output_stats(embedding_modality.stats) - assert stats.num_instances == self.num_instances - assert stats.output_shape == (num_windows, 8, 8) - - windowed_modality = embedding_modality.context(window_operator) + r = self.data_generator.create1DModality( + self.num_instances, 200, modality_type + ) + for aggregation in self.aggregations: + with self.subTest(modality=modality_type.name, aggregation=aggregation): + windowed_modality = r.window_aggregation(window_size, aggregation) + self.verify_window_operation( + aggregation, r, windowed_modality, window_size + ) - def test_window_aggregation_on_2d_modality(self): - data, _ = self.data_generator.create_2d_modality(self.num_instances, (100, 8)) - embedding_modality = TransformedModality( - self.data_generator, "test_transformation" - ) - embedding_modality.data = data - embedding_modality.stats = RepresentationStats(self.num_instances, (100, 8)) + def test_window_aggregation_on_nd_modality(self): + # Window aggregation only changes the first (time) axis and keeps the + # feature axes as they are. The expected shape is therefore + # (num_windows,) + dims[1:] for any number of dimensions. num_windows = 10 - for window_operator in [ - StaticWindow(num_windows=num_windows), - DynamicWindow(num_windows=num_windows), - WindowAggregation(window_size=10), - ]: - stats = window_operator.get_output_stats(embedding_modality.stats) - assert stats.num_instances == self.num_instances - assert stats.output_shape == (num_windows, 8) - - windowed_modality = embedding_modality.context(window_operator) + for dims in [(100, 8, 8), (100, 8)]: + if len(dims) == 3: + data, _ = self.data_generator.create_3d_modality( + self.num_instances, dims + ) + else: + data, _ = self.data_generator.create_2d_modality( + self.num_instances, dims + ) + embedding_modality = TransformedModality( + self.data_generator, "test_transformation" + ) + embedding_modality.data = data + embedding_modality.stats = RepresentationStats(self.num_instances, dims) + + for window_operator in [ + StaticWindow(num_windows=num_windows), + DynamicWindow(num_windows=num_windows), + WindowAggregation(window_size=10), + ]: + with self.subTest(dims=dims, operator=type(window_operator).__name__): + stats = window_operator.get_output_stats(embedding_modality.stats) + self.assertEqual(stats.num_instances, self.num_instances) + self.assertEqual(stats.output_shape, (num_windows,) + dims[1:]) + + embedding_modality.context(window_operator) def _timeseries_modality(self, signal_length=100): return self.data_generator.create1DModality(