diff --git a/docs/index.html b/docs/index.html index d0e1c9b..812e868 100644 --- a/docs/index.html +++ b/docs/index.html @@ -128,11 +128,11 @@

OPL – Optimisation problem library

fn_building_spatial Building spatial design Problem - binary | continuous + continuous | binary >=2 2 - unknown | box + box | unknown >=2 @@ -154,7 +154,7 @@

OPL – Optimisation problem library

BSO-toolbox C++ - {'40 seconds', '1 second'} + {'1 second', '40 seconds'} https://github.com/TUe-excellent-buildings/BSO-toolbox Building Spatial Design toolbox (TU/e) @@ -214,7 +214,7 @@

OPL – Optimisation problem library

26 1 noisy - unknown | box + box | unknown >=14 noisy @@ -292,7 +292,7 @@

OPL – Optimisation problem library

fn_gasoline Gasoline direct injection engine design Problem - integer | continuous + continuous | integer 14 2 multi-fidelity @@ -333,7 +333,7 @@

OPL – Optimisation problem library

fn_invdeceptive_deceptive_rotell InverseDeceptiveTrap+RotatedEllipsoid / DeceptiveTrap+RotatedEllipsoid Problem - binary | continuous + continuous | binary >=2 2 @@ -456,7 +456,7 @@

OPL – Optimisation problem library

fn_onemax_sphere_deceptive_rotell Onemax+Sphere / DeceptiveTrap+RotatedEllipsoid Problem - binary | continuous + continuous | binary >=2 2 @@ -497,7 +497,7 @@

OPL – Optimisation problem library

fn_onemax_sphere_zeromax_sphere Onemax+Sphere / Zeromax+Sphere Problem - binary | continuous + continuous | binary >=2 2 @@ -661,7 +661,7 @@

OPL – Optimisation problem library

gen_ealain Ealain Generator - binary | continuous | integer + continuous | binary | integer >=3 [1, 10, 2, 3, 4, 5, 6, 7, 8, 9] dynamic | multi-fidelity @@ -767,11 +767,11 @@

OPL – Optimisation problem library

>=1 - IOHGNBG | GNBG-II + GNBG-II | IOHGNBG - https://github.com/IOHprofiler/IOHGNBG | https://github.com/rohitsalgotra/GNBG-II - IOHprofiler version of GNBG | Generalized Numerical Benchmark Generator version 2 + https://github.com/rohitsalgotra/GNBG-II | https://github.com/IOHprofiler/IOHGNBG + Generalized Numerical Benchmark Generator version 2 | IOHprofiler version of GNBG @@ -890,11 +890,11 @@

OPL – Optimisation problem library

>=1 - MA-BBOB (IOHexperimenter) | IOHexperimenter + IOHexperimenter | MA-BBOB (IOHexperimenter) C++/Python - https://github.com/IOHprofiler/IOHexperimenter/blob/master/example/Competitions/MA-BBOB/Example_MABBOB.ipynb | https://github.com/IOHprofiler/IOHexperimenter - Example notebook for MA-BBOB in IOHexperimenter | IOHprofiler experimenter framework + https://github.com/IOHprofiler/IOHexperimenter | https://github.com/IOHprofiler/IOHexperimenter/blob/master/example/Competitions/MA-BBOB/Example_MABBOB.ipynb + IOHprofiler experimenter framework | Example notebook for MA-BBOB in IOHexperimenter @@ -1030,7 +1030,7 @@

OPL – Optimisation problem library

gen_randoptgen RandOptGen Generator - binary | continuous | integer + continuous | binary | integer >=3 [1, 10, 2, 3, 4, 5, 6, 7, 8, 9] @@ -1194,7 +1194,7 @@

OPL – Optimisation problem library

suite_amvop AMVOP Suite - categorical | continuous | integer + continuous | integer | categorical >=3 1 @@ -1833,11 +1833,11 @@

OPL – Optimisation problem library

>=1 - IOHexperimenter | CEC2022 reference code + CEC2022 reference code | IOHexperimenter C++/Python - https://github.com/IOHprofiler/IOHexperimenter | https://github.com/P-N-Suganthan/2022-SO-BO - IOHprofiler experimenter framework | Suganthan's reference implementation + https://github.com/P-N-Suganthan/2022-SO-BO | https://github.com/IOHprofiler/IOHexperimenter + Suganthan's reference implementation | IOHprofiler experimenter framework @@ -1932,7 +1932,7 @@

OPL – Optimisation problem library

suite_cuter CUTEr Suite - binary | continuous | integer + continuous | binary | integer >=3 1 @@ -1973,11 +1973,11 @@

OPL – Optimisation problem library

suite_cutest CUTEst Suite - binary | continuous | integer + continuous | binary | integer >=3 1 - unknown | box + box | unknown >=2 @@ -2178,11 +2178,11 @@

OPL – Optimisation problem library

suite_expobench EXPObench Suite - continuous | categorical | integer + integer | categorical | continuous 30-405 1 noisy - unknown | box + box | unknown >=2 ["observational", "real-life"] @@ -2204,7 +2204,7 @@

OPL – Optimisation problem library

10-135 EXPObench Python - {'2 seconds', '80 seconds'} + {'80 seconds', '2 seconds'} https://github.com/AlgTUDelft/ExpensiveOptimBenchmark EXPensive Optimization benchmark library (wind farm layout, gas filter design, pipe shape, hyperparameter tuning, hospital simulation) @@ -2245,7 +2245,7 @@

OPL – Optimisation problem library

coco-gbea - {'34 seconds', '5 seconds'} + {'5 seconds', '34 seconds'} https://github.com/ttusar/coco-gbea Game-Benchmark for Evolutionary Algorithms (COCO fork) @@ -2325,11 +2325,11 @@

OPL – Optimisation problem library

>=1 - IOHGNBG | GNBG-II + GNBG-II | IOHGNBG - https://github.com/IOHprofiler/IOHGNBG | https://github.com/rohitsalgotra/GNBG-II - IOHprofiler version of GNBG | Generalized Numerical Benchmark Generator version 2 + https://github.com/rohitsalgotra/GNBG-II | https://github.com/IOHprofiler/IOHGNBG + Generalized Numerical Benchmark Generator version 2 | IOHprofiler version of GNBG @@ -2424,7 +2424,7 @@

OPL – Optimisation problem library

suite_l1_zdt L1-ZDT Suite - binary | continuous + continuous | binary >=2 2 @@ -2506,7 +2506,7 @@

OPL – Optimisation problem library

suite_l2_zdt L2-ZDT Suite - binary | continuous + continuous | binary >=2 2 @@ -2588,7 +2588,7 @@

OPL – Optimisation problem library

suite_l3_zdt L3-ZDT Suite - binary | continuous + continuous | binary >=2 2 @@ -2875,7 +2875,7 @@

OPL – Optimisation problem library

suite_modact MODAct Suite - continuous | integer + integer | continuous 40 [2, 3, 4, 5] @@ -2899,11 +2899,11 @@

OPL – Optimisation problem library

20 20 - pymoo | modact + modact | pymoo Python {'20ms'} - https://github.com/anyoptimization/pymoo | https://github.com/epfl-lamd/modact - Multi-objective optimization in Python | EPFL-LAMD modact package + https://github.com/epfl-lamd/modact | https://github.com/anyoptimization/pymoo + EPFL-LAMD modact package | Multi-objective optimization in Python @@ -3080,7 +3080,7 @@

OPL – Optimisation problem library

suite_rwmvop RWMVOP Suite - categorical | continuous | integer + continuous | integer | categorical >=3 1 @@ -3367,7 +3367,7 @@

OPL – Optimisation problem library

suite_zdt ZDT Suite - binary | continuous + continuous | binary >=2 2 diff --git a/docs/problems.html b/docs/problems.html index ab6e46f..7156317 100644 --- a/docs/problems.html +++ b/docs/problems.html @@ -102,11 +102,11 @@ fn_building_spatial Building spatial design Problem - binary | continuous + continuous | binary >=2 2 - unknown | box + box | unknown >=2 @@ -128,7 +128,7 @@ BSO-toolbox C++ - {'40 seconds', '1 second'} + {'1 second', '40 seconds'} https://github.com/TUe-excellent-buildings/BSO-toolbox Building Spatial Design toolbox (TU/e) @@ -188,7 +188,7 @@ 26 1 noisy - unknown | box + box | unknown >=14 noisy @@ -266,7 +266,7 @@ fn_gasoline Gasoline direct injection engine design Problem - integer | continuous + continuous | integer 14 2 multi-fidelity @@ -307,7 +307,7 @@ fn_invdeceptive_deceptive_rotell InverseDeceptiveTrap+RotatedEllipsoid / DeceptiveTrap+RotatedEllipsoid Problem - binary | continuous + continuous | binary >=2 2 @@ -430,7 +430,7 @@ fn_onemax_sphere_deceptive_rotell Onemax+Sphere / DeceptiveTrap+RotatedEllipsoid Problem - binary | continuous + continuous | binary >=2 2 @@ -471,7 +471,7 @@ fn_onemax_sphere_zeromax_sphere Onemax+Sphere / Zeromax+Sphere Problem - binary | continuous + continuous | binary >=2 2 @@ -635,7 +635,7 @@ gen_ealain Ealain Generator - binary | continuous | integer + continuous | binary | integer >=3 [1, 10, 2, 3, 4, 5, 6, 7, 8, 9] dynamic | multi-fidelity @@ -741,11 +741,11 @@ >=1 - IOHGNBG | GNBG-II + GNBG-II | IOHGNBG - https://github.com/IOHprofiler/IOHGNBG | https://github.com/rohitsalgotra/GNBG-II - IOHprofiler version of GNBG | Generalized Numerical Benchmark Generator version 2 + https://github.com/rohitsalgotra/GNBG-II | https://github.com/IOHprofiler/IOHGNBG + Generalized Numerical Benchmark Generator version 2 | IOHprofiler version of GNBG @@ -864,11 +864,11 @@ >=1 - MA-BBOB (IOHexperimenter) | IOHexperimenter + IOHexperimenter | MA-BBOB (IOHexperimenter) C++/Python - https://github.com/IOHprofiler/IOHexperimenter/blob/master/example/Competitions/MA-BBOB/Example_MABBOB.ipynb | https://github.com/IOHprofiler/IOHexperimenter - Example notebook for MA-BBOB in IOHexperimenter | IOHprofiler experimenter framework + https://github.com/IOHprofiler/IOHexperimenter | https://github.com/IOHprofiler/IOHexperimenter/blob/master/example/Competitions/MA-BBOB/Example_MABBOB.ipynb + IOHprofiler experimenter framework | Example notebook for MA-BBOB in IOHexperimenter @@ -1004,7 +1004,7 @@ gen_randoptgen RandOptGen Generator - binary | continuous | integer + continuous | binary | integer >=3 [1, 10, 2, 3, 4, 5, 6, 7, 8, 9] @@ -1168,7 +1168,7 @@ suite_amvop AMVOP Suite - categorical | continuous | integer + continuous | integer | categorical >=3 1 @@ -1807,11 +1807,11 @@ >=1 - IOHexperimenter | CEC2022 reference code + CEC2022 reference code | IOHexperimenter C++/Python - https://github.com/IOHprofiler/IOHexperimenter | https://github.com/P-N-Suganthan/2022-SO-BO - IOHprofiler experimenter framework | Suganthan's reference implementation + https://github.com/P-N-Suganthan/2022-SO-BO | https://github.com/IOHprofiler/IOHexperimenter + Suganthan's reference implementation | IOHprofiler experimenter framework @@ -1906,7 +1906,7 @@ suite_cuter CUTEr Suite - binary | continuous | integer + continuous | binary | integer >=3 1 @@ -1947,11 +1947,11 @@ suite_cutest CUTEst Suite - binary | continuous | integer + continuous | binary | integer >=3 1 - unknown | box + box | unknown >=2 @@ -2152,11 +2152,11 @@ suite_expobench EXPObench Suite - continuous | categorical | integer + integer | categorical | continuous 30-405 1 noisy - unknown | box + box | unknown >=2 ["observational", "real-life"] @@ -2178,7 +2178,7 @@ 10-135 EXPObench Python - {'2 seconds', '80 seconds'} + {'80 seconds', '2 seconds'} https://github.com/AlgTUDelft/ExpensiveOptimBenchmark EXPensive Optimization benchmark library (wind farm layout, gas filter design, pipe shape, hyperparameter tuning, hospital simulation) @@ -2219,7 +2219,7 @@ coco-gbea - {'34 seconds', '5 seconds'} + {'5 seconds', '34 seconds'} https://github.com/ttusar/coco-gbea Game-Benchmark for Evolutionary Algorithms (COCO fork) @@ -2299,11 +2299,11 @@ >=1 - IOHGNBG | GNBG-II + GNBG-II | IOHGNBG - https://github.com/IOHprofiler/IOHGNBG | https://github.com/rohitsalgotra/GNBG-II - IOHprofiler version of GNBG | Generalized Numerical Benchmark Generator version 2 + https://github.com/rohitsalgotra/GNBG-II | https://github.com/IOHprofiler/IOHGNBG + Generalized Numerical Benchmark Generator version 2 | IOHprofiler version of GNBG @@ -2398,7 +2398,7 @@ suite_l1_zdt L1-ZDT Suite - binary | continuous + continuous | binary >=2 2 @@ -2480,7 +2480,7 @@ suite_l2_zdt L2-ZDT Suite - binary | continuous + continuous | binary >=2 2 @@ -2562,7 +2562,7 @@ suite_l3_zdt L3-ZDT Suite - binary | continuous + continuous | binary >=2 2 @@ -2849,7 +2849,7 @@ suite_modact MODAct Suite - continuous | integer + integer | continuous 40 [2, 3, 4, 5] @@ -2873,11 +2873,11 @@ 20 20 - pymoo | modact + modact | pymoo Python {'20ms'} - https://github.com/anyoptimization/pymoo | https://github.com/epfl-lamd/modact - Multi-objective optimization in Python | EPFL-LAMD modact package + https://github.com/epfl-lamd/modact | https://github.com/anyoptimization/pymoo + EPFL-LAMD modact package | Multi-objective optimization in Python @@ -3054,7 +3054,7 @@ suite_rwmvop RWMVOP Suite - categorical | continuous | integer + continuous | integer | categorical >=3 1 @@ -3341,7 +3341,7 @@ suite_zdt ZDT Suite - binary | continuous + continuous | binary >=2 2 diff --git a/problems.yaml b/problems.yaml index 9ff660a..c79b63d 100644 --- a/problems.yaml +++ b/problems.yaml @@ -3595,3 +3595,4 @@ suite_zdt: max: null min: 1 type: binary + diff --git a/src/opltools/cli.py b/src/opltools/cli.py index 4a35d92..b497ffb 100644 --- a/src/opltools/cli.py +++ b/src/opltools/cli.py @@ -1,21 +1,28 @@ import sys import argparse from pydantic import ValidationError +import yaml from pydantic_yaml import parse_yaml_raw_as -from .schema import Library +from opltools.schema import Library + +UNIQUE_FIELDS = ["name"] +UNIQUE_WARNING_FIELDS = ["reference", "implementation"] def cmd_validate(args): - try: - with open(args.file) as f: - raw = f.read() - except OSError as e: - print(f"Error reading file: {e}", file=sys.stderr) - return 1 try: - parse_yaml_raw_as(Library, raw) + with open(args.file, "r") as f: + raw = f.read() + lib = parse_yaml_raw_as(Library, raw) + Library.model_validate( + lib, + context={ + "unique_error_fields": args.unique_error_field, + "unique_warning_fields": args.unique_warning_field, + }, + ) print(f"{args.file}: OK") return 0 except ValidationError as e: @@ -35,6 +42,21 @@ def main(): "validate", help="Validate a YAML file against the Library schema" ) validate_parser.add_argument("file", help="YAML file to validate") + # Add unique error fields + validate_parser.add_argument( + "--unique-error-field", + action="append", + help="Field that must be unique across all entries (can be specified multiple times)", + ) + validate_parser.add_argument( + "--unique-warning-field", + action="append", + help="Field that should be unique across all entries (can be specified multiple times)", + ) + # specify default unique fields if not provided + validate_parser.set_defaults( + unique_error_field=UNIQUE_FIELDS, unique_warning_field=UNIQUE_WARNING_FIELDS + ) args = parser.parse_args() diff --git a/src/opltools/schema.py b/src/opltools/schema.py index ea80c89..8ed0dcc 100644 --- a/src/opltools/schema.py +++ b/src/opltools/schema.py @@ -1,7 +1,15 @@ from enum import Enum from typing import Any from typing_extensions import Self -from pydantic import BaseModel, RootModel, ConfigDict, model_validator +from typing import List, Dict, Set +from pydantic import ( + BaseModel, + RootModel, + ConfigDict, + model_validator, + ValidationInfo, + field_validator, +) from .yesnosome import YesNoSome from .utils import ValueRange, union_range @@ -88,6 +96,15 @@ def __hash__(self): return hash(self.title) + hash(self.link) +def forbid_value(field: str, forbidden: str): + def validator(cls, v: str): + if v == forbidden: + raise ValueError(f"{field} cannot be '{forbidden}'") + return v + + return field_validator(field)(validator) + + class Implementation(Thing): type: OPLType = OPLType.implementation name: str @@ -97,6 +114,8 @@ class Implementation(Thing): evaluation_time: set[str] | None = None requirements: str | list[str] | None = None + _v = forbid_value("name", "template") # to prevent copy-paste errors + class ProblemLike(Thing): name: str @@ -118,6 +137,8 @@ class ProblemLike(Thing): code_examples: set[str] | None = None source: set[str] | None = None + _v = forbid_value("name", "template") # to prevent copy-paste errors + def __hash__(self): return hash((self.type, self.name)) @@ -136,6 +157,65 @@ class Generator(ProblemLike): type: OPLType = OPLType.generator +class ValidationRule: + def __init__( + self, + field_name: str, + group: List[OPLType] | None, + error_on_duplicate: bool = True, + ): + self.field_name = field_name + self.group = group + self.error_on_duplicate = error_on_duplicate + self.seen = set() + self.duplicates = set() + + def update_seen(self, entry: Thing): + if self.group is None or entry.OPLType in self.group: + value = getattr(entry, self.field_name, None) + if value is None: + return + if value in self.seen: + self.duplicates.add(value) + else: + self.seen.add(value) + + def _process_duplicates(self): + if self.duplicates: + if self.error_on_duplicate: + print( + f"::error::Duplicate values for field '{self.field_name}': {self.duplicates}" + ) + return False + else: + print( + f"::warning::Duplicate values for field '{self.field_name}': {self.duplicates}" + ) + return True + + +class Validator: + def __init__(self, duplicate_settings: List[Dict[str, Any]]): + rules = [] + for setting in duplicate_settings: + field_name = setting["field_name"] + group = setting.get("group", None) + error_on_duplicate = setting.get("error_on_duplicate", True) + rules.append(ValidationRule(field_name, group, error_on_duplicate)) + self.rules = rules + + def update_seen(self, entry: Thing): + for rule in self.rules: + rule.update_seen(entry) + + def process_duplicates(self): + all_valid = True + for rule in self.rules: + if not rule._process_duplicates(): + all_valid = False + return all_valid + + class Library(RootModel): root: dict[str, Problem | Generator | Suite | Implementation] = {} @@ -168,12 +248,33 @@ def _percolate_set(self, thing: Any, children: set | None, property: str): thing_set.update(child_set) @model_validator(mode="after") - def _validate(self) -> Self: + def _validate(self, info: ValidationInfo) -> Self: + # Check for duplicates and # First check and fixup all problems for id, thing in self.root.items(): if isinstance(thing, Problem) and thing.implementations: self._percolate_set(thing, thing.implementations, "evaluation_time") + # Then check and fixup all suites because changes from the problems need to propagate to the suites + duplicate_settings = ( + info.context.get("duplicate_settings", []) if info.context else [] + ) + validator = Validator(duplicate_settings) + + # First check and fixup all problems + for id, thing in self.root.items(): + validator.update_seen(thing) + if isinstance(thing, Problem) and thing.implementations: + self._percolate_set(thing, thing.implementations, "evaluation_time") + + if not validator.process_duplicates(): + raise ValueError( + "Duplicate values found in fields: " + + ", ".join( + rule.field_name for rule in validator.rules if rule.duplicates + ) + ) + # Then check and fixup all suites because changes from the problems need to propagate to the suites for id, thing in self.root.items(): if isinstance(thing, Suite) and thing.problems: @@ -186,7 +287,6 @@ def _validate(self) -> Self: raise ValueError( f"Suite {id} references problem with id '{problem_id}' but id is a {self.root[problem_id].type.name}." ) - self._percolate_set(thing, thing.problems, "fidelity_levels") self._percolate_set(thing, thing.problems, "variables") self._percolate_set(thing, thing.problems, "constraints") diff --git a/tests/test_library.py b/tests/test_library.py index 542bc7b..2c9be02 100644 --- a/tests/test_library.py +++ b/tests/test_library.py @@ -21,13 +21,15 @@ def test_single_problem(self): assert isinstance(lib.root["p1"], Problem) def test_multiple_things(self): - lib = Library(root={ - "p1": Problem(name="P1"), - "p2": Problem(name="P2"), - "g1": Generator(name="G1"), - "s1": Suite(name="S1", problems={"p1", "p2"}), - "impl1": Implementation(name="impl1", description="d"), - }) + lib = Library( + root={ + "p1": Problem(name="P1"), + "p2": Problem(name="P2"), + "g1": Generator(name="G1"), + "s1": Suite(name="S1", problems={"p1", "p2"}), + "impl1": Implementation(name="impl1", description="d"), + } + ) assert len(lib.root) == 5 assert isinstance(lib.root["p1"], Problem) assert isinstance(lib.root["g1"], Generator) @@ -36,63 +38,135 @@ def test_multiple_things(self): def test_suite_references_missing_problem(self): with pytest.raises(ValidationError, match="undefined id"): - Library(root={ - "s1": Suite(name="S1", problems={"does-not-exist"}), - }) + Library( + root={ + "s1": Suite(name="S1", problems={"does-not-exist"}), + } + ) def test_suite_references_non_problem(self): with pytest.raises(ValidationError, match="but id is a"): - Library(root={ - "g1": Generator(name="G1"), - "s1": Suite(name="S1", problems={"g1"}), - }) + Library( + root={ + "g1": Generator(name="G1"), + "s1": Suite(name="S1", problems={"g1"}), + } + ) def test_suite_with_no_problems_is_valid(self): lib = Library(root={"s1": Suite(name="S1")}) assert lib.root["s1"].problems is None def test_fixup_fidelity_populates_from_problems(self): - lib = Library(root={ - "p1": Problem(name="P1", fidelity_levels={1, 2}), - "p2": Problem(name="P2", fidelity_levels={2, 3}), - "s1": Suite(name="S1", problems={"p1", "p2"}), - }) + lib = Library( + root={ + "p1": Problem(name="P1", fidelity_levels={1, 2}), + "p2": Problem(name="P2", fidelity_levels={2, 3}), + "s1": Suite(name="S1", problems={"p1", "p2"}), + } + ) assert lib.root["s1"].fidelity_levels == {1, 2, 3} def test_fixup_fidelity_extends_existing(self): - lib = Library(root={ - "p1": Problem(name="P1", fidelity_levels={5}), - "s1": Suite(name="S1", problems={"p1"}, fidelity_levels={10}), - }) + lib = Library( + root={ + "p1": Problem(name="P1", fidelity_levels={5}), + "s1": Suite(name="S1", problems={"p1"}, fidelity_levels={10}), + } + ) assert lib.root["s1"].fidelity_levels == {5, 10} def test_fixup_fidelity_with_problems_without_levels(self): - lib = Library(root={ - "p1": Problem(name="P1"), - "p2": Problem(name="P2", fidelity_levels={7}), - "s1": Suite(name="S1", problems={"p1", "p2"}), - }) + lib = Library( + root={ + "p1": Problem(name="P1"), + "p2": Problem(name="P2", fidelity_levels={7}), + "s1": Suite(name="S1", problems={"p1", "p2"}), + } + ) assert lib.root["s1"].fidelity_levels == {7} def test_fixup_fidelity_all_problems_without_levels(self): - lib = Library(root={ - "p1": Problem(name="P1"), - "s1": Suite(name="S1", problems={"p1"}), - }) + lib = Library( + root={ + "p1": Problem(name="P1"), + "s1": Suite(name="S1", problems={"p1"}), + } + ) assert lib.root["s1"].fidelity_levels == set() def test_fixup_evaluation_time_percolates_from_implementation_to_suite(self): - lib = Library(root={ - "impl1": Implementation( - name="impl1", description="d", evaluation_time={"fast"} - ), - "impl2": Implementation( - name="impl2", description="d", evaluation_time={"8 minutes"} - ), - "p1": Problem(name="P1", implementations={"impl1"}), - "p2": Problem(name="P2", implementations={"impl2"}), - "s1": Suite(name="S1", problems={"p1", "p2"}), - }) + lib = Library( + root={ + "impl1": Implementation( + name="impl1", description="d", evaluation_time={"fast"} + ), + "impl2": Implementation( + name="impl2", description="d", evaluation_time={"8 minutes"} + ), + "p1": Problem(name="P1", implementations={"impl1"}), + "p2": Problem(name="P2", implementations={"impl2"}), + "s1": Suite(name="S1", problems={"p1", "p2"}), + } + ) assert lib.root["p1"].evaluation_time == {"fast"} assert lib.root["p2"].evaluation_time == {"8 minutes"} assert lib.root["s1"].evaluation_time == {"fast", "8 minutes"} + + +class TestLibraryValidation: + def test_invalid_root_type(self): + with pytest.raises(ValidationError): + Library(root="not a dict") + + def test_invalid_entry_type(self): + with pytest.raises(ValidationError): + Library(root={"p1": "not a problem"}) + + def test_suite_references_nonexistent_problem(self): + with pytest.raises(ValidationError): + Library(root={"s1": Suite(name="S1", problems={"p1"})}) + + def test_suite_references_non_problem(self): + with pytest.raises(ValidationError): + Library( + root={ + "g1": Generator(name="G1"), + "s1": Suite(name="S1", problems={"g1"}), + } + ) + + def test_valid_library(self): + lib = Library( + root={ + "p1": Problem(name="P1", fidelity_levels={1}), + "p2": Problem(name="P2", fidelity_levels={2}), + "s1": Suite(name="S1", problems={"p1", "p2"}), + "g1": Generator(name="G1"), + "impl1": Implementation(name="impl1", description="d"), + } + ) + assert isinstance(lib, Library) + + def test_duplicates(self): + lib = Library( + root={ + "p1": Problem(name="P1", fidelity_levels={1}), + "p2": Problem(name="P1", fidelity_levels={2}), # duplicate name + "s1": Suite(name="S1", problems={"p1", "p2"}), + "g1": Generator(name="G1"), + "impl1": Implementation(name="impl1", description="d"), + "impl2": Implementation( + name="impl1", description="d" + ), # duplicate name + } + ) + assert isinstance(lib, Library) + with pytest.raises(ValidationError): + Library.model_validate( + lib, + context={ + "unique_error_fields": ["name"], + "unique_warning_fields": [], + }, + ) diff --git a/utils/validate_yaml.py b/utils/validate_yaml.py index 79f4238..ca1b5be 100644 --- a/utils/validate_yaml.py +++ b/utils/validate_yaml.py @@ -9,7 +9,28 @@ sys.path.insert(0, str(parent)) # Now you can import normally -from yaml_to_html import default_columns as REQUIRED_FIELDS +# from yaml_to_html import default_columns as REQUIRED_FIELDS +REQUIRED_FIELDS = [ + "name", + "textual description", + "suite/generator/single", + "objectives", + "dimensionality", + "variable type", + "constraints", + "dynamic", + "noise", + "multi-fidelity", + "source (real-world/artificial)", + "reference", + "implementation", +] + + +from pydantic import ValidationError +from pydantic_yaml import parse_yaml_raw_as +from src.opltools.schema import Library + OPTIONAL_FIELDS = ["multimodal"] UNIQUE_FIELDS = ["name"] @@ -85,6 +106,77 @@ def check_fields(data: Dict) -> bool: return True +def update_seen(fields, seen, duplicates, entry): + entry_type = entry.get("type", "unknown") + for field in fields: + value = entry.get(field, None) + if value is None: + continue + seen_value = f"{entry_type}:{value}" + if seen_value in seen[field]: + duplicates[field].add(seen_value) + else: + seen[field].add(seen_value) + return seen, duplicates + + +def check_duplicates(data, warning_fields, error_fields): + # Run checks for each entry and collect duplicates + fields = set(warning_fields + error_fields) + seen = {field: set() for field in fields} + duplicates = {field: set() for field in fields} + for _, entry in data.items(): + seen, duplicates = update_seen(fields, seen, duplicates, entry) + + duplicate_warnings = { + field: list(dups) + for field, dups in duplicates.items() + if dups and field in warning_fields + } + if len(duplicate_warnings) > 0: + print(f"::warning::Duplication warnings {duplicate_warnings}") + duplicate_errors = { + field: list(dups) + for field, dups in duplicates.items() + if dups and field in error_fields + } + if len(duplicate_errors) > 0: + print(f"::error::Duplication errors {duplicate_errors}") + return len(duplicate_errors) == 0 + + +def check_parsing(filepath): + try: + with open(filepath, "r") as f: + raw = f.read() + parse_yaml_raw_as(Library, raw) + return True + except ValidationError as e: + print(f"::error::YAML parsing error: {e}") + return False + + +def validate_yaml(filepath): + status = check_parsing(filepath) + if not status: + sys.exit(1) + + status, data = read_data(filepath) + if status != 0 or data is None: + sys.exit(1) + if not check_duplicates( + data, warning_fields=UNIQUE_WARNING_FIELDS, error_fields=UNIQUE_FIELDS + ): + sys.exit(1) + if not check_format(data): + sys.exit(1) + for i, entry in enumerate(data): + if not check_fields(entry): + print(f"::error::Validation failed for entry {i+1}.") + sys.exit(1) + print("YAML syntax is valid.") + + def check_novelty(data: Dict, checked_data: List[Dict]) -> bool: for field in UNIQUE_FIELDS + UNIQUE_WARNING_FIELDS: # skip empty fields @@ -127,17 +219,6 @@ def validate_data(data: List[Dict]) -> bool: return True -def validate_yaml(filepath: str) -> None: - status, data = read_data(filepath) - if status != 0 or data is None: - sys.exit(1) - valid = validate_data(data) - if not valid: - sys.exit(1) - else: - sys.exit(0) - - if __name__ == "__main__": if len(sys.argv) < 2: print("::error::Usage: python validate_yaml.py ")