diff --git a/README.md b/README.md index 27176d7..bb6464c 100644 --- a/README.md +++ b/README.md @@ -156,6 +156,43 @@ to a JSON file, a JSON string, a pre-instantiated config object, or `None` (see [How to implement new metrics](#how-to-implement-new-metrics)). If no `writer_config_path` is given, results are printed to the console. +### Coverage-gap diversity from MUPs + +`diversity_coverageGap` calculates the exact finite-domain DNF union induced +by a supplied set of Maximal Uncovered Patterns (MUPs). The paper-defined +coverage gap and exact pattern counts are included in `DQexplanation`. +`DQvalue` is the complementary coverage-space score (`1 - coverage_gap`) so +that it follows the METIS convention that higher scores indicate better data +quality. + +```python +from metis.metric.diversity.diversity_coverageGap_config import ( + diversity_coverageGap_config, +) + +config = diversity_coverageGap_config( + mups_path="path/to/mups_bluenile.csv_mincov_19000.txt", + attributes=[ + "shape", "color", "cut", "clarity", + "polish", "symmetry", "florescence", + ], + mincov=19000, +) + +orchestrator.assess( + metrics=["diversity_coverageGap"], + metric_configs=[config], +) +``` + +MUP fields are mapped positionally to `attributes`. Use the same attribute +order as during MUP discovery. The wildcard defaults to `x`; trailing fields +are interpreted exactly as in FLAPS: the final value is parsed as the MUP's +actual coverage and validated against `mincov`. The coverage value verifies the +frontier but does not change the geometric DNF volume. In the GUI, select +**Diversity: Coverage Gap** and upload the MUP file directly in its metric +configuration. + ### Data loader configs Datasets are described by small JSON configs (see `data/*.json`). File paths @@ -214,6 +251,7 @@ Writer config details are also covered in | Consistency | `consistency_ruleBasedHinrichs` | Rule-based consistency score after Hinrichs (attribute and tuple rules) | | Consistency | `consistency_ruleBasedPipino` | Rule-based consistency score after Pipino (boolean rules) | | Correctness | `correctness_heinrich` | Cell-wise correctness against a reference dataset after Heinrich | +| Diversity | `diversity_coverageGap` | Exact MUP-induced DNF coverage space; details include the coverage gap | | Minimality | `minimality_duplicateCount` | Duplicate rows in the dataset | | Timeliness | `timeliness_heinrich` | Decay-based timeliness of date columns after Heinrich | | Validity | `validity_outOfVocabulary` | Share of values outside a known vocabulary | diff --git a/docs/GUI.md b/docs/GUI.md index 1754d03..d0a14d7 100644 --- a/docs/GUI.md +++ b/docs/GUI.md @@ -75,7 +75,7 @@ metric is selected and no blockers remain. availability warnings. Metrics whose native dependencies are missing (for example FAHES for `completeness_nullAndDMVRatio`) are disabled with a warning. -- Metrics are configured inline through one of three editors, chosen by the +- Metrics are configured inline through the appropriate editor, chosen by the metric's metadata (see the config conventions in the [README](../README.md#config-conventions)): - a form editor for plain dataclass configs @@ -83,6 +83,9 @@ metric is selected and no blockers remain. - an inline rule editor for functional dependencies (`consistency_countFDViolations`) - `timeliness_heinrich` gets a dedicated per-column editor + - `diversity_coverageGap` gets a dedicated MUP-file uploader and positional + dataset-attribute mapping; its `mincov` is inferred from filenames such as + `*_mincov_19000.txt` when available - Select all and deselect buttons exist per dimension. The page lists blockers (missing required configs, missing reference dataset) before letting you continue. diff --git a/gui/ui/components/config_editors/mups_editor.py b/gui/ui/components/config_editors/mups_editor.py new file mode 100644 index 0000000..25d1657 --- /dev/null +++ b/gui/ui/components/config_editors/mups_editor.py @@ -0,0 +1,174 @@ +"""Upload and configure a FLAPS MUP file for coverage-gap assessment.""" + +from __future__ import annotations + +import csv +import io +import re + +import streamlit as st + +_MINCOV_RE = re.compile(r"mincov[_=-]?(\d+)", re.IGNORECASE) +_IDENTIFIER_NAMES = {"id", "row_id", "rowid", "index"} + + +def render(config_class, key_prefix: str, df_columns: list[str]): + """Render the MUP upload and positional attribute mapping controls.""" + st.caption( + "Upload the MUP output for this dataset, then select the dataset " + "attributes used during MUP discovery. Attributes are interpreted in " + "dataset-column order. The final field of each row is read as that " + "MUP's actual coverage." + ) + + delimiter = st.text_input( + "MUP delimiter", + value=",", + max_chars=1, + key=f"{key_prefix}__delimiter", + ) + wildcard = st.text_input( + "Wildcard token", + value="x", + key=f"{key_prefix}__wildcard", + help="Token used by FLAPS for an unspecified attribute.", + ) + uploaded = st.file_uploader( + "MUP file", + type=["txt", "csv"], + key=f"{key_prefix}__mups_upload", + help="FLAPS rows contain the positional pattern followed by its actual coverage.", + ) + + if uploaded is None: + st.caption("Upload a MUP file to complete this metric configuration.") + return None + + raw = uploaded.getvalue() + try: + content = raw.decode("utf-8-sig") + except UnicodeDecodeError: + content = raw.decode("latin-1") + + if len(delimiter) != 1: + st.error("The MUP delimiter must be exactly one character.") + return None + + first_fields = _first_mup_fields(content, delimiter) + file_id = f"{uploaded.name}::{uploaded.size}" + file_state_key = f"{key_prefix}__mups_file_id" + attributes_key = f"{key_prefix}__attributes" + mincov_key = f"{key_prefix}__mincov" + if st.session_state.get(file_state_key) != file_id: + st.session_state[file_state_key] = file_id + st.session_state[attributes_key] = _default_attributes( + df_columns, len(first_fields) + ) + inferred_mincov = _infer_mincov(uploaded.name) + st.session_state[mincov_key] = ( + str(inferred_mincov) if inferred_mincov is not None else "" + ) + + selected = st.multiselect( + "Diversity attributes", + options=df_columns, + key=attributes_key, + help=( + "Select exactly the attributes used to discover these MUPs. Their " + "dataset order must match the positional fields in the MUP file." + ), + ) + attributes = [column for column in df_columns if column in set(selected)] + + mincov_raw = st.text_input( + "Minimum coverage threshold (mincov, optional)", + key=mincov_key, + placeholder="e.g. 19000", + help=( + "Checks that every MUP's final coverage value is below the " + "threshold. The supplied MUP frontier determines the DNF count." + ), + ) + + if not attributes: + st.caption("Select at least one diversity attribute.") + return None + if first_fields and len(first_fields) < len(attributes) + 1: + st.error( + f"The first MUP row has {len(first_fields)} fields, but " + f"{len(attributes)} pattern fields plus one final coverage field " + "are required." + ) + return None + + intermediate = len(first_fields) - len(attributes) - 1 if first_fields else 0 + mapping = ", ".join( + f"{index + 1}: `{attribute}`" for index, attribute in enumerate(attributes) + ) + st.caption(f"Positional mapping — {mapping}") + if not first_fields: + st.warning( + "The uploaded file contains no MUP rows. If this is intentional, " + "the coverage gap is 0 and the coverage-space score is 1." + ) + else: + st.caption( + "The final field of every MUP row is parsed as its actual coverage; " + "it validates the frontier but is not a DNF dimension." + ) + if intermediate > 0: + st.caption( + f"The {intermediate} field{'s' if intermediate != 1 else ''} between " + "the pattern and final coverage will be ignored as metadata." + ) + + mincov = None + if mincov_raw.strip(): + try: + mincov = int(mincov_raw) + except ValueError: + st.error("mincov must be a positive integer.") + return None + + try: + config = config_class( + mups_content=content, + mups_filename=uploaded.name, + attributes=attributes, + mincov=mincov, + wildcard=wildcard, + delimiter=delimiter, + ) + config.validate() + return config + except (TypeError, ValueError) as exc: + st.error(f"Config error: {exc}") + return None + + +def _first_mup_fields(content: str, delimiter: str) -> list[str]: + for fields in csv.reader(io.StringIO(content), delimiter=delimiter): + if not fields or all(not field.strip() for field in fields): + continue + if fields[0].lstrip().startswith("#"): + continue + return fields + return [] + + +def _default_attributes(df_columns: list[str], mup_field_count: int) -> list[str]: + non_identifiers = [ + column for column in df_columns if column.lower() not in _IDENTIFIER_NAMES + ] + # FLAPS output commonly appends one coverage value after the pattern. + likely_pattern_width = max(1, mup_field_count - 1) + if len(non_identifiers) == likely_pattern_width: + return non_identifiers + if len(df_columns) == likely_pattern_width: + return list(df_columns) + return non_identifiers or list(df_columns) + + +def _infer_mincov(filename: str) -> int | None: + match = _MINCOV_RE.search(filename) + return int(match.group(1)) if match else None diff --git a/gui/ui/icons.py b/gui/ui/icons.py index b028399..b52c3f2 100644 --- a/gui/ui/icons.py +++ b/gui/ui/icons.py @@ -7,6 +7,7 @@ "Completeness": ":material/water_drop:", "Consistency": ":material/link:", "Correctness": ":material/check_circle:", + "Diversity": ":material/diversity_3:", "Minimality": ":material/compress:", "Timeliness": ":material/schedule:", "Validity": ":material/fact_check:", diff --git a/gui/ui/pages/metrics_page.py b/gui/ui/pages/metrics_page.py index 7a3091c..b542e8b 100644 --- a/gui/ui/pages/metrics_page.py +++ b/gui/ui/pages/metrics_page.py @@ -17,6 +17,7 @@ from metis.utils.dq_granularity import DQGranularity from ui.components.config_editors import ( callable_editor, + mups_editor, simple_editor, timeliness_editor, ) @@ -718,6 +719,16 @@ def _render_inline_config(info: MetricInfo, df) -> None: AppState.set_metric_config(info.name, cfg) return + if info.name == "diversity_coverageGap": + cfg = mups_editor.render( + info.config_class, + key_prefix=info.name, + df_columns=list(df.columns), + ) + if cfg is not None: + AppState.set_metric_config(info.name, cfg) + return + if info.config_class: # If cell-level output isn't recommended for this metric, pre-select # column-axis aggregation so the user doesn't have to know that the diff --git a/metis/metric/__init__.py b/metis/metric/__init__.py index 23f914c..2e3490a 100644 --- a/metis/metric/__init__.py +++ b/metis/metric/__init__.py @@ -8,6 +8,7 @@ from .consistency.consistency_ruleBasedHinrichs import consistency_ruleBasedHinrichs from .consistency.consistency_ruleBasedPipino import consistency_ruleBasedPipino from .correctness.correctness_heinrich import correctness_heinrich +from .diversity.diversity_coverageGap import diversity_coverageGap from .metric import Metric from .minimality.minimality_duplicateCount import minimality_duplicateCount from .minimality.minimality_clustering import minimality_clustering diff --git a/metis/metric/diversity/__init__.py b/metis/metric/diversity/__init__.py new file mode 100644 index 0000000..f81d5cc --- /dev/null +++ b/metis/metric/diversity/__init__.py @@ -0,0 +1,5 @@ +"""Coverage-based diversity metrics.""" + +from .diversity_coverageGap import diversity_coverageGap + +__all__ = ["diversity_coverageGap"] diff --git a/metis/metric/diversity/coverage_space_counter.py b/metis/metric/diversity/coverage_space_counter.py new file mode 100644 index 0000000..629259e --- /dev/null +++ b/metis/metric/diversity/coverage_space_counter.py @@ -0,0 +1,150 @@ +"""Exact finite-domain DNF counting for MUP-induced pattern regions.""" + +from __future__ import annotations + +from dataclasses import dataclass +from functools import lru_cache +from math import prod +from typing import Iterable + +Literal = tuple[int, int] +Term = tuple[Literal, ...] +Terms = tuple[Term, ...] + + +@dataclass(frozen=True) +class CoverageSpaceScore: + """Exact size and normalized score of an uncovered pattern space.""" + + uncovered_patterns: int + total_patterns: int + effective_term_count: int + + @property + def coverage_gap(self) -> float: + return self.uncovered_patterns / self.total_patterns + + @property + def coverage_space(self) -> float: + return 1.0 - self.coverage_gap + + +class CoverageSpaceCounter: + """Count the union of MUP specialization regions without materializing it. + + Each term is a conjunction of ``(attribute_index, value_index)`` literals. + Terms are combined disjunctively. Every attribute has its concrete domain + values plus one wildcard state, matching the pattern lattice definition in + FLAPS. + """ + + def __init__(self, domain_sizes: Iterable[int]) -> None: + self.domain_sizes = tuple(int(size) for size in domain_sizes) + if not self.domain_sizes: + raise ValueError("At least one diversity attribute is required.") + if any(size <= 0 for size in self.domain_sizes): + raise ValueError("Every diversity attribute must have a non-empty domain.") + + remaining = [1] * (len(self.domain_sizes) + 1) + for index in range(len(self.domain_sizes) - 1, -1, -1): + remaining[index] = remaining[index + 1] * (self.domain_sizes[index] + 1) + self._remaining_products = tuple(remaining) + + def calculate(self, raw_terms: Iterable[Iterable[Literal]]) -> CoverageSpaceScore: + """Return the exact uncovered and total pattern-space sizes.""" + terms = self._normalize_terms(raw_terms) + + @lru_cache(maxsize=None) + def count(active_terms: Terms, next_attribute: int) -> int: + if not active_terms: + return 0 + if not active_terms[0]: + return self._remaining_products[next_attribute] + if next_attribute >= len(self.domain_sizes): + return 0 + + branching_attribute = min( + attribute + for term in active_terms + for attribute, _ in term + if attribute >= next_attribute + ) + branch_count = count_with_attribute(active_terms, branching_attribute) + if branching_attribute == next_attribute: + return branch_count + + skipped_assignments = ( + self._remaining_products[next_attribute] + // self._remaining_products[branching_attribute] + ) + return skipped_assignments * branch_count + + def count_with_attribute(active_terms: Terms, attribute: int) -> int: + unconstrained: list[Term] = [] + terms_by_value: dict[int, list[Term]] = {} + + for term in active_terms: + literal_index = next( + (i for i, literal in enumerate(term) if literal[0] == attribute), + None, + ) + if literal_index is None: + unconstrained.append(term) + continue + + _, value = term[literal_index] + reduced = term[:literal_index] + term[literal_index + 1 :] + terms_by_value.setdefault(value, []).append(reduced) + + base_terms = self._canonicalize(unconstrained) + skipped_concrete_values = self.domain_sizes[attribute] - len(terms_by_value) + result = (skipped_concrete_values + 1) * count(base_terms, attribute + 1) + + for value_terms in terms_by_value.values(): + combined = self._canonicalize([*unconstrained, *value_terms]) + result += count(combined, attribute + 1) + return result + + uncovered = count(terms, 0) + return CoverageSpaceScore( + uncovered_patterns=uncovered, + total_patterns=prod(size + 1 for size in self.domain_sizes), + effective_term_count=len(terms), + ) + + def _normalize_terms(self, raw_terms: Iterable[Iterable[Literal]]) -> Terms: + terms: list[Term] = [] + for raw_term in raw_terms: + term = tuple(sorted((int(attribute), int(value)) for attribute, value in raw_term)) + previous_attribute = -1 + for attribute, value in term: + if attribute < 0 or attribute >= len(self.domain_sizes): + raise ValueError(f"Invalid attribute index in MUP: {attribute}.") + if value < 0 or value >= self.domain_sizes[attribute]: + raise ValueError( + f"Invalid value index {value} for attribute {attribute}." + ) + if attribute == previous_attribute: + raise ValueError( + f"A MUP constrains attribute {attribute} more than once." + ) + previous_attribute = attribute + terms.append(term) + return self._canonicalize(terms) + + @staticmethod + def _canonicalize(raw_terms: Iterable[Term]) -> Terms: + """Remove duplicate terms and terms subsumed by a general term.""" + terms: list[Term] = [] + for raw_term in raw_terms: + term = tuple(raw_term) + term_set = frozenset(term) + if any(frozenset(existing).issubset(term_set) for existing in terms): + continue + terms = [ + existing + for existing in terms + if not term_set.issubset(frozenset(existing)) + ] + terms.append(term) + return tuple(sorted(terms)) diff --git a/metis/metric/diversity/diversity_coverageGap.py b/metis/metric/diversity/diversity_coverageGap.py new file mode 100644 index 0000000..5f7d449 --- /dev/null +++ b/metis/metric/diversity/diversity_coverageGap.py @@ -0,0 +1,201 @@ +"""MUP-based diversity assessment using exact finite-domain DNF counting.""" + +from __future__ import annotations + +import csv +import io +from pathlib import Path + +import pandas as pd + +from metis.metric.config import MetricConfig +from metis.metric.diversity.coverage_space_counter import CoverageSpaceCounter, Literal +from metis.metric.diversity.diversity_coverageGap_config import ( + diversity_coverageGap_config, +) +from metis.metric.metric import Metric +from metis.utils.dq_dimension import DQDimension +from metis.utils.dq_granularity import DQGranularity +from metis.utils.result import DQResult + + +class diversity_coverageGap(Metric): + """Calculate coverage gap from a supplied MUP frontier. + + The exact paper-defined coverage gap is included in ``DQexplanation``. + ``DQvalue`` is its complement (coverage space) so the metric follows the + existing METIS convention that larger DQ scores indicate better quality. + """ + + _gui_requires_reference: bool = False + _gui_config_required: bool = True + _gui_callable_config: bool = False + _gui_recommended_granularities: frozenset = frozenset({DQGranularity.TABLE}) + _gui_description: str = ( + "Exact coverage-based diversity from Maximal Uncovered Patterns (MUPs). " + "Counts the MUP-induced finite-domain DNF union. The METIS score is " + "coverage space (1 − coverage gap); exact gap and counts are shown in details." + ) + + def assess( + self, + data: pd.DataFrame, + reference: pd.DataFrame | None = None, + metric_config: str | MetricConfig | None = None, + ) -> list[DQResult]: + config = self.load_config(metric_config or "", diversity_coverageGap_config) + attributes = list(config.attributes or []) + missing = [attribute for attribute in attributes if attribute not in data.columns] + if missing: + raise ValueError(f"Configured diversity attributes are missing: {missing}.") + + domains = [self._domain(data[attribute]) for attribute in attributes] + domain_indexes = [ + {value: index for index, value in enumerate(domain)} for domain in domains + ] + text = self._read_mups(config) + terms, raw_mup_count, mup_coverages, intermediate_field_count = self._parse_mups( + text=text, + attributes=attributes, + domain_indexes=domain_indexes, + wildcard=config.wildcard, + delimiter=config.delimiter, + mincov=config.mincov, + ) + + score = CoverageSpaceCounter(len(domain) for domain in domains).calculate(terms) + source_name = config.mups_filename + if source_name is None and config.mups_path: + source_name = Path(config.mups_path).name + + explanation = { + "coverage_gap": score.coverage_gap, + "coverage_space": score.coverage_space, + "uncovered_patterns": str(score.uncovered_patterns), + "total_patterns": str(score.total_patterns), + "mup_count": raw_mup_count, + "mup_coverage_min": min(mup_coverages) if mup_coverages else None, + "mup_coverage_max": max(mup_coverages) if mup_coverages else None, + "mup_coverages_validated_against_mincov": config.mincov is not None, + "effective_dnf_terms": score.effective_term_count, + "dnf_method": "exact finite-domain DNF counting", + "attributes": attributes, + "domain_sizes": { + attribute: len(domain) + for attribute, domain in zip(attributes, domains) + }, + "mincov": config.mincov, + "mups_file": source_name, + "ignored_intermediate_fields_per_mup": intermediate_field_count, + } + + return [DQResult( + timestamp=pd.Timestamp.now(), + DQdimension=DQDimension.DIVERSITY, + DQmetric=self.__class__.__name__, + DQgranularity=DQGranularity.TABLE, + DQvalue=score.coverage_space, + DQexplanation=explanation, + columnNames=attributes, + configJson=config.to_json(), + )] + + @staticmethod + def _domain(series: pd.Series) -> tuple[str, ...]: + values = tuple(dict.fromkeys(diversity_coverageGap._normalize_value(v) for v in series)) + if not values: + raise ValueError( + f"Diversity attribute '{series.name}' has no observed domain values." + ) + return values + + @staticmethod + def _normalize_value(value) -> str: + if pd.isna(value): + return "" + if isinstance(value, float) and value.is_integer(): + return str(int(value)) + return str(value).strip() + + @staticmethod + def _read_mups(config: diversity_coverageGap_config) -> str: + if config.mups_content is not None: + return config.mups_content + + path = Path(config.mups_path or "") + try: + return path.read_text(encoding="utf-8-sig") + except UnicodeDecodeError: + return path.read_text(encoding="latin-1") + except OSError as exc: + raise ValueError(f"Could not read MUP file '{path}': {exc}") from exc + + @staticmethod + def _parse_mups( + text: str, + attributes: list[str], + domain_indexes: list[dict[str, int]], + wildcard: str, + delimiter: str, + mincov: int | None, + ) -> tuple[list[tuple[Literal, ...]], int, list[int], int | list[int]]: + terms: list[tuple[Literal, ...]] = [] + mup_coverages: list[int] = [] + expected_fields = len(attributes) + intermediate_counts: set[int] = set() + + reader = csv.reader(io.StringIO(text), delimiter=delimiter) + for line_number, raw_fields in enumerate(reader, start=1): + if not raw_fields or all(not field.strip() for field in raw_fields): + continue + if raw_fields[0].lstrip().startswith("#"): + continue + if len(raw_fields) < expected_fields + 1: + raise ValueError( + f"MUP line {line_number} has {len(raw_fields)} fields; " + f"{expected_fields} pattern fields plus the final MUP coverage " + "field are required." + ) + + intermediate_counts.add(len(raw_fields) - expected_fields - 1) + term: list[Literal] = [] + for attribute_index, attribute in enumerate(attributes): + value = raw_fields[attribute_index].strip() + if value == wildcard: + continue + normalized = diversity_coverageGap._normalize_value(value) + value_index = domain_indexes[attribute_index].get(normalized) + if value_index is None: + raise ValueError( + f"Unknown value '{value}' for diversity attribute " + f"'{attribute}' on MUP line {line_number}." + ) + term.append((attribute_index, value_index)) + terms.append(tuple(term)) + + coverage_raw = raw_fields[-1].strip() + try: + coverage = int(coverage_raw) + except ValueError as exc: + raise ValueError( + f"MUP coverage '{coverage_raw}' on line {line_number} is not an integer." + ) from exc + if coverage < 0: + raise ValueError( + f"MUP coverage on line {line_number} must be non-negative." + ) + if mincov is not None and coverage >= mincov: + raise ValueError( + f"MUP coverage {coverage} on line {line_number} is not below " + f"mincov {mincov}." + ) + mup_coverages.append(coverage) + + intermediate_field_count: int | list[int] + if not intermediate_counts: + intermediate_field_count = 0 + elif len(intermediate_counts) == 1: + intermediate_field_count = next(iter(intermediate_counts)) + else: + intermediate_field_count = sorted(intermediate_counts) + return terms, len(terms), mup_coverages, intermediate_field_count diff --git a/metis/metric/diversity/diversity_coverageGap_config.py b/metis/metric/diversity/diversity_coverageGap_config.py new file mode 100644 index 0000000..f5f2a04 --- /dev/null +++ b/metis/metric/diversity/diversity_coverageGap_config.py @@ -0,0 +1,57 @@ +"""Configuration for the MUP-based coverage-gap metric.""" + +from __future__ import annotations + +from dataclasses import dataclass + +from metis.metric.config import MetricConfig + + +@dataclass +class diversity_coverageGap_config(MetricConfig): + """Describe the MUP source and its positional dataset-column mapping. + + ``mups_path`` is intended for Python/API use. The Streamlit GUI supplies + ``mups_content`` directly after upload. Exactly one source must be set. + MUP fields are matched positionally to ``attributes``. The final field of + every non-empty row is the MUP's actual dataset coverage. It is validated + against ``mincov`` when the threshold is available, but it does not affect + the geometric DNF union count. + """ + + mups_path: str | None = None + mups_content: str | None = None + mups_filename: str | None = None + attributes: list[str] | None = None + mincov: int | None = None + wildcard: str = "x" + delimiter: str = "," + + def validate(self) -> None: + sources = int(self.mups_path is not None) + int(self.mups_content is not None) + if sources != 1: + raise ValueError("Provide exactly one of mups_path or mups_content.") + if self.mups_path is not None and not self.mups_path.strip(): + raise ValueError("mups_path must not be empty.") + if not self.attributes: + raise ValueError( + "Select the dataset attributes that correspond positionally to the MUP fields." + ) + if len(set(self.attributes)) != len(self.attributes): + raise ValueError("attributes must not contain duplicates.") + if self.mincov is not None and self.mincov <= 0: + raise ValueError("mincov must be positive when provided.") + if not self.wildcard: + raise ValueError("wildcard must not be empty.") + if len(self.delimiter) != 1: + raise ValueError("delimiter must be exactly one character.") + + def to_json(self) -> dict: + return { + "name": self.__class__.__name__, + "mups_file": self.mups_filename or self.mups_path, + "attributes": list(self.attributes or []), + "mincov": self.mincov, + "wildcard": self.wildcard, + "delimiter": self.delimiter, + } diff --git a/metis/utils/dq_dimension.py b/metis/utils/dq_dimension.py index 5efaae9..4d30c27 100644 --- a/metis/utils/dq_dimension.py +++ b/metis/utils/dq_dimension.py @@ -12,3 +12,4 @@ class DQDimension(StrEnum): MINIMALITY = "Minimality" VALIDITY = "Validity" READABILITY = "Readability" + DIVERSITY = "Diversity"