diff --git a/.github/workflows/dataset-validator[tabular].yml b/.github/workflows/dataset-validator[tabular].yml new file mode 100644 index 00000000..143b88cf --- /dev/null +++ b/.github/workflows/dataset-validator[tabular].yml @@ -0,0 +1,44 @@ +name: Run Tabular Validator + +on: + pull_request: + branches: + - main + paths: + - 'data/**' + workflow_dispatch: + inputs: + mode: + description: "Validation mode" + required: false + default: "tabular" + +jobs: + validate: + name: Validate Statistics Data [Tabular] + runs-on: ubuntu-latest + + env: + DATASET_PATH: ${{ vars.DATASET_DIR_PATH_VAR || '../../data' }} + MODE: ${{ github.event.inputs.mode || 'tabular' }} + + steps: + - name: Checkout repository + uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: '3.12' + + - name: Install dependencies + run: pip install -r scripts/validator/requirements.txt + + - name: Run validator + working-directory: scripts/validator + run: | + echo "Running with:" + echo "Dataset Directory: $DATASET_PATH" + echo "Validation Mode: $MODE" + + python main.py "$DATASET_PATH" "$MODE" \ No newline at end of file diff --git a/.gitignore b/.gitignore index 3deeccb3..3bef95e2 100644 --- a/.gitignore +++ b/.gitignore @@ -15,4 +15,7 @@ ingestion/.env # Generated zip files (not used by website) website/static/downloads/archive_Data.zip website/static/downloads/sources_Data.zip -website/static/downloads/statistics_Data.zip \ No newline at end of file +website/static/downloads/statistics_Data.zip + +# Virtual environment +venv/ \ No newline at end of file diff --git a/scripts/validator/README.md b/scripts/validator/README.md new file mode 100644 index 00000000..879b4f40 --- /dev/null +++ b/scripts/validator/README.md @@ -0,0 +1,73 @@ +# Data Validator Program + +A command-line tool for validating datasets using configurable validator types. + +--- + +## Getting Started + +### Prerequisites + +- Python 3.x +- `pip` and `venv` + +### Local Setup + +```bash +# Navigate to the project directory +cd scripts/validator + +# Create a virtual environment +python3 -m venv venv + +# Activate the virtual environment +source venv/bin/activate + +# Install dependencies +pip install -r requirements.txt +``` + +### Running the Program + +```bash +python main.py +``` + +**Example:** + +```bash +python main.py ../../data/statistics tabular +``` + +--- + +## Supported Validators + +| Validator | Description | Validations | Status | +|-----------|-------------|-------------|--------| +| `tabular` | Validates tabular data | `schema-validation` — Checks that the data conforms to a predefined schema (structures, types, constraints)

`duplicate-columns` — Detects columns that appear more than once

`row-column-mismatch` — Detects rows where the number of fields does not match the number of defined columns

`data-types-mismatch` — Identifies values that do not match the expected data type for their column (e.g. text in a numeric field)

`empty-values` — Flag cells that are null, blank, or contain only whitespace where a value is required

`value-overflow` — Catches values that exceed the maximum allowed length or numeric range for their column | ✅ Available | + +--- + +## Project Structure + +``` +scripts/validator/ +│ +├── main.py # Entry point +├── requirements.txt # Python dependencies +├── README.md # Project documentation +│ +├── core/ +│ ├── baseRunner.py # Base runner logic +│ └── baseValidator.py # Base validator interface +│ +├── models/ +│ └── tabularSchema.json # Schema definition for tabular validation +│ +├── utils/ +│ └── utils.py # Shared utility functions +│ +└── validators/ + └── tabular.py # Tabular validator implementation +``` \ No newline at end of file diff --git a/scripts/validator/core/baseRunner.py b/scripts/validator/core/baseRunner.py new file mode 100644 index 00000000..6dc80e2d --- /dev/null +++ b/scripts/validator/core/baseRunner.py @@ -0,0 +1,34 @@ +import sys +from pathlib import Path +from utils.utils import Utils + +def run_validation(file_path, validator): + validator = validator() + all_errors = [] + all_warnings = [] + + paths = list(Path(file_path).rglob("data.json")) + + if not paths: + print("[INFO] No data.json files found") + sys.exit(0) + + for path in paths: + errors, warnings = validator.validate_data(path) + all_errors.extend(errors) + all_warnings.extend(warnings) + + if all_errors: + print(f" - {len(all_errors)} errors found") + for error in all_errors: + print(Utils.format_issue(error)) + + if all_warnings: + print(f" - {len(all_warnings)} warnings found") + for warning in all_warnings: + print(Utils.format_issue(warning)) + + if not all_errors and not all_warnings: + print("All data is valid ✅") + + sys.exit(1 if all_errors else 0) diff --git a/scripts/validator/core/baseValidator.py b/scripts/validator/core/baseValidator.py new file mode 100644 index 00000000..998523fb --- /dev/null +++ b/scripts/validator/core/baseValidator.py @@ -0,0 +1,3 @@ +class BaseValidator: + def validate(self, file_path): + raise NotImplementedError \ No newline at end of file diff --git a/scripts/validator/main.py b/scripts/validator/main.py new file mode 100644 index 00000000..a9a303d9 --- /dev/null +++ b/scripts/validator/main.py @@ -0,0 +1,16 @@ +from core.baseRunner import run_validation +from validators.tabular import TabularValidator +import sys + +def main(file_path, validator): + if validator == "tabular": + run_validation(file_path, TabularValidator) + else: + print("Invalid validator") + sys.exit(1) + +if __name__ == "__main__": + if len(sys.argv) < 3: + print("Usage: python main.py ") + sys.exit(1) + main(sys.argv[1], sys.argv[2]) \ No newline at end of file diff --git a/scripts/validator/models/tabularSchema.json b/scripts/validator/models/tabularSchema.json new file mode 100644 index 00000000..d3bdb1e3 --- /dev/null +++ b/scripts/validator/models/tabularSchema.json @@ -0,0 +1,22 @@ +{ + "type": "object", + "required": ["columns", "rows"], + "properties": { + "columns": { + "type": "array", + "items": { + "type": "string" + }, + "minItems": 1 + }, + "rows": { + "type": "array", + "items": { + "type": "array", + "items": {}, + "minItems": 1 + } + } + }, + "additionalProperties": false +} \ No newline at end of file diff --git a/scripts/validator/requirements.txt b/scripts/validator/requirements.txt new file mode 100644 index 00000000..6e221097 --- /dev/null +++ b/scripts/validator/requirements.txt @@ -0,0 +1,3 @@ +jsonschema>=4.25,<5 +pytest>=8.0.0 + diff --git a/scripts/validator/tests/test_base_runner.py b/scripts/validator/tests/test_base_runner.py new file mode 100644 index 00000000..717c60a9 --- /dev/null +++ b/scripts/validator/tests/test_base_runner.py @@ -0,0 +1,67 @@ +import pytest +from unittest.mock import patch, MagicMock +from core.baseRunner import run_validation + +class MockValidator: + def validate_data(self, path): + # We'll override this in specific tests + return [], [] + +@patch("core.baseRunner.Path") +def test_run_validation_no_files_found(mock_path_class, capsys): + # Setup mock Path.rglob to return an empty list + mock_path_instance = MagicMock() + mock_path_instance.rglob.return_value = [] + mock_path_class.return_value = mock_path_instance + + with pytest.raises(SystemExit) as excinfo: + run_validation("mock_dir", MockValidator) + + assert excinfo.value.code == 0 + captured = capsys.readouterr() + assert "[INFO] No data.json files found" in captured.out + +@patch("core.baseRunner.Path") +def test_run_validation_all_valid(mock_path_class, capsys): + # Setup mock paths to return two file + mock_path_instance = MagicMock() + # Need to return an interable for rglob + mock_path_instance.rglob.return_value = ["mock_data_1.json", "mock_data_2.json"] + mock_path_class.return_value = mock_path_instance + + class SuccessValidator(MockValidator): + def validate(self, path): + return [], [] + + with pytest.raises(SystemExit) as excinfo: + run_validation("mock_dir", SuccessValidator) + + assert excinfo.value.code == 0 + captured = capsys.readouterr() + assert "All data is valid ✅" in captured.out + +@patch("core.baseRunner.Path") +def test_run_validation_with_errors_and_warnings(mock_path_class, capsys): + mock_path_instance = MagicMock() + mock_path_instance.rglob.return_value = ["mock_data.json"] + mock_path_class.return_value = mock_path_instance + + class FailureValidator(MockValidator): + def validate_data(self, path): + errors = [{ + "type": "error", "file": str(path), "row": 1, "column": "id", "message": "is wrong" + }] + warnings = [{ + "type": "warning", "file": str(path), "row": 2, "column": "name", "message": "is empty" + }] + return errors, warnings + + with pytest.raises(SystemExit) as excinfo: + run_validation("mock_dir", FailureValidator) + + assert excinfo.value.code == 1 + captured = capsys.readouterr() + assert "1 errors found" in captured.out + assert "1 warnings found" in captured.out + assert "[ERROR]" in captured.out + assert "[WARNING]" in captured.out diff --git a/scripts/validator/tests/test_base_validator.py b/scripts/validator/tests/test_base_validator.py new file mode 100644 index 00000000..098f97ee --- /dev/null +++ b/scripts/validator/tests/test_base_validator.py @@ -0,0 +1,7 @@ +import pytest +from core.baseValidator import BaseValidator + +def test_base_validator_validate_raises_not_implemented(): + validator = BaseValidator() + with pytest.raises(NotImplementedError): + validator.validate("some_file.json") diff --git a/scripts/validator/tests/test_main.py b/scripts/validator/tests/test_main.py new file mode 100644 index 00000000..d610a66e --- /dev/null +++ b/scripts/validator/tests/test_main.py @@ -0,0 +1,16 @@ +from unittest.mock import patch +from main import main + +@patch("main.sys.exit") +@patch("main.run_validation") +def test_main_valid_validator(mock_run_validation, mock_sys_exit): + main("some_dir", "tabular") + mock_run_validation.assert_called_once() + assert mock_sys_exit.call_count == 0 + +@patch("main.sys.exit") +@patch("main.print") +def test_main_invalid_validator(mock_print, mock_sys_exit): + main("some_dir", "unknown_validator") + mock_print.assert_called_with("Invalid validator") + mock_sys_exit.assert_called_with(1) diff --git a/scripts/validator/tests/test_tabular.py b/scripts/validator/tests/test_tabular.py new file mode 100644 index 00000000..24e39a74 --- /dev/null +++ b/scripts/validator/tests/test_tabular.py @@ -0,0 +1,116 @@ +import pytest +import json +from unittest.mock import patch, mock_open +from validators.tabular import TabularValidator + +@pytest.fixture +def tabular_validator(): + return TabularValidator() + +def test_check_duplicate_columns(tabular_validator): + # No duplicates + errors = tabular_validator._check_duplicate_columns("mock.json", ["id", "name", "age"]) + assert len(errors) == 0 + + # Duplicates exist + errors = tabular_validator._check_duplicate_columns("mock.json", ["id", "name", "id", "age", "name"]) + assert len(errors) == 1 + assert errors[0]["type"] == "error" + assert "Duplicate column names found" in errors[0]["message"] + assert set(errors[0]["column"]) == {"id", "name"} + +def test_check_row_column_mismatch(tabular_validator): + # Match + errors = tabular_validator._check_row_column_mismatch("mock.json", 0, [1, "test"], 2) + assert len(errors) == 0 + + # Mismatch + errors = tabular_validator._check_row_column_mismatch("mock.json", 1, [1, "test", "extra"], 2) + assert len(errors) == 1 + assert errors[0]["type"] == "error" + assert errors[0]["row"] == 1 + assert "has 3 value(s), expected 2 value(s)" in errors[0]["message"] + +def test_check_data_types(tabular_validator): + columns = ["id", "score", "name"] + first_row = [1, 5.5, "Alice"] # int, float, str + + # Valid row matching first row types + row_valid = [2, 10.0, "Bob"] + errors = tabular_validator._check_data_types("mock.json", 1, row_valid, first_row, columns) + assert len(errors) == 0 + + # Allow int for float columns + row_valid_int_for_float = [3, 10, "Charlie"] + errors = tabular_validator._check_data_types("mock.json", 2, row_valid_int_for_float, first_row, columns) + assert len(errors) == 0 + + # Invalid row + row_invalid = ["four", "five", 6] + errors = tabular_validator._check_data_types("mock.json", 3, row_invalid, first_row, columns) + assert len(errors) == 3 + assert "expected int" in errors[0]["message"] + assert "expected float or whole number" in errors[1]["message"] + assert "expected str" in errors[2]["message"] + +def test_check_empty_values(tabular_validator): + columns = ["id", "name", "notes"] + row_empty = [1, "", None] + warnings = tabular_validator._check_empty_values("mock.json", 0, row_empty, columns) + assert len(warnings) == 2 + assert warnings[0]["column"] == "name" + assert warnings[0]["message"] == "has empty value" + assert warnings[1]["column"] == "notes" + assert warnings[1]["message"] == "has empty value" + +def test_check_value_overflow(tabular_validator): + columns = ["id", "big_num"] + row = [1, 2_147_483_648] # Over max int32 + warnings = tabular_validator._check_value_overflow("mock.json", 0, row, columns) + assert len(warnings) == 1 + assert warnings[0]["column"] == "big_num" + assert "which is a BIGINT as it exceeds PostgreSQL's 32-bit integer limit." in warnings[0]["message"] + +@patch("validators.tabular.open") +def test_validate_invalid_json(mock_file_open, tabular_validator): + mock_file_open.side_effect = FileNotFoundError() + errors, warnings = tabular_validator.validate_data("missing.json") + assert len(errors) == 1 + assert len(warnings) == 0 + assert "Invalid JSON" in errors[0] + +@patch("validators.tabular.open", new_callable=mock_open, read_data='{"invalid": "schema"}') +def test_validate_schema_error(_mock_file_open, tabular_validator): + errors, warnings = tabular_validator.validate_data("schema_error.json") + assert len(errors) == 1 + assert len(warnings) == 0 + assert "Schema error → 'columns' is a required property" in errors[0]["message"] + +@patch("validators.tabular.open", new_callable=mock_open) +def test_validate_success(mock_file_open, tabular_validator): + valid_data = { + "columns": ["id", "name"], + "rows": [ + [1, "Alice"], + [2, "Bob"] + ] + } + mock_file_open.return_value.read.return_value = json.dumps(valid_data) + errors, warnings = tabular_validator.validate_data("valid.json") + assert len(errors) == 0 + assert len(warnings) == 0 + +@patch("validators.tabular.open", new_callable=mock_open) +def test_validate_with_warnings_and_errors(mock_file_open, tabular_validator): + invalid_data = { + "columns": ["id", "id", "name"], + "rows": [ + [1, 1, "Alice"], + [2, "not-int", "Bob"] + ] + } + mock_file_open.return_value.read.return_value = json.dumps(invalid_data) + errors, warnings = tabular_validator.validate_data("invalid.json") + + # Should have duplicate column error, and type mismatch error for 'not-int' + assert len(errors) >= 2 diff --git a/scripts/validator/tests/test_utils.py b/scripts/validator/tests/test_utils.py new file mode 100644 index 00000000..17c4acae --- /dev/null +++ b/scripts/validator/tests/test_utils.py @@ -0,0 +1,79 @@ +from utils.utils import Utils + + +def test_format_issue_with_row_and_single_column(): + issue = { + "type": "error", + "file": "data.json", + "row": 5, + "column": "name", + "message": "is invalid", + } + result = Utils.format_issue(issue) + assert result == "[ERROR] data.json: Row 5, Column 'name' is invalid" + + +def test_format_issue_with_row_and_multiple_columns(): + issue = { + "type": "warning", + "file": "data.csv", + "row": 10, + "column": ["age", "dob"], + "message": "have conflicting values", + } + result = Utils.format_issue(issue) + assert ( + result + == "[WARNING] data.csv: Row 10, Columns [age, dob] have conflicting values" + ) + + +def test_format_issue_without_row_and_column(): + issue = { + "type": "error", + "file": "config.json", + "message": "File not found or unreadable", + } + result = Utils.format_issue(issue) + assert result == "[ERROR] config.json: File not found or unreadable" + + +def test_format_issue_without_rows_and_columns(): + issue = { + "type": "error", + "file": "file.csv", + "row": None, + "column": ["header1", "header2"], + "message": "No rows found", + } + result = Utils.format_issue(issue) + assert result == "[ERROR] file.csv: , Columns [header1, header2] No rows found" + + +def test_format_issue_without_row_with_column(): + issue = { + "type": "error", + "file": "schema.json", + "column": ["header1", "header2"], + "message": "Duplicate columns", + } + result = Utils.format_issue(issue) + assert ( + result == "[ERROR] schema.json: , Columns [header1, header2] Duplicate columns" + ) + + +def test_fits_in_int32(): + # Boundary values + assert Utils.fits_in_int32(-2_147_483_648) is True + assert Utils.fits_in_int32(2_147_483_647) is True + + # Internal values + assert Utils.fits_in_int32(0) is True + assert Utils.fits_in_int32(1000) is True + assert Utils.fits_in_int32(-50000) is True + + # Out of bounds + assert Utils.fits_in_int32(-2_147_483_649) is False + assert Utils.fits_in_int32(2_147_483_648) is False + assert Utils.fits_in_int32(10_000_000_000) is False diff --git a/scripts/validator/utils/utils.py b/scripts/validator/utils/utils.py new file mode 100644 index 00000000..cd1c9b41 --- /dev/null +++ b/scripts/validator/utils/utils.py @@ -0,0 +1,22 @@ +class Utils: + @staticmethod + def format_issue(issue): + location = "" + + if issue.get("row") is not None: + location += f"Row {issue['row']}" + + if issue.get("column"): + if isinstance(issue["column"], list): + cols = ", ".join(issue["column"]) + location += f", Columns [{cols}]" + else: + location += f", Column '{issue['column']}'" + + return f"[{issue['type'].upper()}] {issue['file']}: {location} {issue['message']}" + + @staticmethod + def fits_in_int32(value: int) -> bool: + INT32_MIN = -2_147_483_648 + INT32_MAX = 2_147_483_647 + return INT32_MIN <= value <= INT32_MAX \ No newline at end of file diff --git a/scripts/validator/validators/tabular.py b/scripts/validator/validators/tabular.py new file mode 100644 index 00000000..336d78dc --- /dev/null +++ b/scripts/validator/validators/tabular.py @@ -0,0 +1,146 @@ +from core.baseValidator import BaseValidator +import json +from jsonschema import validate, ValidationError +from collections import Counter +from utils.utils import Utils +from pathlib import Path + +class TabularValidator(BaseValidator): + def __init__(self): + with open(Path(__file__).parent / "../models/tabularSchema.json") as f: + self.schema = json.load(f) + + def _check_duplicate_columns(self, file_path, columns): + if len(columns) != len(set(columns)): + column_counts = Counter(columns) + duplicates = [col for col, count in column_counts.items() if count > 1] + if duplicates: + return [ + { + "type": "error", + "file": file_path, + "row": None, + "column": duplicates, + "message": f"Duplicate column names found: {', '.join(duplicates)}", + } + ] + return [] + + def _check_row_column_mismatch(self, file_path, row_index, row, num_cols): + if len(row) != num_cols: + return [{ + "type": "error", + "file": file_path, + "row": row_index, + "column": None, + "message": f"has {len(row)} value(s), expected {num_cols} value(s)", + }] + return [] + + def _check_data_types(self, file_path, row_index, row, first_row, columns): + errors = [] + for j, value in enumerate(row): + expected_type = type(first_row[j]) + if expected_type is float: + allowed_types = (float, int) + expected_msg = "float or whole number" + else: + allowed_types = (expected_type) + expected_msg = expected_type.__name__ + + if not isinstance(value, allowed_types): + errors.append({ + "type": "error", + "file": file_path, + "row": row_index, + "column": columns[j], + "message": f"has {value} ({type(value).__name__}), expected {expected_msg}", + }) + return errors + + def _check_empty_values(self, file_path, row_index, row, columns): + warnings = [] + for j, value in enumerate(row): + str_value = str(value).strip() if value is not None else "" + if str_value == "": + warnings.append({ + "type": "warning", + "file": file_path, + "row": row_index, + "column": columns[j], + "message": "has empty value", + }) + return warnings + + def _check_value_overflow(self, file_path, row_index, row, columns): + warnings = [] + for j, value in enumerate(row): + if isinstance(value, int): + if not Utils.fits_in_int32(value): + warnings.append({ + "type": "warning", + "file": file_path, + "row": row_index, + "column": columns[j], + "message": f"has {value} ({type(value).__name__}), which is a BIGINT as it exceeds PostgreSQL's 32-bit integer limit.", + }) + return warnings + + def validate_data(self, file_path): + errors = [] + warnings = [] + + # Load JSON + try: + with open(file_path) as f: + data = json.load(f) + except (json.JSONDecodeError, FileNotFoundError) as e: + return [f"[ERROR] {file_path}: Invalid JSON ({e})"], [] + + # 1. Schema validation + try: + # imported validate function from the jsonschema library + validate(instance=data, schema=self.schema) + except ValidationError as e: + errors.append({ + "type": "error", + "file": file_path, + "row": None, + "column": None, + "message": f"Schema error → {e.message}" + }) + return errors, warnings + + # 2. Custom validation + columns = data.get("columns", []) + rows = data.get("rows", []) + num_cols = len(columns) + + if not rows or not columns: + if not rows: + message = "No rows found" + elif not columns: + message = "No columns found" + else: + message = "No rows or columns found" + + errors.append({ + "type": "error", + "file": file_path, + "row": rows if rows else None, + "column": columns if columns else None, + "message": message + }) + return errors, warnings + + errors.extend(self._check_duplicate_columns(file_path, columns)) + + for row_index, row in enumerate(rows, start=1): + errors.extend(self._check_row_column_mismatch(file_path, row_index, row, num_cols)) + warnings.extend(self._check_empty_values(file_path, row_index, row, columns)) + warnings.extend(self._check_value_overflow(file_path, row_index, row, columns)) + errors.extend(self._check_data_types(file_path, row_index, row, rows[0], columns)) + + return errors, warnings + +