diff --git a/.mise.toml b/.mise.toml index be3149d..0213a69 100644 --- a/.mise.toml +++ b/.mise.toml @@ -1,4 +1,4 @@ [tools] python="3.12" poetry="2.4.1" -java="liberica-1.8.0" +java="zulu-17.60.17" diff --git a/.tool-versions b/.tool-versions index e111eee..eb68911 100644 --- a/.tool-versions +++ b/.tool-versions @@ -1,3 +1,3 @@ python 3.12.12 poetry 2.4.1 -java liberica-1.8.0 +java zulu-17.60.17 diff --git a/README.md b/README.md index e94b29e..f91ce2e 100644 --- a/README.md +++ b/README.md @@ -17,6 +17,7 @@ __The DVE offers__: - Format normalization to Parquet for a unified data representation - Data modelling and typecasting - Business-rule validations executed on supported backends such as Spark and DuckDB, with the option to add custom backends +- Validate and enforce referential integrity checks with minimal configuration - Deriving new fields and entities - Clear validation reporting, including summary insights and record-level error messages @@ -41,7 +42,8 @@ Below is a list of features that we would like to implement or have been request | Uplift to Python 3.11 | 0.2.0 | Yes | | Uplift Pyspark to 3.5 | 0.8.0 | Yes | | Allow DVE to run on Python 3.12+ | 0.8.0 | Yes | -| Upgrade to Pydantic 2.0 | 0.9.0 | Yes | +| Upgrade to Pydantic 2.0 | 0.9.0 | Yes | +| Upgrade DuckDB to v1.4 | 0.10.0 | Yes | | Uplift Pyspark to 4.0+ | TBA | No | | Polars upgrade to v1+ | TBA | No | | DuckDB upgrade to v1.5+ | TBA | No | diff --git a/docs/advanced_guidance/json_schemas/contract/components/base_entity.schema.json b/docs/advanced_guidance/json_schemas/contract/components/base_entity.schema.json index b89291a..06607cb 100644 --- a/docs/advanced_guidance/json_schemas/contract/components/base_entity.schema.json +++ b/docs/advanced_guidance/json_schemas/contract/components/base_entity.schema.json @@ -26,8 +26,15 @@ "items": { "type": "string" } + }, + "reader_additional_checks": { + "description": "A mapping of additional checks to perform on entities after initial read", + "type": "object", + "additionalProperties": { + "$ref": "reader_additional_checks.json" } - }, + } +}, "required": ["fields"] } diff --git a/docs/advanced_guidance/json_schemas/contract/components/reader_additional_checks.schema.json b/docs/advanced_guidance/json_schemas/contract/components/reader_additional_checks.schema.json new file mode 100644 index 0000000..6c01eb7 --- /dev/null +++ b/docs/advanced_guidance/json_schemas/contract/components/reader_additional_checks.schema.json @@ -0,0 +1,22 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "data-ingest:contract/components/reader_additional_checks.schema.json", + "title": "reader_additional_checks", + "description": "Additional checks to perform on initially read entities", + "type": "object", + "properties": { + "error_code": { + "description": "The code to be used for the additional check specified", + "type": "string" + }, + "error_message": { + "description": "The message to be displayed for the additional check specified.", + "type": "string" + } + }, + "required": [ + "error_code", + "error_message" + ], + "additionalProperties": false +} \ No newline at end of file diff --git a/docs/advanced_guidance/json_schemas/dataset.schema.json b/docs/advanced_guidance/json_schemas/dataset.schema.json index 4e85011..af8b620 100644 --- a/docs/advanced_guidance/json_schemas/dataset.schema.json +++ b/docs/advanced_guidance/json_schemas/dataset.schema.json @@ -10,6 +10,9 @@ }, "transformations": { "$ref": "transformations/transformations.schema.json" + }, + "entity_relationships": { + "$ref": "entity_relationships.schema.json" } }, "required": [ diff --git a/docs/advanced_guidance/json_schemas/entity_relationships.schema.json b/docs/advanced_guidance/json_schemas/entity_relationships.schema.json new file mode 100644 index 0000000..0c2abbb --- /dev/null +++ b/docs/advanced_guidance/json_schemas/entity_relationships.schema.json @@ -0,0 +1,50 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "data-ingest:entity_relationships.schema.json", + "title": "entity_relationships", + "description": "Description of relationships to link normalised entities back to parent entities.", + "type": "object", + "patternProperties": { + "^[A-Za-z0-9_]+.$": { + "type": "object", + "properties": { + "parent_entity": { + "type": "string" + }, + "join_fields": { + "type": "object", + "additionalProperties": { + "type": "string" + } + }, + "is_root_entity": { + "type": "boolean" + }, + "mandatory": { + "type": "boolean" + }, + "missing_parent_id_error_code": { + "type": "string" + }, + "missing_parent_id_error_message": { + "type": "string" + }, + "no_valid_records_error_code": { + "type": "string" + }, + "no_valid_records_error_message": { + "type": "string" + }, + "empty_entity_error_code": { + "type": "string", + "minLength": 1 + }, + "empty_entity_error_message": { + "type": "string", + "minLength": 1 + } + }, + "additionalProperties": false + } + } +} \ No newline at end of file diff --git a/docs/advanced_guidance/package_documentation/entity_hierarchy.md b/docs/advanced_guidance/package_documentation/entity_hierarchy.md new file mode 100644 index 0000000..4080dcb --- /dev/null +++ b/docs/advanced_guidance/package_documentation/entity_hierarchy.md @@ -0,0 +1,5 @@ +::: dve.core_engine.configuration.v1.hierarchy + handler: python + options: + show_root_heading: true + heading_level: 2 diff --git a/docs/user_guidance/entity_relationships.md b/docs/user_guidance/entity_relationships.md new file mode 100644 index 0000000..bf0f6c1 --- /dev/null +++ b/docs/user_guidance/entity_relationships.md @@ -0,0 +1,63 @@ +--- +title: Entity Relationships +tags: + - Linkage + - Relationships + - Missing + - Parent + - Group + - Rejections +--- + +Sometimes a user may choose to use the file transformation stage to `normalise` a heavily nested dataset into separate entities during the initial reading of data. This would be done by specifying different entities in the dataset section of the contract configuration in the `dischema` file. This allows for easier interaction when customising errors in the data contract or writing transformations in the business rules. + +`Normalising` assets can lead to more complex validations being required. For example in the flights dataset: + +```mermaid +erDiagram + COUNTRY ||--|{ AIRPORT : "" + AIRPORT ||--o{ FLIGHT : "" + FLIGHT ||--o{ PASSENGER : "" + AIRPORT ||--|{ STAFF_MEMBER : "" +``` + +### Missing Parent Records + +It could be that an airport record is deemed invalid and removed. Due to this, any flight records that linked to the now removed airport record are themselves invalid - a situation we refer to as a `missing_parent` issue, but are now existing in an entirely different entity. + +### No Valid Mandatory Records + +It could also be the case that staff records are a mandatory field for airport records. If all staff records for a particular airport record are removed during validation, this itself would invalidate the airport record - a situation we refer to as `no_valid_records` issue - but again the invalid airport record is in a different entity. + +### Dischema + +In order to perform these validations, how to link normalised entities needs to be provided. This can be specified in the `entity_relationships` section of the `dischema`. + +## Entity Relationships Content + +To allow the DVE to link between normalised assets, the following information should be provided (per linkable entity): + +- `parent_entity`: the immediate parent of the entity +- `join_fields`: how to join the entity with its parent in dictionary form (parent_field_name: child_field_name) +- `is_root_entity`: indicates that the entity is a root node in a hierarchical model +- `mandatory`: whether the child entity is a mandatory field in the immediate parent + +There is also the functionality to customise errors related to either missing parent or group rejections: + +- `missing_parent_id_error_code`: the error code to display if a record is rejected as it has no valid parent record +- `missing_parent_id_error_message`: the error message to display if a record is rejected as it has no valid parent record +- `no_valid_records_error_code`: the error code to display if parent records are removed due to no valid children in a mandatory field +- `no_valid_records_error_message`: the error message to display if parent records are removed due to no valid children in a mandatory field +- `empty_entity_error_code`: the error code to display if a __mandatory__ entity has any records post filtering +- `empty_entity_error_message`: the error message to display if a __mandatory__ entity has any records post filtering + +!!! note + Specifying root entities is __optional__. Root entities will be inferred based on their absence. + You may wish to specify root entities so that error codes and messages can be customised (e.g. empty entity). + + When specifying root entities you should ensure that parent_entity and join_fields values are left blank. + +## Entity Hierarchy Object + +The details provided in the entity_relationships section of the dischema are used to create an EntityHierarchy object. +Please refer to [Advanced User Guidance: Entity Hierarchy](../advanced_guidance/package_documentation/entity_hierarchy.md). diff --git a/docs/user_guidance/getting_started.md b/docs/user_guidance/getting_started.md index e938c2d..fbda5ae 100644 --- a/docs/user_guidance/getting_started.md +++ b/docs/user_guidance/getting_started.md @@ -68,6 +68,8 @@ Within the example above, there are two parent keys - `schemas` and `datasets`. !!! note The "splitting" of entities is considerably more useful in situtations where you want to normalise/de-normalise your data. If you're unfamiliar with this concept, you can read more about it [here](https://en.wikipedia.org/wiki/Database_normalization). However, you should keep in mind potential performance impacts of doing this. If you have rules that requires fields from different entities, you will have to perform a `join` between the split entities to be able to perform the rule. +To support with the application of more complex validation relating to parent and child records within normalised data, the [entity_relationships](entity_relationships.md) section of the `dischema` enables users to specify parent-child relationships and to customise error codes related to missing parent and group level validation issues. + For each dataset definition, you will need to provide a `reader_config` which describes how to load the data during the [File Transformation](file_transformation.md) stage. So, in the example above, we expect `movies` to come in as a `JSON` file. However, you can add more readers if you have the same data in different data formats (e.g. `csv`, `xml`, `json`). Regardless of what file format, the [File Transformation](file_transformation.md) stage will convert the submitted data into a "stringified" parquet format which is a requirement for the subsequent stages. To learn more about how you can construct your Data Contract please read [here](data_contract.md). diff --git a/docs/user_guidance/install.md b/docs/user_guidance/install.md index 85186cd..2c86b12 100644 --- a/docs/user_guidance/install.md +++ b/docs/user_guidance/install.md @@ -78,11 +78,12 @@ Once you have installed the DVE you are almost ready to use it. To be able to ru ## DVE Version Compatability Matrix -| DVE Version | Python Version | DuckDB Version | Spark Version | Pydantic Version | -| ------------ | -------------- | -------------- | --------------- | ---------------- | -| >=0.9.0 | >=3.10,<3.13 | 1.1.3 | >=3.5.0,<=3.5.5 | 2.13.4 | -| >=0.8.0 | >=3.10,<3.13 | 1.1.3 | 3.5.2 | 1.10.19 | -| >=0.7.2 | >=3.10,<3.12 | 1.1.* | 3.4.* | 1.10.16 | -| >=0.6 | >=3.10,<3.12 | 1.1.* | 3.4.* | 1.10.15 | -| >=0.2,<0.6 | >=3.10,<3.12 | 1.1.0 | 3.4.4 | 1.10.15 | -| 0.1 | >=3.7.2,<3.8 | 1.1.0 | 3.2.1 | 1.10.15 | +| DVE Version | Python Version | DuckDB Version | Spark Version | Pydantic Version | +| ------------ | -------------- | ---------------- | --------------- | ---------------- | +| >=0.10.0 | >=3.10,<1.13 | __>=1.4,<1.4.5__ | >=3.5.0,<=3.5.5 | 2.13.4 | +| >=0.9.0 | >=3.10,<3.13 | 1.1.3 | >=3.5.0,<=3.5.5 | __2.13.4__ | +| >=0.8.0 | >=3.10,<3.13 | __1.1.3__ | __3.5.2__ | 1.10.19 | +| >=0.7.2 | >=3.10,<3.12 | 1.1.* | 3.4.* | __1.10.16__ | +| >=0.6 | >=3.10,<3.12 | __1.1.*__ | __3.4.*__ | 1.10.15 | +| >=0.2,<0.6 | __>=3.10,<3.12__ | 1.1.0 | 3.4.4 | 1.10.15 | +| 0.1 | >=3.7.2,<3.8 | 1.1.0 | 3.2.1 | 1.10.15 | diff --git a/poetry.lock b/poetry.lock index 0fedfd4..18f789e 100644 --- a/poetry.lock +++ b/poetry.lock @@ -1144,66 +1144,58 @@ files = [ [[package]] name = "duckdb" -version = "1.1.3" +version = "1.4.4" description = "DuckDB in-process database" optional = false -python-versions = ">=3.7.0" +python-versions = ">=3.9.0" groups = ["main"] files = [ - {file = "duckdb-1.1.3-cp310-cp310-macosx_12_0_arm64.whl", hash = "sha256:1c0226dc43e2ee4cc3a5a4672fddb2d76fd2cf2694443f395c02dd1bea0b7fce"}, - {file = "duckdb-1.1.3-cp310-cp310-macosx_12_0_universal2.whl", hash = "sha256:7c71169fa804c0b65e49afe423ddc2dc83e198640e3b041028da8110f7cd16f7"}, - {file = "duckdb-1.1.3-cp310-cp310-macosx_12_0_x86_64.whl", hash = "sha256:872d38b65b66e3219d2400c732585c5b4d11b13d7a36cd97908d7981526e9898"}, - {file = "duckdb-1.1.3-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:25fb02629418c0d4d94a2bc1776edaa33f6f6ccaa00bd84eb96ecb97ae4b50e9"}, - {file = "duckdb-1.1.3-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9e3f5cd604e7c39527e6060f430769b72234345baaa0987f9500988b2814f5e4"}, - {file = "duckdb-1.1.3-cp310-cp310-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:08935700e49c187fe0e9b2b86b5aad8a2ccd661069053e38bfaed3b9ff795efd"}, - {file = "duckdb-1.1.3-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:f9b47036945e1db32d70e414a10b1593aec641bd4c5e2056873d971cc21e978b"}, - {file = "duckdb-1.1.3-cp310-cp310-win_amd64.whl", hash = "sha256:35c420f58abc79a68a286a20fd6265636175fadeca1ce964fc8ef159f3acc289"}, - {file = "duckdb-1.1.3-cp311-cp311-macosx_12_0_arm64.whl", hash = "sha256:4f0e2e5a6f5a53b79aee20856c027046fba1d73ada6178ed8467f53c3877d5e0"}, - {file = "duckdb-1.1.3-cp311-cp311-macosx_12_0_universal2.whl", hash = "sha256:911d58c22645bfca4a5a049ff53a0afd1537bc18fedb13bc440b2e5af3c46148"}, - {file = "duckdb-1.1.3-cp311-cp311-macosx_12_0_x86_64.whl", hash = "sha256:c443d3d502335e69fc1e35295fcfd1108f72cb984af54c536adfd7875e79cee5"}, - {file = "duckdb-1.1.3-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:0a55169d2d2e2e88077d91d4875104b58de45eff6a17a59c7dc41562c73df4be"}, - {file = "duckdb-1.1.3-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9d0767ada9f06faa5afcf63eb7ba1befaccfbcfdac5ff86f0168c673dd1f47aa"}, - {file = "duckdb-1.1.3-cp311-cp311-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:51c6d79e05b4a0933672b1cacd6338f882158f45ef9903aef350c4427d9fc898"}, - {file = "duckdb-1.1.3-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:183ac743f21c6a4d6adfd02b69013d5fd78e5e2cd2b4db023bc8a95457d4bc5d"}, - {file = "duckdb-1.1.3-cp311-cp311-win_amd64.whl", hash = "sha256:a30dd599b8090ea6eafdfb5a9f1b872d78bac318b6914ada2d35c7974d643640"}, - {file = "duckdb-1.1.3-cp312-cp312-macosx_12_0_arm64.whl", hash = "sha256:a433ae9e72c5f397c44abdaa3c781d94f94f4065bcbf99ecd39433058c64cb38"}, - {file = "duckdb-1.1.3-cp312-cp312-macosx_12_0_universal2.whl", hash = "sha256:d08308e0a46c748d9c30f1d67ee1143e9c5ea3fbcccc27a47e115b19e7e78aa9"}, - {file = "duckdb-1.1.3-cp312-cp312-macosx_12_0_x86_64.whl", hash = "sha256:5d57776539211e79b11e94f2f6d63de77885f23f14982e0fac066f2885fcf3ff"}, - {file = "duckdb-1.1.3-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:e59087dbbb63705f2483544e01cccf07d5b35afa58be8931b224f3221361d537"}, - {file = "duckdb-1.1.3-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:4ebf5f60ddbd65c13e77cddb85fe4af671d31b851f125a4d002a313696af43f1"}, - {file = "duckdb-1.1.3-cp312-cp312-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:e4ef7ba97a65bd39d66f2a7080e6fb60e7c3e41d4c1e19245f90f53b98e3ac32"}, - {file = "duckdb-1.1.3-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:f58db1b65593ff796c8ea6e63e2e144c944dd3d51c8d8e40dffa7f41693d35d3"}, - {file = "duckdb-1.1.3-cp312-cp312-win_amd64.whl", hash = "sha256:e86006958e84c5c02f08f9b96f4bc26990514eab329b1b4f71049b3727ce5989"}, - {file = "duckdb-1.1.3-cp313-cp313-macosx_12_0_arm64.whl", hash = "sha256:0897f83c09356206ce462f62157ce064961a5348e31ccb2a557a7531d814e70e"}, - {file = "duckdb-1.1.3-cp313-cp313-macosx_12_0_universal2.whl", hash = "sha256:cddc6c1a3b91dcc5f32493231b3ba98f51e6d3a44fe02839556db2b928087378"}, - {file = "duckdb-1.1.3-cp313-cp313-macosx_12_0_x86_64.whl", hash = "sha256:1d9ab6143e73bcf17d62566e368c23f28aa544feddfd2d8eb50ef21034286f24"}, - {file = "duckdb-1.1.3-cp313-cp313-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:2f073d15d11a328f2e6d5964a704517e818e930800b7f3fa83adea47f23720d3"}, - {file = "duckdb-1.1.3-cp313-cp313-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:d5724fd8a49e24d730be34846b814b98ba7c304ca904fbdc98b47fa95c0b0cee"}, - {file = "duckdb-1.1.3-cp313-cp313-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:51e7dbd968b393343b226ab3f3a7b5a68dee6d3fe59be9d802383bf916775cb8"}, - {file = "duckdb-1.1.3-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:00cca22df96aa3473fe4584f84888e2cf1c516e8c2dd837210daec44eadba586"}, - {file = "duckdb-1.1.3-cp313-cp313-win_amd64.whl", hash = "sha256:77f26884c7b807c7edd07f95cf0b00e6d47f0de4a534ac1706a58f8bc70d0d31"}, - {file = "duckdb-1.1.3-cp37-cp37m-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:a4748635875fc3c19a7320a6ae7410f9295557450c0ebab6d6712de12640929a"}, - {file = "duckdb-1.1.3-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:b74e121ab65dbec5290f33ca92301e3a4e81797966c8d9feef6efdf05fc6dafd"}, - {file = "duckdb-1.1.3-cp37-cp37m-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:9c619e4849837c8c83666f2cd5c6c031300cd2601e9564b47aa5de458ff6e69d"}, - {file = "duckdb-1.1.3-cp37-cp37m-win_amd64.whl", hash = "sha256:0ba6baa0af33ded836b388b09433a69b8bec00263247f6bf0a05c65c897108d3"}, - {file = "duckdb-1.1.3-cp38-cp38-macosx_12_0_arm64.whl", hash = "sha256:ecb1dc9062c1cc4d2d88a5e5cd8cc72af7818ab5a3c0f796ef0ffd60cfd3efb4"}, - {file = "duckdb-1.1.3-cp38-cp38-macosx_12_0_universal2.whl", hash = "sha256:5ace6e4b1873afdd38bd6cc8fcf90310fb2d454f29c39a61d0c0cf1a24ad6c8d"}, - {file = "duckdb-1.1.3-cp38-cp38-macosx_12_0_x86_64.whl", hash = "sha256:a1fa0c502f257fa9caca60b8b1478ec0f3295f34bb2efdc10776fc731b8a6c5f"}, - {file = "duckdb-1.1.3-cp38-cp38-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:6411e21a2128d478efbd023f2bdff12464d146f92bc3e9c49247240448ace5a6"}, - {file = "duckdb-1.1.3-cp38-cp38-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:c5336939d83837af52731e02b6a78a446794078590aa71fd400eb17f083dda3e"}, - {file = "duckdb-1.1.3-cp38-cp38-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f549af9f7416573ee48db1cf8c9d27aeed245cb015f4b4f975289418c6cf7320"}, - {file = "duckdb-1.1.3-cp38-cp38-win_amd64.whl", hash = "sha256:2141c6b28162199999075d6031b5d63efeb97c1e68fb3d797279d31c65676269"}, - {file = "duckdb-1.1.3-cp39-cp39-macosx_12_0_arm64.whl", hash = "sha256:09c68522c30fc38fc972b8a75e9201616b96ae6da3444585f14cf0d116008c95"}, - {file = "duckdb-1.1.3-cp39-cp39-macosx_12_0_universal2.whl", hash = "sha256:8ee97ec337794c162c0638dda3b4a30a483d0587deda22d45e1909036ff0b739"}, - {file = "duckdb-1.1.3-cp39-cp39-macosx_12_0_x86_64.whl", hash = "sha256:a1f83c7217c188b7ab42e6a0963f42070d9aed114f6200e3c923c8899c090f16"}, - {file = "duckdb-1.1.3-cp39-cp39-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:1aa3abec8e8995a03ff1a904b0e66282d19919f562dd0a1de02f23169eeec461"}, - {file = "duckdb-1.1.3-cp39-cp39-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:80158f4c7c7ada46245837d5b6869a336bbaa28436fbb0537663fa324a2750cd"}, - {file = "duckdb-1.1.3-cp39-cp39-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:647f17bd126170d96a38a9a6f25fca47ebb0261e5e44881e3782989033c94686"}, - {file = "duckdb-1.1.3-cp39-cp39-musllinux_1_2_x86_64.whl", hash = "sha256:252d9b17d354beb9057098d4e5d5698e091a4f4a0d38157daeea5fc0ec161670"}, - {file = "duckdb-1.1.3-cp39-cp39-win_amd64.whl", hash = "sha256:eeacb598120040e9591f5a4edecad7080853aa8ac27e62d280f151f8c862afa3"}, - {file = "duckdb-1.1.3.tar.gz", hash = "sha256:68c3a46ab08836fe041d15dcbf838f74a990d551db47cb24ab1c4576fc19351c"}, + {file = "duckdb-1.4.4-cp310-cp310-macosx_10_9_universal2.whl", hash = "sha256:e870a441cb1c41d556205deb665749f26347ed13b3a247b53714f5d589596977"}, + {file = "duckdb-1.4.4-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:49123b579e4a6323e65139210cd72dddc593a72d840211556b60f9703bda8526"}, + {file = "duckdb-1.4.4-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:5e1933fac5293fea5926b0ee75a55b8cfe7f516d867310a5b251831ab61fe62b"}, + {file = "duckdb-1.4.4-cp310-cp310-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:707530f6637e91dc4b8125260595299ec9dd157c09f5d16c4186c5988bfbd09a"}, + {file = "duckdb-1.4.4-cp310-cp310-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:453b115f4777467f35103d8081770ac2f223fb5799178db5b06186e3ab51d1f2"}, + {file = "duckdb-1.4.4-cp310-cp310-win_amd64.whl", hash = "sha256:a3c8542db7ffb128aceb7f3b35502ebaddcd4f73f1227569306cc34bad06680c"}, + {file = "duckdb-1.4.4-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:5ba684f498d4e924c7e8f30dd157da8da34c8479746c5011b6c0e037e9c60ad2"}, + {file = "duckdb-1.4.4-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:5536eb952a8aa6ae56469362e344d4e6403cc945a80bc8c5c2ebdd85d85eb64b"}, + {file = "duckdb-1.4.4-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:47dd4162da6a2be59a0aef640eb08d6360df1cf83c317dcc127836daaf3b7f7c"}, + {file = "duckdb-1.4.4-cp311-cp311-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:6cb357cfa3403910e79e2eb46c8e445bb1ee2fd62e9e9588c6b999df4256abc1"}, + {file = "duckdb-1.4.4-cp311-cp311-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:4c25d5b0febda02b7944e94fdae95aecf952797afc8cb920f677b46a7c251955"}, + {file = "duckdb-1.4.4-cp311-cp311-win_amd64.whl", hash = "sha256:6703dd1bb650025b3771552333d305d62ddd7ff182de121483d4e042ea6e2e00"}, + {file = "duckdb-1.4.4-cp311-cp311-win_arm64.whl", hash = "sha256:bf138201f56e5d6fc276a25138341b3523e2f84733613fc43f02c54465619a95"}, + {file = "duckdb-1.4.4-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:ddcfd9c6ff234da603a1edd5fd8ae6107f4d042f74951b65f91bc5e2643856b3"}, + {file = "duckdb-1.4.4-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:6792ca647216bd5c4ff16396e4591cfa9b4a72e5ad7cdd312cec6d67e8431a7c"}, + {file = "duckdb-1.4.4-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:1f8d55843cc940e36261689054f7dfb6ce35b1f5b0953b0d355b6adb654b0d52"}, + {file = "duckdb-1.4.4-cp312-cp312-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:c65d15c440c31e06baaebfd2c06d71ce877e132779d309f1edf0a85d23c07e92"}, + {file = "duckdb-1.4.4-cp312-cp312-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:b297eff642503fd435a9de5a9cb7db4eccb6f61d61a55b30d2636023f149855f"}, + {file = "duckdb-1.4.4-cp312-cp312-win_amd64.whl", hash = "sha256:d525de5f282b03aa8be6db86b1abffdceae5f1055113a03d5b50cd2fb8cf2ef8"}, + {file = "duckdb-1.4.4-cp312-cp312-win_arm64.whl", hash = "sha256:50f2eb173c573811b44aba51176da7a4e5c487113982be6a6a1c37337ec5fa57"}, + {file = "duckdb-1.4.4-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:337f8b24e89bc2e12dadcfe87b4eb1c00fd920f68ab07bc9b70960d6523b8bc3"}, + {file = "duckdb-1.4.4-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:0509b39ea7af8cff0198a99d206dca753c62844adab54e545984c2e2c1381616"}, + {file = "duckdb-1.4.4-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:fb94de6d023de9d79b7edc1ae07ee1d0b4f5fa8a9dcec799650b5befdf7aafec"}, + {file = "duckdb-1.4.4-cp313-cp313-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:0d636ceda422e7babd5e2f7275f6a0d1a3405e6a01873f00d38b72118d30c10b"}, + {file = "duckdb-1.4.4-cp313-cp313-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:7df7351328ffb812a4a289732f500d621e7de9942a3a2c9b6d4afcf4c0e72526"}, + {file = "duckdb-1.4.4-cp313-cp313-win_amd64.whl", hash = "sha256:6fb1225a9ea5877421481d59a6c556a9532c32c16c7ae6ca8d127e2b878c9389"}, + {file = "duckdb-1.4.4-cp313-cp313-win_arm64.whl", hash = "sha256:f28a18cc790217e5b347bb91b2cab27aafc557c58d3d8382e04b4fe55d0c3f66"}, + {file = "duckdb-1.4.4-cp314-cp314-macosx_10_15_universal2.whl", hash = "sha256:25874f8b1355e96178079e37312c3ba6d61a2354f51319dae860cf21335c3a20"}, + {file = "duckdb-1.4.4-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:452c5b5d6c349dc5d1154eb2062ee547296fcbd0c20e9df1ed00b5e1809089da"}, + {file = "duckdb-1.4.4-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:8e5c2d8a0452df55e092959c0bfc8ab8897ac3ea0f754cb3b0ab3e165cd79aff"}, + {file = "duckdb-1.4.4-cp314-cp314-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:1af6e76fe8bd24875dc56dd8e38300d64dc708cd2e772f67b9fbc635cc3066a3"}, + {file = "duckdb-1.4.4-cp314-cp314-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:d0440f59e0cd9936a9ebfcf7a13312eda480c79214ffed3878d75947fc3b7d6d"}, + {file = "duckdb-1.4.4-cp314-cp314-win_amd64.whl", hash = "sha256:59c8d76016dde854beab844935b1ec31de358d4053e792988108e995b18c08e7"}, + {file = "duckdb-1.4.4-cp314-cp314-win_arm64.whl", hash = "sha256:53cd6423136ab44383ec9955aefe7599b3fb3dd1fe006161e6396d8167e0e0d4"}, + {file = "duckdb-1.4.4-cp39-cp39-macosx_10_9_universal2.whl", hash = "sha256:8097201bc5fd0779d7fcc2f3f4736c349197235f4cb7171622936343a1aa8dbf"}, + {file = "duckdb-1.4.4-cp39-cp39-macosx_10_9_x86_64.whl", hash = "sha256:cd1be3d48577f5b40eb9706c6b2ae10edfe18e78eb28e31a3b922dcff1183597"}, + {file = "duckdb-1.4.4-cp39-cp39-macosx_11_0_arm64.whl", hash = "sha256:e041f2fbd6888da090eca96ac167a7eb62d02f778385dd9155ed859f1c6b6dc8"}, + {file = "duckdb-1.4.4-cp39-cp39-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:7eec0bf271ac622e57b7f6554a27a6e7d1dd2f43d1871f7962c74bcbbede15ba"}, + {file = "duckdb-1.4.4-cp39-cp39-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:5cdc4126ec925edf3112bc656ac9ed23745294b854935fa7a643a216e4455af6"}, + {file = "duckdb-1.4.4-cp39-cp39-win_amd64.whl", hash = "sha256:c9566a4ed834ec7999db5849f53da0a7ee83d86830c33f471bf0211a1148ca12"}, + {file = "duckdb-1.4.4.tar.gz", hash = "sha256:8bba52fd2acb67668a4615ee17ee51814124223de836d9e2fdcbc4c9021b3d3c"}, ] +[package.extras] +all = ["adbc-driver-manager", "fsspec", "ipython", "numpy", "pandas", "pyarrow"] + [[package]] name = "et-xmlfile" version = "2.0.0" @@ -3595,25 +3587,22 @@ test = ["pytest", "pytest-cov"] [[package]] name = "zensical" -version = "0.0.46" +version = "0.0.63" description = "A modern static site generator built by the creators of Material for MkDocs" optional = false python-versions = ">=3.10" groups = ["docs"] files = [ - {file = "zensical-0.0.46-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:d91af81ab058c8693dfd75f2f77b4c73bcba4125681d1d276f38624291820bd2"}, - {file = "zensical-0.0.46-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:d9221264a9a87409900a47e29985607b0c9245dacb89077e87c8e16e31edc167"}, - {file = "zensical-0.0.46-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:ec43018d5343ca2e1d71aa352eeddd560fef504effd03025840a5a783abefa4f"}, - {file = "zensical-0.0.46-cp310-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:26e98fb8ab7ab50cdd20a73e2c7d4d9aae0b46cf2d8691e6bb22f9c261b8a60a"}, - {file = "zensical-0.0.46-cp310-abi3-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:46fe578f26963f8ee89567983e62737b6fadc9197d4742e1020b522e092d7baa"}, - {file = "zensical-0.0.46-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:aef03fa186a5589148e10b62610500989c6b075a2c08e1554233adbf91b2a3dc"}, - {file = "zensical-0.0.46-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:bc7446cdf97a8dea390f20ed2bd6b030cddc1bd36a8ce113ea3efef6fa61c573"}, - {file = "zensical-0.0.46-cp310-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:bbee37801f1ed500f158dc0992c569282950f780ae353c37fe6969f99983d701"}, - {file = "zensical-0.0.46-cp310-abi3-musllinux_1_2_i686.whl", hash = "sha256:9487c147c9cceb50c04d0ad70b024821a6eab1629dafd70ab6d1e86ec841e623"}, - {file = "zensical-0.0.46-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:f42a4683c762f026878d19ede4bcf7bfbb84dbecb5ad923949abb77806ed88a5"}, - {file = "zensical-0.0.46-cp310-abi3-win32.whl", hash = "sha256:85f018f2a7ee76a83915c87ddb12b58cf343fd6154081d33ac95b6751b011dd7"}, - {file = "zensical-0.0.46-cp310-abi3-win_amd64.whl", hash = "sha256:1543a693a160de60e86ca589592401b584670e7e12c5ae30e3c2ba76786f7ec3"}, - {file = "zensical-0.0.46.tar.gz", hash = "sha256:3ec21f4fb1e78cd7c0d6b07ae336b04770e27ba020dabc457b2790e5d34f1978"}, + {file = "zensical-0.0.63-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:069af2ee0254eaa7099faad908f0ae0017c151ef0160b21a039d08efaa61b38a"}, + {file = "zensical-0.0.63-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:6ec02283d946569f68fdd1ab4a1dd2de560c0ad540874d5ecd4a4f1e96bdbe1d"}, + {file = "zensical-0.0.63-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:af2711ff1397338c730cc9232f687066da7181751257f317877501b1b77a44ea"}, + {file = "zensical-0.0.63-cp310-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:f1afb0e2819a588d956c1684c90e23fc0c7efc89e1ec141403a9fb41b75b1993"}, + {file = "zensical-0.0.63-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:925b3dd8bb6812b780f5e00828c42bd1eebd675b1eb6b4a58135926bb716bdde"}, + {file = "zensical-0.0.63-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:6365bce465a7f6755533eaa80aa20c8a48d1b39fa7b12cd394eece3921e2c530"}, + {file = "zensical-0.0.63-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:63706419125f4c46461b322c8aaed1bbc3b917dd389ecf29601d713fc962bfb6"}, + {file = "zensical-0.0.63-cp310-abi3-win_amd64.whl", hash = "sha256:43e223b8d6ad1f08a772232926c123a000abd30ffe98d9db63a2731ec6275329"}, + {file = "zensical-0.0.63-cp310-abi3-win_arm64.whl", hash = "sha256:73c29ca1cd00243384fdc69414fac1c4367814203116acc9d198121b7a91012a"}, + {file = "zensical-0.0.63.tar.gz", hash = "sha256:95f68b494fa6a11a59065f7965d27acfca404dd867f81d31295b5ab7172f2f8a"}, ] [package.dependencies] @@ -3622,7 +3611,7 @@ deepmerge = ">=2.0" jinja2 = ">=3.1" markdown = ">=3.7" pygments = ">=2.20" -pymdown-extensions = ">=10.21.3" +pymdown-extensions = ">=11.0" pyyaml = ">=6.0.2" tomli = ">=2.4.0" @@ -3649,4 +3638,4 @@ type = ["pytest-mypy (>=1.0.1) ; platform_python_implementation != \"PyPy\""] [metadata] lock-version = "2.1" python-versions = ">=3.10,<3.13" -content-hash = "3c6b964ad86fe375ec1862480b207189085a8d1da8e5017a8545b78dc0ff469b" +content-hash = "6d996888c9d149407fb885b454745bf763f9cba56404a29148ffe0b9fd57b0de" diff --git a/pyproject.toml b/pyproject.toml index 0ad62d3..5cac028 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -7,6 +7,7 @@ authors = [ ] readme = "README.md" classifiers = [ + "Development Status :: 4 - Beta", "Programming Language :: Python :: 3", "Programming Language :: Python :: 3.10", "Programming Language :: Python :: 3.11", @@ -35,7 +36,7 @@ python = ">=3.10,<3.13" # breaking changes beyond 3.12 boto3 = ">=1.34.162,<1.36" # breaking change beyond 1.36 botocore = ">=1.34.162,<1.36" # breaking change beyond 1.36 delta-spark = ">=3.0.0,<=3.2.0" -duckdb = "1.1.3" # breaking changes beyond 1.1 +duckdb = ">=1.4,<1.4.5" Jinja2 = "3.1.6" lxml = "6.1.1" numpy = "1.26.4" @@ -108,7 +109,7 @@ mkdocs = "1.6.1" mkdocstrings = { version = "1.0.3", extras = ["python"] } griffelib = "2.0.1" pymdown-extensions = "11.0.1" -zensical = "0.0.46" +zensical = "0.0.63" [tool.ruff] line-length = 100 diff --git a/src/dve/common/error_utils.py b/src/dve/common/error_utils.py index 120c902..23ad714 100644 --- a/src/dve/common/error_utils.py +++ b/src/dve/common/error_utils.py @@ -5,7 +5,7 @@ import logging from collections.abc import Iterable from itertools import chain -from multiprocessing import Queue +from queue import Queue from threading import Thread from typing import Optional, Union diff --git a/src/dve/core_engine/backends/base/contract.py b/src/dve/core_engine/backends/base/contract.py index 948ff77..003f0b4 100644 --- a/src/dve/core_engine/backends/base/contract.py +++ b/src/dve/core_engine/backends/base/contract.py @@ -443,10 +443,6 @@ def apply( ], ) - if contract_metadata.cache_originals: - for entity_name in list(entities): - entities[f"Original{entity_name}"] = entities[entity_name] - return entities, feedback_errors_uri, successful, processing_errors_uri def read_parquet(self, path: URI, **kwargs) -> EntityType: diff --git a/src/dve/core_engine/backends/base/reader.py b/src/dve/core_engine/backends/base/reader.py index ae0e99f..ff3586c 100644 --- a/src/dve/core_engine/backends/base/reader.py +++ b/src/dve/core_engine/backends/base/reader.py @@ -10,6 +10,10 @@ from dve.core_engine.backends.exceptions import MessageBearingError, ReaderLacksEntityTypeSupport from dve.core_engine.backends.types import EntityName, EntityType +from dve.core_engine.configuration.v1 import ( + AllowedAdditionalReaderChecks, + _ReaderAdditionalChecksConfig, +) from dve.core_engine.message import FeedbackMessage from dve.core_engine.type_hints import URI, ArbitraryFunction, WrapDecorator from dve.parser.file_handling.service import open_stream @@ -109,7 +113,10 @@ def read_to_entity_type( entity_name: EntityName, schema: type[BaseModel], all_model_fields: Optional[set[str]] = None, - ) -> EntityType: + additional_checks: Optional[ + dict[AllowedAdditionalReaderChecks, _ReaderAdditionalChecksConfig] + ] = None, + ): """Read to the specified entity type, if supported. NOTE: Simple types should either be returned as strings (if present) or @@ -117,21 +124,47 @@ def read_to_entity_type( data contract. """ - if entity_name == Iterator[dict[str, Any]]: - return self.read_to_py_iterator( + additional_checks = additional_checks or {} + + self.raise_if_not_sensible_file(resource, entity_name) + + if entity_type == Iterator[dict[str, Any]]: + entity = self.read_to_py_iterator( resource, entity_name, schema, all_model_fields # type: ignore ) - self.raise_if_not_sensible_file(resource, entity_name) + else: - try: - reader_func = self.__read_methods__[entity_type] - except KeyError as err: - raise ReaderLacksEntityTypeSupport(entity_type=entity_type) from err + try: + reader_func = self.__read_methods__[entity_type] + except KeyError as err: + raise ReaderLacksEntityTypeSupport(entity_type=entity_type) from err - return reader_func( - self, resource, entity_name, schema, all_model_fields=all_model_fields # type: ignore - ) + entity = reader_func( + self, + resource, + entity_name, + schema, + all_model_fields=all_model_fields, # type: ignore + ) + + if config := additional_checks.get("check_empty"): + if self.check_entity_empty(entity): + raise MessageBearingError( + f"The mandatory entity {entity_name} is empty", + messages=[ + FeedbackMessage( + entity=entity_name, + record=None, + failure_type="submission", + error_location=entity_name, + error_code=config.error_code, + error_message=config.error_message, + ) + ], + ) + + return entity def add_record_index(self, entity: EntityType, **kwargs) -> EntityType: """Add a record index to the entity""" @@ -141,6 +174,10 @@ def drop_record_index(self, entity: EntityType, **kwargs) -> EntityType: """Drop a record index to the entity""" raise NotImplementedError(f"drop_record_index not implemented in {self.__class__}") + def check_entity_empty(self, entity: EntityType) -> bool: + """Determine if the entity supplied is empty""" + raise NotImplementedError(f"check_entity_empty not implemented in {self.__class__}") + def write_parquet( self, entity: EntityType, diff --git a/src/dve/core_engine/backends/base/rules.py b/src/dve/core_engine/backends/base/rules.py index 9b6b4fe..dbe2d9d 100644 --- a/src/dve/core_engine/backends/base/rules.py +++ b/src/dve/core_engine/backends/base/rules.py @@ -3,7 +3,7 @@ import logging from abc import ABC, abstractmethod from collections import defaultdict -from collections.abc import Iterable +from collections.abc import Iterable, Iterator from typing import Any, ClassVar, Generic, NoReturn, Optional, TypeVar from uuid import uuid4 @@ -27,6 +27,7 @@ CopyEntity, DeferredFilter, EntityRemoval, + GroupIdentification, HeaderJoin, ImmediateFilter, InnerJoin, @@ -43,8 +44,11 @@ TableUnion, ) from dve.core_engine.backends.types import Entities, EntityType, StageSuccessful +from dve.core_engine.configuration.v1.hierarchy import EntityHierarchy, HierarchyNode from dve.core_engine.exceptions import CriticalProcessingError from dve.core_engine.loggers import get_logger +from dve.core_engine.message import FeedbackMessage +from dve.core_engine.templating import template_object from dve.core_engine.type_hints import URI, DVEStageName, EntityName, Messages, TemplateVariables T_contra = TypeVar("T_contra", bound=AbstractStep, contravariant=True) @@ -307,7 +311,10 @@ def join_header(self, entities: Entities, *, config: HeaderJoin) -> Messages: """ raise NotImplementedError - def identify_orphans(self, entities: Entities, *, config: OrphanIdentification) -> Messages: + @abstractmethod + def identify_orphans( + self, entities: Entities, *, config: OrphanIdentification + ) -> Iterable: """Identify records in an entity which don't have at least one corresponding match in the target. A new boolean column will be added to `entity` ('IsOrphaned') indicating whether the condition matched. @@ -320,6 +327,14 @@ def identify_orphans(self, entities: Entities, *, config: OrphanIdentification) """ raise NotImplementedError + @abstractmethod + def check_mandatory_group(self, entities: Entities, *, config: GroupIdentification) -> Iterator: + """ + Check that a mandatory key in an entity has at least one valid entry in the all the child + entities. + """ + raise NotImplementedError + @abstractmethod def union(self, entities: Entities, *, config: TableUnion) -> Messages: """Union two entities together, taking the columns from each by name. @@ -352,6 +367,145 @@ def notify(self, entities: Entities, *, config: Notification) -> Messages: """ + def identify_and_remove_orphans( + self, + working_directory: URI, + entities: Entities, + entity_hierarchy: EntityHierarchy, + key_fields: Optional[dict[str, list[str]]] = None, + ) -> tuple[Messages, dict[EntityName, bool]]: + """ + Identifies and removes orphan records by traversing the EntityHierarchy object. + An orphan is a child record whose parent FK does not exist in the parent entity. + Processes recursively: removes orphans at each level, then processes children. + """ + + def process_node(node: HierarchyNode) -> bool: + """Identify orphans and remove in a given node""" + issues_found: bool = False + if node.parent_entity is None: + return issues_found + + self.logger.info(f"Checking for orphan records in {node.entity_name}") + + join_expr = " AND ".join( + f"{node.parent_entity}.{k} = {node.entity_name}.{v}" + for k, v in node.join_fields.items() + ) + location = list(node.join_fields.values())[0] + with BackgroundMessageWriter( + working_directory=working_directory, + dve_stage=self.__stage_name__, + key_fields=key_fields, + logger=self.logger, + ) as msg_writer: + _orph_records = self.identify_orphans( + entities=entities, + config=OrphanIdentification( + id=list(node.join_fields.values())[0], + entity_name=node.entity_name, + target_name=node.parent_entity, + join_condition=join_expr, + ), + ) + _messages = [ + FeedbackMessage( + entity=node.entity_name, + record=record, # type: ignore + error_location=location, + error_message=template_object( + node.missing_parent_id_error_message, record + ), + failure_type="record", + error_type="record", + error_code=node.missing_parent_id_error_code, + reporting_field=location, + category="Parent Missing", + ) + for record in _orph_records + ] + msg_writer.write_queue.put(_messages) + + return len(_messages) > 0 + + entity_issues_found: dict[EntityName, bool] = {} + + for tree in entity_hierarchy.entity_trees.values(): + for node in tree.iterate_root_down(): + entity_issues_found[node.entity_name] = process_node(node) + + entities.update(entities) + + return [], entity_issues_found + + def identify_and_remove_missing_mandatory_groups( + self, + working_directory: URI, + entities: Entities, + entity_hierarchy: EntityHierarchy, + key_fields: Optional[dict[str, list[str]]] = None, + ) -> tuple[Messages, dict[EntityName, bool]]: + """ + Identify that an entity with a mandatory key has at least one valid child record. + """ + + def process_node(node: HierarchyNode) -> bool: + """Identify at least one valid child for a mandatory entity at a given node.""" + if node.parent_entity is None or not node.mandatory: + return False + + self.logger.info( + f"Identifying that mandatory entity `{node.parent_entity}` has at least 1 valid child record" # pylint: disable=C0301 + ) + + join_expr = " AND ".join( + f"{node.parent_entity}.{k} = {node.entity_name}.{v}" + for k, v in node.join_fields.items() + ) + + with BackgroundMessageWriter( + working_directory=working_directory, + dve_stage=self.__stage_name__, + key_fields=key_fields, + logger=self.logger, + ) as msg_writer: + location = next(iter(node.join_fields.values())) + missing_children_records = self.check_mandatory_group( + entities=entities, + config=GroupIdentification( + entity_name=node.parent_entity, + target_name=node.entity_name, + join_condition=join_expr, + ), + ) + _messages = [ + FeedbackMessage( + entity=node.parent_entity, + record=record, # type: ignore + error_location=location, + error_message=template_object(node.no_valid_records_error_message, record), + failure_type="record", + error_type="record", + error_code=node.no_valid_records_error_code, + reporting_field=location, + category="Children missing", + ) + for record in missing_children_records + ] + msg_writer.write_queue.put(_messages) + return len(_messages) > 0 + + entity_issues_found: dict[EntityName, bool] = {} + + for tree in entity_hierarchy.entity_trees.values(): + for node in tree.iterate_lowest_descendent_up(): + if node.parent_entity and node.mandatory: + entity_issues_found[node.parent_entity] = process_node(node) + + # entities.update(entities) + + return [], entity_issues_found + # pylint: disable=R0912,R0914 def apply_sync_filters( self, @@ -428,6 +582,7 @@ def apply_sync_filters( excluded_columns=filter_column_names, reporting=rule.reporting, parent=rule.parent, + error_on_null=True, ), ) if not success: @@ -459,6 +614,7 @@ def apply_sync_filters( expression=f"NOT ({rule.expression})", reporting=rule.reporting, parent=rule.parent, + error_on_null=True, ), ) if not success: @@ -691,3 +847,8 @@ def filter_data_contract_record_rejections( ): """Method to filter out record rejection errors from the data contract for a given entity""" raise NotImplementedError() + + @staticmethod + def get_entity_count(entity: EntityType) -> int: + """Method to get count of records in entity""" + raise NotImplementedError() diff --git a/src/dve/core_engine/backends/implementations/duckdb/duckdb_helpers.py b/src/dve/core_engine/backends/implementations/duckdb/duckdb_helpers.py index 588cd7e..84be049 100644 --- a/src/dve/core_engine/backends/implementations/duckdb/duckdb_helpers.py +++ b/src/dve/core_engine/backends/implementations/duckdb/duckdb_helpers.py @@ -3,6 +3,7 @@ """Helper objects for duckdb data contract implementation""" +import itertools from collections.abc import Generator, Iterator from dataclasses import is_dataclass from datetime import date, datetime, time @@ -99,8 +100,11 @@ def __call__(self): def table_exists(connection: DuckDBPyConnection, table_name: str) -> bool: """check if a table exists in a given DuckDBPyConnection""" - return table_name in map(lambda x: x[0], connection.sql("SHOW TABLES").fetchall()) + return table_name in get_all_existing_ddb_tables(connection) +def get_all_existing_ddb_tables(connection: DuckDBPyConnection) -> tuple[str]: + """Fetch all tables available ina given duckdb connection""" + return tuple(itertools.chain.from_iterable(connection.sql("SHOW TABLES").fetchall())) def relation_is_empty(relation: DuckDBPyRelation) -> bool: """Check if a duckdb relation is empty""" @@ -284,11 +288,11 @@ def _ddb_filter_contract_errors( "RecordIndex": "INTEGER", "FailureType": "STRING", "Status": "STRING", - "Entity": "STRING", + "OriginalEntity": "STRING", }, ) .filter( - f"FailureType == 'record' AND Status != 'informational' AND Entity = '{entity_name}'" + f"FailureType == 'record' AND Status != 'informational' AND OriginalEntity = '{entity_name}'" # pylint: disable=C0301 ) # pylint: disable=C0301 .select("RecordIndex") .distinct() @@ -322,6 +326,16 @@ def duckdb_get_entity_count(cls): return cls +def _duckdb_check_entity_empty(self, entity: DuckDBPyRelation) -> bool: # pylint: disable=W0613 + return entity.shape[0] == 0 + + +def duckdb_check_entity_empty(cls): + """Class decorator to check whether a supplied entity is empty""" + cls.check_entity_empty = _duckdb_check_entity_empty + return cls + + def get_all_registered_udfs(connection: DuckDBPyConnection) -> set[str]: """Function to supply the names of a registered functions stored in the supplied duckdb connection. Creates the temp table used to store registered functions (if not exists). diff --git a/src/dve/core_engine/backends/implementations/duckdb/readers/csv.py b/src/dve/core_engine/backends/implementations/duckdb/readers/csv.py index 723e5e3..012673a 100644 --- a/src/dve/core_engine/backends/implementations/duckdb/readers/csv.py +++ b/src/dve/core_engine/backends/implementations/duckdb/readers/csv.py @@ -22,6 +22,7 @@ UnableToParseCSVError, ) from dve.core_engine.backends.implementations.duckdb.duckdb_helpers import ( + duckdb_check_entity_empty, duckdb_record_index, duckdb_write_parquet, get_duckdb_type_from_annotation, @@ -35,6 +36,7 @@ from dve.parser.file_handling import get_content_length +@duckdb_check_entity_empty @duckdb_record_index @duckdb_write_parquet class DuckDBCSVReader(CSVFileReader): diff --git a/src/dve/core_engine/backends/implementations/duckdb/readers/json.py b/src/dve/core_engine/backends/implementations/duckdb/readers/json.py index 79d74c6..84b601d 100644 --- a/src/dve/core_engine/backends/implementations/duckdb/readers/json.py +++ b/src/dve/core_engine/backends/implementations/duckdb/readers/json.py @@ -10,6 +10,7 @@ from dve.core_engine.backends.base.reader import BaseFileReader, read_function from dve.core_engine.backends.implementations.duckdb.duckdb_helpers import ( + duckdb_check_entity_empty, duckdb_record_index, duckdb_write_parquet, get_duckdb_type_from_annotation, @@ -18,6 +19,7 @@ from dve.core_engine.type_hints import URI, EntityName +@duckdb_check_entity_empty @duckdb_record_index @duckdb_write_parquet class DuckDBJSONReader(BaseFileReader): diff --git a/src/dve/core_engine/backends/implementations/duckdb/readers/xml.py b/src/dve/core_engine/backends/implementations/duckdb/readers/xml.py index 42e281a..ac11169 100644 --- a/src/dve/core_engine/backends/implementations/duckdb/readers/xml.py +++ b/src/dve/core_engine/backends/implementations/duckdb/readers/xml.py @@ -10,7 +10,10 @@ from dve.core_engine.backends.base.reader import read_function from dve.core_engine.backends.exceptions import MessageBearingError -from dve.core_engine.backends.implementations.duckdb.duckdb_helpers import duckdb_write_parquet +from dve.core_engine.backends.implementations.duckdb.duckdb_helpers import ( + duckdb_check_entity_empty, + duckdb_write_parquet, +) from dve.core_engine.backends.readers.xml import XMLStreamReader from dve.core_engine.backends.utilities import ( get_polars_type_from_annotation, @@ -20,6 +23,7 @@ from dve.core_engine.type_hints import URI +@duckdb_check_entity_empty @polars_record_index @duckdb_write_parquet class DuckDBXMLStreamReader(XMLStreamReader): diff --git a/src/dve/core_engine/backends/implementations/duckdb/rules.py b/src/dve/core_engine/backends/implementations/duckdb/rules.py index dc73dad..edb7d71 100644 --- a/src/dve/core_engine/backends/implementations/duckdb/rules.py +++ b/src/dve/core_engine/backends/implementations/duckdb/rules.py @@ -1,6 +1,6 @@ """Business rule definitions for duckdb backend""" - -from collections.abc import Callable +# pylint: disable=R0801 +from collections.abc import Callable, Iterable, Iterator from typing import get_type_hints from uuid import uuid4 @@ -23,12 +23,14 @@ from dve.core_engine.backends.implementations.duckdb.duckdb_helpers import ( DDBStruct, ddb_filter_contract_errors, + duckdb_get_entity_count, duckdb_read_parquet, duckdb_record_index, duckdb_rel_to_dictionaries, duckdb_write_parquet, get_all_registered_udfs, get_duckdb_type_from_annotation, + relation_is_empty, ) from dve.core_engine.backends.implementations.duckdb.types import ( DuckDBEntities, @@ -43,6 +45,7 @@ Aggregation, AntiJoin, ConfirmJoinHasMatch, + GroupIdentification, HeaderJoin, ImmediateFilter, InnerJoin, @@ -53,12 +56,14 @@ SemiJoin, TableUnion, ) +from dve.core_engine.constants import RECORD_INDEX_COLUMN_NAME from dve.core_engine.functions import implementations as functions from dve.core_engine.message import FeedbackMessage from dve.core_engine.templating import template_object from dve.core_engine.type_hints import Messages +@duckdb_get_entity_count @duckdb_record_index @duckdb_write_parquet @duckdb_read_parquet @@ -364,7 +369,7 @@ def join_header(self, entities: DuckDBEntities, *, config: HeaderJoin) -> Messag ), ) - target_schema = DDBStruct(dict(zip(target_rel.columns, target_rel.dtypes)))() + target_schema = DDBStruct(dict(zip(target_rel.columns, target_rel.dtypes)))() # type: ignore # pylint:disable=C0301 joined_rel = source_rel.select( StarExpression(exclude=[]), @@ -375,8 +380,11 @@ def join_header(self, entities: DuckDBEntities, *, config: HeaderJoin) -> Messag return [] def identify_orphans( - self, entities: DuckDBEntities, *, config: OrphanIdentification - ) -> Messages: + self, + entities: DuckDBEntities, + *, + config: OrphanIdentification, + ) -> Iterable: """Identify records in an entity which don't have at least one corresponding match in the target. A new boolean column will be added to `entity` ('IsOrphaned') indicating whether the condition matched. @@ -390,41 +398,89 @@ def identify_orphans( target_rel: DuckDBPyRelation = entities[config.target_name] target_rel = target_rel.set_alias(config.target_name) - key_name = f"key_{uuid4().hex}" - source_rel = source_rel.select(f"*, row_number() over () as {key_name}").set_alias( - config.entity_name - ) + if relation_is_empty(source_rel): + self.logger.info(f"{config.entity_name} is empty. Skipping orphan check.") + return [] + match_name = f"matched_{uuid4().hex}" target_rel = target_rel.select( StarExpression(exclude=[]), ConstantExpression(1).alias(match_name) ).set_alias(config.target_name) - joined_rel: DuckDBPyRelation = source_rel.join( - target_rel, condition=config.join_condition, how="left" - ).aggregate(f"{key_name}, coalesce(count({match_name})==0, TRUE) AS IsOrphaned") + orphaned_rel: DuckDBPyRelation = ( + source_rel.join(target_rel, condition=config.join_condition, how="left") + .aggregate( + f"{config.entity_name}.{RECORD_INDEX_COLUMN_NAME}, {config.entity_name}.{config.id}, coalesce(count({match_name}), 0)==0 AS IsOrphaned" # pylint: disable=C0301 + ) + .filter("IsOrphaned") + .select(RECORD_INDEX_COLUMN_NAME) + .set_alias("orphan") + ) - if "IsOrphaned" not in source_rel.columns: - result: DuckDBPyRelation = source_rel.join( - joined_rel, condition=key_name, how="left" - ).select(StarExpression(exclude=[key_name])) - else: - result = source_rel.set_alias("source").join( - joined_rel.set_alias("joined"), - condition=f"source.{key_name} = joined.{key_name}", - how="left", + if relation_is_empty(orphaned_rel): + self.logger.info( + f"Found 0 orphan records between {config.entity_name} and {config.target_name}" + ) + return [] + + message_rel = ( + entities[config.entity_name] + .set_alias(config.entity_name) + .join( + orphaned_rel, + f"{config.entity_name}.{RECORD_INDEX_COLUMN_NAME} = orphan.{RECORD_INDEX_COLUMN_NAME}", # pylint: disable=C0301 + "semi", ) + ) + filtered_rel = ( + entities[config.entity_name] + .set_alias(config.entity_name) + .join( + orphaned_rel, + f"{config.entity_name}.{RECORD_INDEX_COLUMN_NAME} = orphan.{RECORD_INDEX_COLUMN_NAME}", # pylint: disable=C0301 + "anti", + ) + ) - columns = {name: f"source.{name}" for name in source_rel.columns} - if "IsOrphaned" in source_rel.columns: - columns["IsOrphaned"] = ColumnExpression("source.IsOrphaned") | ColumnExpression("joined.IsOrphaned") # type: ignore # pylint: disable=line-too-long - columns.pop(key_name, None) + entities[config.entity_name] = filtered_rel - result = result.select( - ",".join([f"{column} as {name}" for name, column in columns.items()]) - ) + return duckdb_rel_to_dictionaries(message_rel) - entities[config.new_entity_name or config.entity_name] = result - return [] + def check_mandatory_group( + self, entities: DuckDBEntities, *, config: GroupIdentification + ) -> Iterator: + """ + Check that a mandatory key in an entity has at least one valid entry in the all the + child entities. + """ + source_rel: DuckDBPyRelation = entities[config.entity_name] + source_rel = source_rel.set_alias(config.entity_name) + target_rel: DuckDBPyRelation = entities[config.target_name] + target_rel = target_rel.set_alias(config.target_name) + + source_columns = [f"{config.entity_name}.{c.strip()}" for c in source_rel.columns] + _pk, fk = config.join_condition.split("=") + + joined_rel = source_rel.join(target_rel, config.join_condition, "left").select( + *source_columns, + ColumnExpression(fk.strip()).alias("fk"), + ) + + missing_children_rel = joined_rel.filter("fk IS NULL") + filtered_rel = joined_rel.filter("fk IS NOT NULL").select(StarExpression(exclude=["fk"])) + + _no_valid_child_records: tuple[int] = missing_children_rel.count("*").fetchone() # type: ignore # pylint: disable=C0301 + if _no_valid_child_records: + _no_valid_children = _no_valid_child_records[0] + else: + _no_valid_children = 0 + self.logger.info( + f"Found {_no_valid_children} records with no valid children in {config.entity_name}." + ) # pylint: disable=C0301 + + entities[config.entity_name] = filtered_rel + + return duckdb_rel_to_dictionaries(missing_children_rel) def union(self, entities: DuckDBEntities, *, config: TableUnion) -> Messages: """Union two entities together, taking the columns from each by name. @@ -496,7 +552,12 @@ def notify(self, entities: DuckDBEntities, *, config: Notification) -> Messages: """ messages: Messages = [] entity = entities[config.entity_name] - + if config.error_if_expression_null: + if self.get_entity_count(entity.filter(f"({config.expression}) IS NULL")) > 0: + raise ValueError( + f"The filter evaluated for error code {config.reporting.code}" + + f" in entity {config.entity_name} produced some NULL results. Please investigate." # pylint: disable=C0301 + ) matched = entity.filter(config.expression) if config.excluded_columns: matched = matched.select(StarExpression(exclude=config.excluded_columns)) diff --git a/src/dve/core_engine/backends/implementations/spark/contract.py b/src/dve/core_engine/backends/implementations/spark/contract.py index d2fd9ae..432a731 100644 --- a/src/dve/core_engine/backends/implementations/spark/contract.py +++ b/src/dve/core_engine/backends/implementations/spark/contract.py @@ -156,8 +156,9 @@ def apply_data_contract( fld, fld_info.annotation ).alias(fld) if fld in record_df.columns - else lit(None).cast( - get_type_from_annotation(fld_info.annotation)).alias(fld) + else lit(None) + .cast(get_type_from_annotation(fld_info.annotation)) + .alias(fld) ) for fld, fld_info in entity_fields.items() ], diff --git a/src/dve/core_engine/backends/implementations/spark/readers/csv.py b/src/dve/core_engine/backends/implementations/spark/readers/csv.py index 2df30c5..5cd2f56 100644 --- a/src/dve/core_engine/backends/implementations/spark/readers/csv.py +++ b/src/dve/core_engine/backends/implementations/spark/readers/csv.py @@ -12,6 +12,7 @@ from dve.core_engine.backends.exceptions import EmptyFileError from dve.core_engine.backends.implementations.spark.spark_helpers import ( get_type_from_annotation, + spark_check_entity_empty, spark_record_index, spark_write_parquet, ) @@ -20,6 +21,7 @@ from dve.parser.file_handling import get_content_length +@spark_check_entity_empty @spark_record_index @spark_write_parquet class SparkCSVReader(CSVFileReader): diff --git a/src/dve/core_engine/backends/implementations/spark/readers/json.py b/src/dve/core_engine/backends/implementations/spark/readers/json.py index 6123009..6404231 100644 --- a/src/dve/core_engine/backends/implementations/spark/readers/json.py +++ b/src/dve/core_engine/backends/implementations/spark/readers/json.py @@ -11,6 +11,7 @@ from dve.core_engine.backends.exceptions import EmptyFileError from dve.core_engine.backends.implementations.spark.spark_helpers import ( get_type_from_annotation, + spark_check_entity_empty, spark_record_index, spark_write_parquet, ) @@ -18,6 +19,7 @@ from dve.parser.file_handling import get_content_length +@spark_check_entity_empty @spark_record_index @spark_write_parquet class SparkJSONReader(BaseFileReader): diff --git a/src/dve/core_engine/backends/implementations/spark/readers/xml.py b/src/dve/core_engine/backends/implementations/spark/readers/xml.py index ba42d29..4d6df6a 100644 --- a/src/dve/core_engine/backends/implementations/spark/readers/xml.py +++ b/src/dve/core_engine/backends/implementations/spark/readers/xml.py @@ -17,6 +17,7 @@ from dve.core_engine.backends.implementations.spark.spark_helpers import ( df_is_empty, get_type_from_annotation, + spark_check_entity_empty, spark_record_index, spark_write_parquet, ) @@ -29,6 +30,7 @@ """The mode to use when parsing XML files with Spark.""" +@spark_check_entity_empty @spark_record_index @spark_write_parquet class SparkXMLStreamReader(XMLStreamReader): @@ -55,6 +57,7 @@ def read_to_dataframe( ) +@spark_check_entity_empty @spark_record_index @spark_write_parquet class SparkXMLReader(BasicXMLFileReader): # pylint: disable=too-many-instance-attributes diff --git a/src/dve/core_engine/backends/implementations/spark/rules.py b/src/dve/core_engine/backends/implementations/spark/rules.py index 825ee15..84df52b 100644 --- a/src/dve/core_engine/backends/implementations/spark/rules.py +++ b/src/dve/core_engine/backends/implementations/spark/rules.py @@ -1,6 +1,6 @@ """Step implementations in Spark.""" - -from collections.abc import Callable +# pylint: disable=R0801 +from collections.abc import Callable, Iterator from typing import Optional from uuid import uuid4 @@ -15,6 +15,7 @@ get_all_registered_udfs, object_to_spark_literal, spark_filter_contract_errors, + spark_get_entity_count, spark_read_parquet, spark_record_index, spark_write_parquet, @@ -34,6 +35,7 @@ ColumnAddition, ColumnRemoval, ConfirmJoinHasMatch, + GroupIdentification, HeaderJoin, ImmediateFilter, InnerJoin, @@ -51,6 +53,7 @@ from dve.core_engine.type_hints import Messages +@spark_get_entity_count @spark_record_index @spark_write_parquet @spark_read_parquet @@ -338,7 +341,8 @@ def union(self, entities: SparkEntities, *, config: TableUnion) -> Messages: def identify_orphans( self, entities: SparkEntities, *, config: OrphanIdentification - ) -> Messages: + ) -> tuple[Messages, int]: + # TODO - adjust this to new setup of identify and remove orphans source_df: DataFrame = entities[config.entity_name] source_df = source_df.alias(config.entity_name) target_df: DataFrame = entities[config.target_name] @@ -371,7 +375,13 @@ def identify_orphans( result = result.select(*[column.alias(name) for name, column in columns.items()]) entities[config.new_entity_name or config.entity_name] = result - return [] + return [], 0 + + def check_mandatory_group( + self, entities: SparkEntities, *, config: GroupIdentification + ) -> Iterator: + # TODO - implement for spark + raise NotImplementedError def filter(self, entities: SparkEntities, *, config: ImmediateFilter) -> Messages: """Filter an entity immediately, and do not emit any messages. @@ -393,6 +403,13 @@ def notify(self, entities: SparkEntities, *, config: Notification) -> Messages: messages: Messages = [] entity = entities[config.entity_name] + if config.error_if_expression_null: + if self.get_entity_count(entity.filter(f"({config.expression}) IS NULL")) > 0: + raise ValueError( + f"The filter evaluated for error code {config.reporting.code}" + + f" in entity {config.entity_name} produced some NULL results. Please investigate." # pylint: disable=C0301 + ) + matched = entity.filter(config.expression) if config.excluded_columns: matched = matched.drop(*config.excluded_columns) diff --git a/src/dve/core_engine/backends/implementations/spark/spark_helpers.py b/src/dve/core_engine/backends/implementations/spark/spark_helpers.py index 8c14132..1257306 100644 --- a/src/dve/core_engine/backends/implementations/spark/spark_helpers.py +++ b/src/dve/core_engine/backends/implementations/spark/spark_helpers.py @@ -415,14 +415,14 @@ def _spark_filter_contract_errors( st.StructField("RecordIndex", st.IntegerType()), st.StructField("FailureType", st.StringType()), st.StructField("Status", st.StringType()), - st.StructField("Entity", st.StringType()), + st.StructField("OriginalEntity", st.StringType()), ] ), ) .filter( (sf.col("FailureType") == sf.lit("record")) & (sf.col("Status") != sf.lit("informational")) - & (sf.col("Entity") == sf.lit(entity_name)) + & (sf.col("OriginalEntity") == sf.lit(entity_name)) ) .distinct() .orderBy(sf.asc(sf.col("RecordIndex"))) @@ -456,6 +456,16 @@ def spark_get_entity_count(cls): return cls +def _spark_check_entity_empty(self, entity: DataFrame) -> bool: # pylint: disable=W0613 + return entity.count() == 0 + + +def spark_check_entity_empty(cls): + """Class decorator to check whether a supplied entity is empty""" + cls.check_entity_empty = _spark_check_entity_empty + return cls + + def get_all_registered_udfs(spark: SparkSession) -> set[str]: """Function to supply the names of a registered functions stored in the supplied spark session. diff --git a/src/dve/core_engine/backends/metadata/contract.py b/src/dve/core_engine/backends/metadata/contract.py index e3eb0c0..2e7565d 100644 --- a/src/dve/core_engine/backends/metadata/contract.py +++ b/src/dve/core_engine/backends/metadata/contract.py @@ -35,7 +35,7 @@ class DataContractMetadata(BaseModel, frozen=True, arbitrary_types_allowed=True) reporting_fields: dict[EntityName, ReportingFields] """The per-entity reporting fields.""" cache_originals: bool = False - """Whether to cache the original entities after loading.""" + """WARNING - Depreciated functionality. Whether to cache the original entities after loading.""" _schemas: dict[EntityName, type[BaseModel]] = PrivateAttr(default_factory=dict) """The pydantic models of the schmas.""" diff --git a/src/dve/core_engine/backends/metadata/rules.py b/src/dve/core_engine/backends/metadata/rules.py index f3a6305..a1db7cc 100644 --- a/src/dve/core_engine/backends/metadata/rules.py +++ b/src/dve/core_engine/backends/metadata/rules.py @@ -33,6 +33,7 @@ "CopyEntity", "DeferredFilter", "EntityRemoval", + "GroupIdentification", "HeaderJoin", "ImmediateFilter", "InnerJoin", @@ -40,6 +41,7 @@ "OneToOneJoin", "OneToOneJoin", "OrphanIdentification", + "OrphanRemoval", "ParentMetadata", "RenameEntity", "Rule", @@ -280,6 +282,8 @@ class Notification(AbstractStep): """Columns to be excluded from the record in the report.""" reporting: ReportingConfig """The reporting information for the filter.""" + error_if_expression_null: bool = False + """Raise error if the results of evaluating the expression passed leads to some NULL results""" def get_required_entities(self) -> set[EntityName]: return {self.entity_name} @@ -558,6 +562,17 @@ class OrphanIdentification(AbstractConditionalJoin): """A step within a rule. This is either a rule config or the literal string 'sync'.""" +class OrphanRemoval(BaseStep): + """Remove an orphan record from the `entity`.""" + + reporting: ReportingConfig + """The reporting information for the row removal.""" + + +class GroupIdentification(AbstractConditionalJoin): + """Identify mandatory records which do not have any valid child records""" + + class Rule(BaseModel): """A rule, made up of multiple steps.""" diff --git a/src/dve/core_engine/backends/readers/xml.py b/src/dve/core_engine/backends/readers/xml.py index badc03a..a3fd437 100644 --- a/src/dve/core_engine/backends/readers/xml.py +++ b/src/dve/core_engine/backends/readers/xml.py @@ -255,6 +255,8 @@ def _get_elements_from_stream(self, stream: IO[bytes]) -> Iterator[XMLElement]: remove_comments=True, dtd_validation=False, resolve_entities=False, + no_network=True, + load_dtd=False, ) tree: etree._ElementTree = etree.parse(stream, parser) @@ -375,6 +377,8 @@ def _get_elements_from_stream(self, stream: IO[bytes]) -> Iterator[XMLElement]: remove_comments=True, dtd_validation=False, resolve_entities=False, + no_network=True, + load_dtd=False, ) container_contexts = 1 if not self.root_tag else 0 diff --git a/src/dve/core_engine/configuration/v1/__init__.py b/src/dve/core_engine/configuration/v1/__init__.py index 959596f..421c56e 100644 --- a/src/dve/core_engine/configuration/v1/__init__.py +++ b/src/dve/core_engine/configuration/v1/__init__.py @@ -1,9 +1,10 @@ """The loader for the first JSON-based dataset configuration.""" import json -from typing import Any, Optional, Union +from typing import Any, Optional, Type, Union -from pydantic import BaseModel, Field, PrivateAttr, validate_call +from pydantic import BaseModel, Field, PrivateAttr, field_validator, model_validator, validate_call +from pydantic_core.core_schema import FieldValidationInfo from typing_extensions import Literal from dve.core_engine.backends.base.reference_data import ReferenceConfig, ReferenceConfigUnion @@ -22,7 +23,14 @@ ) from dve.core_engine.configuration.v1.steps import StepConfigUnion from dve.core_engine.message import DataContractErrorDetail -from dve.core_engine.type_hints import EntityName, ErrorCategory, ErrorType, TemplateVariables +from dve.core_engine.type_hints import ( + EntityName, + ErrorCategory, + ErrorCode, + ErrorMessage, + ErrorType, + TemplateVariables, +) from dve.core_engine.validation import RowValidator from dve.parser.file_handling import joinuri, open_stream, resolve_location from dve.parser.type_hints import URI, Extension @@ -38,6 +46,8 @@ FieldName = str """The name of a field within a model/schema.""" +JoinFields = Optional[dict[str, str]] +"""The fields required ( parent > child ) to join a child entity back to the parent""" TypeOrDef = Union[ # pylint: disable=C0103 TypeName, "_CallableTypeDefinition", "_ModelTypeDefinition", "_TypeAliasDefinition" ] @@ -47,6 +57,8 @@ """The operation """ RuleType = type[AbstractStep] """The metadata step type implemented by the rule.""" +AllowedAdditionalReaderChecks = Literal["check_empty"] +"""Additional checks to be performed in the file_transformation stage""" class _BaseTypeDefintion(BaseModel): @@ -81,6 +93,67 @@ class _TypeAliasDefinition(_BaseTypeDefintion): """The name of the Python type.""" +class _LinkageConfig(BaseModel): + """Specify how to link entities back to parents if required""" + + parent_entity: Optional[EntityName] = None + """The name of the parent entity""" + join_fields: JoinFields = Field(default_factory=dict) + """The fields that can be used to link back to the parent entity""" + is_root_entity: bool = False + """Whether the entity is the highest level parent in a tree""" + mandatory: bool = False + """If the entity is a child, is it a mandatory field of the parent""" + no_valid_records_error_code: Optional[ErrorCode] = "NoValidRecords" + """The error code to emit if the entity has no valid records and is mandatory in the parent entity""" # pylint: disable=C0301 + no_valid_records_error_message: Optional[ErrorMessage] = ( + "parent record removed as no valid child records" + ) + """The error message to emit if the entity has no valid records and is mandatory in the parent entity""" # pylint: disable=C0301 + missing_parent_id_error_code: Optional[ErrorCode] = "MissingParentRecord" + """The error code to emit if the entity contains records that are orphaned by parent record rejections""" # pylint: disable=C0301 + missing_parent_id_error_message: Optional[ErrorMessage] = ( + "Records removed due to no valid parent record" + ) + """The error code to emit if the entity contains records that are orphaned by parent record rejections""" # pylint: disable=C0301 + empty_entity_error_code: ErrorCode = "EmptyEntity" + """The error code to emit if a mandatory entity has no valid remaining records""" + empty_entity_error_message: ErrorMessage = "no valid records remaining" + """The error message to emit if a mandatory entity has no valid remaining records""" + + @model_validator(mode="after") + def _check_root_no_parent_or_join_keys(self): + if self.is_root_entity and (self.parent_entity or self.join_fields): + raise ValueError( + "If entity is root, neither parent_entity nor join keys should be specified" + ) + return self + + @model_validator(mode="after") + def _check_non_root_entities_have_a_defined_parent(self): + """Check that non root entities have a parent defined.""" + if not self.is_root_entity and self.parent_entity is None: + raise ValueError( + 'Non-root entity has no defined parent entity. If you intend this to be a root ' \ + 'entity you must specify `"is_root_entity": true` for the entity. ' \ + 'Otherwise you must specify a `"parent_entity": ""` for this entity.' + ) + return self + + @model_validator(mode="after") + def _check_root_mandatory(self): + if self.is_root_entity and not self.mandatory: + raise ValueError("If entity is root, it must be labelled mandatory") + return self + + @model_validator(mode="after") + def _check_parent_entity_with_join_keys(self): + if self.parent_entity or self.join_fields: + if not (self.parent_entity and self.join_fields): + raise ValueError("Both parent_entity and join_fields must be supplied if one is") + return self + + class _SchemaConfig(BaseModel): """Configuration for a component schema within a dataset.""" @@ -90,6 +163,11 @@ class _SchemaConfig(BaseModel): """A list of the field names within the schema which _must_ be provided.""" +class _ReaderAdditionalChecksConfig(BaseModel): + error_code: str + error_message: str + + class _ReaderConfig(BaseModel): # type: ignore """Reader configuration options for a model.""" @@ -110,6 +188,10 @@ class _ModelConfig(_SchemaConfig): """A single key field to be used by the model.""" reader_config: dict[Extension, _ReaderConfig] """Reader configuration options for the model.""" + reader_additional_checks: dict[AllowedAdditionalReaderChecks, _ReaderAdditionalChecksConfig] = ( + Field(default_factory=dict) + ) + """Additional checks to be performed after the entity is read""" aliases: dict[FieldName, FieldName] = Field(default_factory=dict) """An alias field name mapping.""" @@ -136,7 +218,7 @@ class V1DataContractConfig(BaseModel): """Configuration for the data contract component of the dataset.""" cache_originals: bool = False - """Whether to cache the original entities after loading.""" + """WARNING - Depreciated functionality. Whether to cache the original entities after loading.""" error_details: Optional[URI] = None """Optional URI containing custom data contract error codes and messages""" types: dict[TypeName, TypeOrDef] = Field(default_factory=dict) @@ -177,6 +259,8 @@ class V1EngineConfig(BaseEngineConfig): default_factory=dict ) """Rule store rules from the loaded rule stores.""" + entity_relationships: dict[EntityName, _LinkageConfig] = Field(default_factory=dict) + """The parent-child relationships linking the defined entities""" @validate_call def _update_rule_store(self, rule_store: dict[RuleName, BusinessComponentSpecConfigUnion]): @@ -322,14 +406,13 @@ def get_contract_metadata(self) -> DataContractMetadata: } reporting_fields[entity_name] = dataset_config.reporting_fields validators[entity_name] = RowValidator( - contract_dict, entity_name, error_info=error_info + contract_dict, entity_name, error_info=error_info.get(entity_name) ) return DataContractMetadata( reader_metadata=reader_metadata, validators=validators, reporting_fields=reporting_fields, - cache_originals=self.contract.cache_originals, ) def load_error_message_info(self, uri): diff --git a/src/dve/core_engine/configuration/v1/hierarchy.py b/src/dve/core_engine/configuration/v1/hierarchy.py new file mode 100644 index 0000000..35997eb --- /dev/null +++ b/src/dve/core_engine/configuration/v1/hierarchy.py @@ -0,0 +1,210 @@ +"""Classes to help determine and store entity hierarchy information.""" + +import json +from typing import Any, Iterable, Optional, Union + +from pydantic import BaseModel, Field, model_validator + +from dve.core_engine.configuration.v1 import V1EngineConfig, _LinkageConfig +from dve.core_engine.type_hints import EntityName, ErrorCode, ErrorMessage +from dve.metadata_parser.exc import EntityNotFoundError +from dve.parser.file_handling.service import open_stream +from dve.parser.type_hints import URI + + +class HierarchyNode(BaseModel): + """Stores entity hierarchy information""" + + entity_name: str + parent_entity: Optional[str] = None + children: list["HierarchyNode"] = Field(default_factory=list) + mandatory: bool = False + join_fields: dict[str, str] = Field(default_factory=dict) + no_valid_records_error_code: ErrorCode = "NoValidRecords" + no_valid_records_error_message: ErrorMessage = "parent record removed as no valid child records" + missing_parent_id_error_code: Optional[ErrorCode] = "MissingParentRecord" + missing_parent_id_error_message: Optional[ErrorMessage] = ( + "Records removed due to no valid parent record" + ) + empty_entity_error_code: ErrorCode = "EmptyEntity" + empty_entity_error_message: ErrorMessage = "no valid records remaining" + + @model_validator(mode="after") + def validate_empty_error_details(self): + """ + Removes the default messaging for checking empty entities as not performed on + non mandatory nodes/entities + """ + if not self.mandatory: + self.empty_entity_error_code = None + self.empty_entity_error_message = None + return self + + def get_descendents(self) -> list["HierarchyNode"]: + """Recursively list all descendents of the node""" + descendents = [] + for node in self.children: # type: ignore + descendents.append(node) + descendents.extend(node.get_descendents()) + return descendents + + def get_descendent_names(self) -> list[str]: + """Recursively list all names of descendents of the node""" + return [node.entity_name for node in self.get_descendents()] + + def get_node(self, entity_name: str) -> Union["HierarchyNode", None]: + """Recursively search for node and return if found""" + node = None + if self.entity_name == entity_name: + return self + for child in self.children: # type: ignore + node = child.get_node(entity_name) + if node: + break + return node + + def add_child_node(self, parent_entity: str, child_info: "HierarchyNode") -> None: + """Add a child node if the parent exists in the hierarchy""" + try: + self.get_node(parent_entity).children.append(child_info) # type: ignore + except AttributeError as exc: + raise EntityNotFoundError( + f"Can't find parent node {parent_entity} in {self.entity_name}" + ) from exc + + def as_dict(self) -> dict[str, dict[str, Any]]: + """Get dictionary representation of entity hierarchy""" + child_dict: dict[str, dict[str, Any]] = {} + for node in self.children: # type: ignore + child_dict.update(node.as_dict()) + + ret_dict = self.model_dump(exclude={"entity_name", "children"}) + ret_dict.update({"children": child_dict}) + + return {self.entity_name: ret_dict} + + def _get_full_tree(self): + """Get all nodes in tree, including the root""" + desc = self.get_descendents() + desc.insert(0, self) + return desc + + def iterate_root_down(self): + """Iterate through nodes from root to lowest descendent""" + yield from self._get_full_tree() + + def iterate_lowest_descendent_up(self): + """Iterate through nodes from lowest descendent to root""" + yield from self._get_full_tree()[::-1] + + +class EntityHierarchy: + """Determines and stores entity hierarchy information from config""" + + def __init__(self, entity_trees: dict[EntityName, HierarchyNode]): + self.entity_trees = entity_trees + + @staticmethod + def determine_trees( + all_datasets: Iterable[str], entity_relationships: dict[str, _LinkageConfig] + ) -> dict[EntityName, HierarchyNode]: + """Determine the entity hierarchy trees and store as HierarchyNodes""" + root_entities: dict[str, _LinkageConfig] = dict( + filter(lambda x: x[1].is_root_entity, entity_relationships.items()) + ) + top_level_parents: dict[EntityName, HierarchyNode] = { + entity_name: HierarchyNode( + entity_name=entity_name, + parent_entity=None, + **config.model_dump( + exclude={ + "parent_entity", + "missing_parent_id_error_code", + "missing_parent_id_error_message", + } + ), + missing_parent_id_error_code=None, + missing_parent_id_error_message=None, + ) + for entity_name, config in root_entities.items() + } + + if default_roots := [ + entity_name for entity_name in all_datasets if entity_name not in entity_relationships + ]: + for entity_name in default_roots: + top_level_parents[entity_name] = HierarchyNode( + entity_name=entity_name, + parent_entity=None, + missing_parent_id_error_code=None, + missing_parent_id_error_message=None, + ) + + for name, linkage_detail in entity_relationships.items(): + for main_entity, parent_node in top_level_parents.items(): + if linkage_detail.is_root_entity: + break + + if ( + linkage_detail.parent_entity == main_entity + or linkage_detail.parent_entity in parent_node.get_descendent_names() + ): + parent_node.add_child_node( + linkage_detail.parent_entity, # type: ignore + HierarchyNode(entity_name=name, **linkage_detail.model_dump()), + ) + break + else: + raise EntityNotFoundError( + f"Can't find parent entity {linkage_detail.parent_entity} defined to " + + f"establish hierarchy for {name} - please ensure it is defined above " + + "any child entities in the dischema." + ) + return top_level_parents + + @classmethod + def from_dischema(cls, dischema_uri: URI): + """Create entity hierarchy direct from dischema""" + with open_stream(dischema_uri) as dischema: + config_dict = json.load(dischema) + all_datasets = config_dict.get("contract", {}).get("datasets", {}).keys() + entity_relationships = { + k: _LinkageConfig(**v) for k, v in config_dict.get("entity_relationships", {}).items() + } + return cls(entity_trees=cls.determine_trees(all_datasets, entity_relationships)) + + @classmethod + def from_engine_config(cls, engine_config: V1EngineConfig): + """Create entity hierarchy direct from engine config""" + return cls( + entity_trees=cls.determine_trees( + all_datasets=engine_config.contract.datasets.keys(), + entity_relationships=engine_config.entity_relationships, + ) + ) + + def get_all_mandatory_nodes( + self, + node: Optional[HierarchyNode] = None, + mandatory_nodes: Optional[list[HierarchyNode]] = None, + nodes_visited: Optional[set[EntityName]] = None, + ) -> list[HierarchyNode]: + """Find and return all mandatory nodes""" + if mandatory_nodes is None: + mandatory_nodes = [] + + if nodes_visited is None: + nodes_visited = set() + + if node is None: + for _node in self.entity_trees.values(): + self.get_all_mandatory_nodes(_node, mandatory_nodes, nodes_visited) + + if node: + if node.mandatory and node.entity_name not in nodes_visited: + nodes_visited.add(node.entity_name) + mandatory_nodes.append(node) + for child_node in node.children: + self.get_all_mandatory_nodes(child_node, mandatory_nodes, nodes_visited) + + return mandatory_nodes diff --git a/src/dve/core_engine/message.py b/src/dve/core_engine/message.py index 78024e9..05dbc17 100644 --- a/src/dve/core_engine/message.py +++ b/src/dve/core_engine/message.py @@ -90,7 +90,7 @@ def extract_error_value(records, error_location): class UserMessage: """The structure of the message that is used to populate the error report.""" - Entity: Optional[str] + ReportingEntity: Optional[str] """The entity that the message pertains to (if applicable).""" Key: Optional[str] "The key field(s) in string format to allow users to identify the record" @@ -176,7 +176,8 @@ class FeedbackMessage: # pylint: disable=too-many-instance-attributes """The category of the error.""" HEADER: ClassVar[list[str]] = [ - "Entity", + "ReportingEntity", + "OriginalEntity", "Key", "FailureType", "Status", @@ -307,6 +308,7 @@ def to_row( return ( self.entity, + self.original_entity, key, self.failure_type, "informational" if self.is_informational else "error", diff --git a/src/dve/core_engine/models.py b/src/dve/core_engine/models.py index bba2986..49dba23 100644 --- a/src/dve/core_engine/models.py +++ b/src/dve/core_engine/models.py @@ -82,7 +82,9 @@ def _ensure_just_file_stem( @property def file_name_with_ext(self): """Return file name with extension.""" - return f"{self.file_name}.{self.file_extension}" + if self.file_extension: + return f"{self.file_name}.{self.file_extension}" + return self.file_name @classmethod def from_metadata_file(cls, submission_id: str, metadata_uri: Location): diff --git a/src/dve/core_engine/type_hints.py b/src/dve/core_engine/type_hints.py index 154ada6..50781ec 100644 --- a/src/dve/core_engine/type_hints.py +++ b/src/dve/core_engine/type_hints.py @@ -133,12 +133,21 @@ """A string indicating the field that the error pertains to.""" FieldValue = Optional[Any] """The value that caused the error.""" -ErrorCategory = Literal["Blank", "Wrong format", "Bad value", "Bad file"] +ErrorCategory = Literal[ + "Blank", + "Wrong format", + "Bad value", + "Bad file", + "Parent Missing", + "Children missing", + "Empty entity", +] """A string indicating the category of the error.""" RecordIndex = Optional[int] """The record index that the error relates to (if applicable)""" MessageTuple = tuple[ + Optional[EntityName], Optional[EntityName], Key, FailureType, diff --git a/src/dve/metadata_parser/models.py b/src/dve/metadata_parser/models.py index 49e2386..0b7d535 100644 --- a/src/dve/metadata_parser/models.py +++ b/src/dve/metadata_parser/models.py @@ -391,6 +391,7 @@ class DatasetSpecification(BaseModel): """Configuration options for a dataset.""" cache_originals: bool = False + """WARNING - Depreciated functionality.""" types: dict[TypeName, FieldSpecification] = Field(default_factory=dict) """Predefined types to be used within schema/dataset definitions.""" schemas: dict[EntityName, EntitySpecification] = Field(default_factory=dict) diff --git a/src/dve/parser/file_handling/service.py b/src/dve/parser/file_handling/service.py index 9ee9d9f..fbdc8ab 100644 --- a/src/dve/parser/file_handling/service.py +++ b/src/dve/parser/file_handling/service.py @@ -273,9 +273,12 @@ def copy_resource(source_uri: URI, target_uri: URI, overwrite: bool = False) -> _transfer_resource(source_uri, target_uri, overwrite, "copy") -def move_resource(source_uri: URI, target_uri: URI, overwrite: bool = False) -> None: - """Move a resource from one location to another.""" +def move_resource(source_uri: URI, target_uri: URI, overwrite: bool = False) -> URI: + """ + Move a resource from one location to another. Returns the target_uri. + """ _transfer_resource(source_uri, target_uri, overwrite, "move") + return target_uri def create_directory(target_uri: URI): diff --git a/src/dve/pipeline/pipeline.py b/src/dve/pipeline/pipeline.py index a9be3ff..60e266a 100644 --- a/src/dve/pipeline/pipeline.py +++ b/src/dve/pipeline/pipeline.py @@ -1,4 +1,4 @@ -# pylint: disable=protected-access,too-many-instance-attributes,too-many-arguments,line-too-long +# pylint: disable=protected-access,too-many-instance-attributes,too-many-arguments,line-too-long,too-many-lines """Generic Pipeline object to define how DVE should be interacted with.""" import json @@ -18,6 +18,7 @@ import dve.reporting.excel_report as er from dve.common.error_utils import ( + BackgroundMessageWriter, dump_feedback_errors, dump_processing_errors, get_feedback_errors_uri, @@ -34,6 +35,7 @@ from dve.core_engine.backends.readers.utilities import get_all_model_fields from dve.core_engine.backends.types import EntityType from dve.core_engine.backends.utilities import stringify_model +from dve.core_engine.configuration.v1.hierarchy import EntityHierarchy from dve.core_engine.exceptions import CriticalProcessingError from dve.core_engine.loggers import get_logger from dve.core_engine.message import FeedbackMessage @@ -215,10 +217,10 @@ def write_file_to_parquet( for model_name, model in models.items(): self._logger.info(f"Transforming {model_name} to stringified parquet") - reader: BaseFileReader = load_reader( - dataset, model_name, ext, self.backend_reader_kwargs - ) try: + reader: BaseFileReader = load_reader( + dataset, model_name, ext, self.backend_reader_kwargs + ) if not entity_type: reader.write_parquet( reader.read_to_py_iterator( @@ -237,10 +239,14 @@ def write_file_to_parquet( model_name, stringify_model(model), # type: ignore get_all_model_fields(models.values()), # type: ignore + dataset[model_name].reader_additional_checks, ), f"{out}{model_name}", ) except MessageBearingError as exc: + self._logger.error( + f"While processing {model_name}, an issue was encountered", exc_info=exc + ) errors.extend(exc.messages) return list(dict.fromkeys(errors)) # remove any duplicate errors @@ -543,7 +549,45 @@ def data_contract_step( return processed_files, failed_processing - def apply_business_rules( # pylint: disable=R0914 + def check_mandatory_entities_have_records( + self, + working_directory: URI, + entities: EntityManager, + entity_hierarchy: EntityHierarchy, + key_fields: Optional[dict[str, list[str]]] = None, + ) -> None: + """ + Check that mandatory entities have at least one record post business rules. Otherwise, + raise a submission rejection error message. + """ + with BackgroundMessageWriter( + working_directory=working_directory, + dve_stage="business_rules", + key_fields=key_fields, + logger=self._logger, + ) as msg_writer: + _msgs = [] + for node in entity_hierarchy.get_all_mandatory_nodes(): + entity_name = node.entity_name + if node.mandatory and self.get_entity_count(entities[entity_name]) == 0: + self._logger.info( + f"Found 0 records in mandatory entity {entity_name} after applying all business rules" # pylint: disable=C0301 + ) + _msgs.append( + FeedbackMessage( + entity=entity_name, + record=None, + error_location=entity_name, + error_message=node.empty_entity_error_message, + failure_type="submission", + error_type="submission", + error_code=node.empty_entity_error_code, + category="Empty entity", + ) + ) + msg_writer.write_queue.put(_msgs) + + def apply_business_rules( # pylint: disable=R0914,R0915 self, submission_info: SubmissionInfo, submission_status: Optional[SubmissionStatus] = None ) -> tuple[SubmissionInfo, SubmissionStatus]: """Apply the business rules to a given submission, the submission may have failed at the @@ -583,7 +627,6 @@ def apply_business_rules( # pylint: disable=R0914 entities[file_name] = self.step_implementations.add_record_index( # type: ignore self.step_implementations.read_parquet(parquet_uri) # type: ignore ) - entities[f"Original{file_name}"] = self.step_implementations.read_parquet(parquet_uri) # type: ignore sub_info_entity = ( self._audit_tables._submission_info.conv_to_entity( # pylint: disable=protected-access @@ -596,8 +639,13 @@ def apply_business_rules( # pylint: disable=R0914 key_fields = {model: conf.reporting_fields for model, conf in model_config.items()} + entity_hierarchy = EntityHierarchy.from_engine_config(config) + _errors_uri, rules_success = self.step_implementations.apply_rules( # type: ignore - working_directory, entity_manager, rules, key_fields + working_directory, + entity_manager, + rules, + key_fields, ) rule_messages = load_feedback_messages( @@ -614,21 +662,17 @@ def apply_business_rules( # pylint: disable=R0914 for entity_name, entity in entity_manager.entities.items(): # Note BI filtering done within the apply_rules self._logger.info(f"applying data contract filter to {entity_name}.") - if not entity_name.startswith("Original"): - filtered_entity = self._step_implementations.filter_data_contract_record_rejections( - working_directory, - entity, - entity_name, - ) - else: - self._logger.info(f"Skipping {entity_name}. Marked original.") - filtered_entity = entity + filtered_entity = self._step_implementations.filter_data_contract_record_rejections( + working_directory, + entity, + entity_name, + ) projected = self._step_implementations.write_parquet( # type: ignore filtered_entity, fh.joinuri( self.processed_files_path, submission_info.submission_id, - "business_rules", + "temp_business_rules", entity_name, ), ) @@ -636,10 +680,101 @@ def apply_business_rules( # pylint: disable=R0914 projected ) + _, orph_issues_1 = self.step_implementations.identify_and_remove_orphans( # type: ignore + working_directory, + entity_manager.entities, + entity_hierarchy, + key_fields, + ) + + _, grp_issues_1 = self.step_implementations.identify_and_remove_missing_mandatory_groups( # type: ignore + working_directory, + entity_manager.entities, + entity_hierarchy, + key_fields, + ) + + # Perform a second time incase the mandatory groups result in new orphans + _, orph_issues_2 = self.step_implementations.identify_and_remove_orphans( # type: ignore + working_directory, + entity_manager.entities, + entity_hierarchy, + key_fields, + ) + + entity_issues: dict[EntityName, bool] = { + entity: any( + val + for val in ( + orph_issues_1.get(entity, False), + grp_issues_1.get(entity, False), + orph_issues_2.get(entity, False), + ) + ) + for entity in orph_issues_1.keys() + } + + unchanged_entities: list[EntityName] = [] + for entity_name, entity in entity_manager.entities.items(): + if entity_issues.get(entity_name, False): + self._logger.info(f"Writing {entity_name} out to disk.") + final_projection = self._step_implementations.write_parquet( # type: ignore + entity, + fh.joinuri( + self.processed_files_path, + submission_info.submission_id, + "business_rules", + entity_name, + ), + ) + + entity_manager.entities[entity_name] = self.step_implementations.read_parquet( # type: ignore + final_projection + ) + else: + unchanged_entities.append(entity_name) + + for entity_name in unchanged_entities: + self._logger.info(f"Moving {entity_name} from temp_business_rules to business_rules") + final_projection = fh.move_resource( + source_uri=fh.joinuri( + self.processed_files_path, + submission_info.submission_id, + "temp_business_rules", + entity_name, + ), + target_uri=fh.joinuri( + self.processed_files_path, + submission_info.submission_id, + "business_rules", + entity_name, + ), + overwrite=True, + ) + + entity_manager.entities[entity_name] = self.step_implementations.read_parquet( # type: ignore + final_projection + ) + + fh.remove_prefix( + fh.joinuri( + self.processed_files_path, submission_info.submission_id, "temp_business_rules" + ) + ) + + self.check_mandatory_entities_have_records( + working_directory, entity_manager, entity_hierarchy + ) + submission_status.number_of_records = self.get_entity_count( - entity=entity_manager.entities[f"""Original{rules.global_variables.get( - 'entity', - submission_info.dataset_id)}"""] + entity=self.step_implementations.read_parquet( # type: ignore + fh.joinuri( + self.processed_files_path, + submission_info.submission_id, + "data_contract", + rules.global_variables.get('entity', submission_info.dataset_id) + ) + ) ) submission_status.number_of_records_rejected = ( submission_status.number_of_records @@ -776,7 +911,7 @@ def _get_error_dataframes(self, submission_id: str): .alias("error_type") # type: ignore ) df = df.select( - pl.col("Entity").alias("Table"), # type: ignore + pl.col("ReportingEntity").alias("Table"), # type: ignore pl.col("error_type").alias("Type"), # type: ignore pl.col("ErrorCode").alias("Error_Code"), # type: ignore pl.col("ReportingField").alias("Data_Item"), # type: ignore diff --git a/src/dve/pipeline/utils.py b/src/dve/pipeline/utils.py index e6122c2..0b946b1 100644 --- a/src/dve/pipeline/utils.py +++ b/src/dve/pipeline/utils.py @@ -11,9 +11,11 @@ import dve.core_engine.backends.implementations.duckdb # pylint: disable=unused-import import dve.core_engine.backends.implementations.spark # pylint: disable=unused-import import dve.parser.file_handling as fh +from dve.core_engine.backends.exceptions import MessageBearingError from dve.core_engine.backends.readers import _READER_REGISTRY from dve.core_engine.configuration.v1 import SchemaName, V1EngineConfig, _ModelConfig from dve.core_engine.loggers import get_logger +from dve.core_engine.message import FeedbackMessage from dve.core_engine.type_hints import URI, SubmissionResult from dve.metadata_parser.model_generator import JSONtoPyd @@ -52,7 +54,31 @@ def load_reader( backend_reader_kwargs: Optional[dict[str, Any]] = None, ): """Loads the readers for the diven feed, model name and file extension""" - reader_config = dataset[model_name].reader_config[f".{file_extension.lower()}"] + try: + reader_config = dataset[model_name].reader_config[f".{file_extension.lower()}"] + except KeyError as exc: + if file_extension: + err_msg = ( + f"The supplied file extension `{file_extension}`" + + f" is not a supported file format for {model_name}." + ) + else: + err_msg = "No supplied file extension. Unable to parse file without a file extension." + + raise MessageBearingError( + f"The file extension provided ({file_extension}) is not supported for this collection.", + messages=[ + FeedbackMessage( + entity=model_name, + record=None, + failure_type="submission", + error_location="Whole File", + error_code="InvalidFileExtension", + error_message=err_msg, + ) + ], + ) from exc + reader = _READER_REGISTRY[reader_config.reader]( **reader_config.kwargs_, **backend_reader_kwargs if backend_reader_kwargs else {} ) @@ -68,7 +94,7 @@ def unpersist_all_rdds(spark: SparkSession): rdd.unpersist() -def deadletter_file(source_uri: URI) -> None: +def deadletter_file(source_uri: URI) -> URI | None: """Move files that can't be processed to a deadletter location""" try: source_parent: URI = source_uri.rsplit("/", 1)[0] diff --git a/src/dve/reporting/__init__.py b/src/dve/reporting/__init__.py index 9a93c67..ab78e11 100644 --- a/src/dve/reporting/__init__.py +++ b/src/dve/reporting/__init__.py @@ -1 +1,2 @@ """Error reports module.""" +# pylint: disable=R0801 diff --git a/src/dve/reporting/error_report.py b/src/dve/reporting/error_report.py index 95137b5..bf1708a 100644 --- a/src/dve/reporting/error_report.py +++ b/src/dve/reporting/error_report.py @@ -91,7 +91,7 @@ def create_error_dataframe(errors: deque[FeedbackMessage], key_fields): .alias("error_type") ) df = df.select( # type: ignore - col("Entity").alias("Table"), # type: ignore + col("ReportingEntity").alias("Table"), # type: ignore col("error_type").alias("Type"), # type: ignore col("ErrorCode").alias("Error_Code"), # type: ignore col("ReportingField").alias("Data_Item"), # type: ignore diff --git a/src/dve/reporting/excel_report.py b/src/dve/reporting/excel_report.py index 5876cdd..e0c4891 100644 --- a/src/dve/reporting/excel_report.py +++ b/src/dve/reporting/excel_report.py @@ -153,17 +153,21 @@ def _add_submission_info(self, status: str, summary: Worksheet): ), # pylint: disable=C0301 ] ) - if status not in ( + if status in ( ErrorReportStatus.PROCESSING_FAILED, ErrorReportStatus.FILE_REJECTION, ): - summary.append( - [ - "", - "Total Number of Records Rejected", - self.submission_status.number_of_records_rejected, - ] - ) + _records_rejected = self.submission_status.number_of_records + else: + _records_rejected = self.submission_status.number_of_records_rejected + summary.append( + [ + "", + "Total Number of Records Rejected", + _records_rejected, + ] + ) + summary.append(["", ""]) diff --git a/tests/features/flights.feature b/tests/features/flights.feature new file mode 100644 index 0000000..bcfad68 --- /dev/null +++ b/tests/features/flights.feature @@ -0,0 +1,288 @@ +Feature: Pipeline tests using the flights dataset + Test hierarchical record rejection and ensuring that records are removed correctly including + any "orphan" records generated from record removal in parent entities. + + Scenario: A perfect flights file + Given I submit the flights file perfect_flights.xml for processing + And A duckdb pipeline is configured with schema file 'flights.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And the airport entity is stored as a parquet after the file_transformation phase + And the staff entity is stored as a parquet after the file_transformation phase + And the flights entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the latest audit record for the submission is marked with processing status data_contract + When I run the data contract phase + Then there are no file rejections from the data_contract phase + And there are no record rejections from the data_contract phase + When I run the business rules phase + Then there are no file rejections from the business_rules phase + And there are no record rejections from the business_rules phase + And the final entities have the following row counts + | entity_name | row_count | + | country | 1 | + | airport | 3 | + | staff | 15 | + | flights | 10 | + | passengers | 25 | + When I run the error report phase + Then An error report is produced + And The statistics entry for the submission shows the following information + | parameter | value | + | record_count | 1 | + | number_submission_rejections | 0 | + | number_record_rejections | 0 | + | number_warnings | 0 | + + Scenario: A flights submission where the root record is rejected + Given I submit the flights file missing_country_id.xml for processing + And A duckdb pipeline is configured with schema file 'flights.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And the airport entity is stored as a parquet after the file_transformation phase + And the staff entity is stored as a parquet after the file_transformation phase + And the flights entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the latest audit record for the submission is marked with processing status data_contract + When I run the data contract phase + Then there are no file rejections from the data_contract phase + And there is 1 record rejection from the data_contract phase + And there are errors with the following details and associated error_count from the data_contract phase + | FailureType | ErrorCode | error_count | + | record | CountryIdIsMissing | 1 | + When I run the business rules phase + Then there are errors with the following details and associated error_count from the business_rules phase + | ErrorType | ErrorCode | error_count | + | record | AirportHasNoCountry | 3 | + | record | StaffHasNoAirport | 15 | + | record | FlightHasNoAirport | 10 | + | record | PassengerHasNoFlight | 25 | + | submission | NoValidCountries | 1 | + | submission | NoValidAirports | 1 | + | submission | NoValidStaff | 1 | + And the final entities have the following row counts + | entity_name | row_count | + | country | 0 | + | airport | 0 | + | staff | 0 | + | flights | 0 | + | passengers | 0 | + When I run the error report phase + Then An error report is produced + And The statistics entry for the submission shows the following information + | parameter | value | + | record_count | 1 | + | number_submission_rejections | 3 | + | number_record_rejections | 54 | + | number_warnings | 0 | + + Scenario: A flights submission where a child primary key is rejected + Given I submit the flights file missing_flight_id.xml for processing + And A duckdb pipeline is configured with schema file 'flights.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And the airport entity is stored as a parquet after the file_transformation phase + And the staff entity is stored as a parquet after the file_transformation phase + And the flights entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the latest audit record for the submission is marked with processing status data_contract + When I run the data contract phase + Then there are no file rejections from the data_contract phase + And there are no record rejections from the data_contract phase + When I run the business rules phase + Then there are errors with the following details and associated error_count from the business_rules phase + | ErrorType | ErrorCode | error_count | + | record | FlightIDMissing | 1 | + | record | PassengerHasNoFlight | 3 | + And the final entities have the following row counts + | entity_name | row_count | + | country | 1 | + | airport | 3 | + | staff | 15 | + | flights | 9 | + | passengers | 22 | + When I run the error report phase + Then An error report is produced + And The statistics entry for the submission shows the following information + | parameter | value | + | record_count | 1 | + | number_submission_rejections | 0 | + | number_record_rejections | 4 | + | number_warnings | 0 | + + Scenario: A flights submission with only country id and name submitted + Given I submit the flights file only_country_id.xml for processing + And A duckdb pipeline is configured with schema file 'flights.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And the airport entity is stored as a parquet after the file_transformation phase + And the flights entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the latest audit record for the submission is marked with processing status data_contract + When I run the data contract phase + Then there are no file rejections from the data_contract phase + And there are no record rejections from the data_contract phase + When I run the business rules phase + Then there are errors with the following details and associated error_count from the business_rules phase + | ErrorType | ErrorCode | error_count | + | record | CountryHasNoAirport | 1 | + | submission | NoValidCountries | 1 | + | submission | NoValidAirports | 1 | + | submission | NoValidStaff | 1 | + And the final entities have the following row counts + | entity_name | row_count | + | country | 0 | + | airport | 0 | + | staff | 0 | + | flights | 0 | + | passengers | 0 | + When I run the error report phase + Then An error report is produced + And The statistics entry for the submission shows the following information + | parameter | value | + | record_count | 1 | + | number_submission_rejections | 3 | + | number_record_rejections | 1 | + | number_warnings | 0 | + + Scenario: A flights submission with a rejection on a node with one mandatory node + Given I submit the flights file singular_node_rejections.xml for processing + And A duckdb pipeline is configured with schema file 'flights.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And the airport entity is stored as a parquet after the file_transformation phase + And the flights entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the latest audit record for the submission is marked with processing status data_contract + When I run the data contract phase + Then there are no file rejections from the data_contract phase + And there are no record rejections from the data_contract phase + When I run the business rules phase + Then there are errors with the following details and associated error_count from the business_rules phase + | ErrorType | Status | ErrorCode | error_count | + | record | error | InvalidFlightDestination | 2 | + | record | error | PassengerHasNoFlight | 4 | + | record | error | AirportHasNoStaff | 1 | + | record | error | CountryHasNoAirport | 1 | + | submission | error | NoValidCountries | 1 | + | submission | error | NoValidAirports | 1 | + | submission | error | NoValidStaff | 1 | + And the final entities have the following row counts + | entity_name | row_count | + | country | 0 | + | airport | 0 | + | staff | 0 | + | flights | 0 | + | passengers | 0 | + When I run the error report phase + Then An error report is produced + And The statistics entry for the submission shows the following information + | parameter | value | + | record_count | 1 | + | number_submission_rejections | 3 | + | number_record_rejections | 8 | + | number_warnings | 0 | + + Scenario: A flights submission with a rejection on a node with two mandatory nodes + Given I submit the flights file multi_node_file_rejection.xml for processing + And A duckdb pipeline is configured with schema file 'flights.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And the airport entity is stored as a parquet after the file_transformation phase + And the flights entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the latest audit record for the submission is marked with processing status data_contract + When I run the data contract phase + Then there are no file rejections from the data_contract phase + And there are no record rejections from the data_contract phase + When I run the business rules phase + Then there are errors with the following details and associated error_count from the business_rules phase + | ErrorType | Status | ErrorCode | error_count | + | record | error | StaffIDMissing | 7 | + | record | error | AirportHasNoStaff | 1 | + | record | error | FlightHasNoAirport | 1 | + | record | error | PassengerHasNoFlight | 1 | + And the final entities have the following row counts + | entity_name | row_count | + | country | 1 | + | airport | 1 | + | staff | 1 | + | flights | 1 | + | passengers | 1 | + When I run the error report phase + Then An error report is produced + And The statistics entry for the submission shows the following information + | parameter | value | + | record_count | 1 | + | number_submission_rejections | 0 | + | number_record_rejections | 10 | + | number_warnings | 0 | + + Scenario: A flights submission with many types of rejections in a single submission + Given I submit the flights file flights_full_regression.xml for processing + And A duckdb pipeline is configured with schema file 'flights.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And the airport entity is stored as a parquet after the file_transformation phase + And the flights entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the latest audit record for the submission is marked with processing status data_contract + When I run the data contract phase + Then there are errors with the following details and associated error_count from the data_contract phase + | FailureType | ErrorCode | error_count | + | record | AirportIdIsMissing | 1 | + When I run the business rules phase + Then there are errors with the following details and associated error_count from the business_rules phase + | ErrorType | Status | ErrorCode | error_count | + | record | error | InvalidFlightDestination | 1 | + | record | error | PassengerNameMissing | 1 | + | record | error | StaffIDMissing | 4 | + | record | error | PassengerHasNoFlight | 3 | + | record | error | StaffHasNoAirport | 1 | + | record | error | FlightHasNoAirport | 2 | + | record | error | AirportHasNoStaff | 1 | + And the final entities have the following row counts + | entity_name | row_count | + | country | 1 | + | airport | 3 | + | staff | 3 | + | flights | 3 | + | passengers | 2 | + When I run the error report phase + Then An error report is produced + And The statistics entry for the submission shows the following information + | parameter | value | + | record_count | 1 | + | number_submission_rejections | 0 | + | number_record_rejections | 14 | + | number_warnings | 0 | + + Scenario: A flights submission where mandatory entity has no records submitted + Given I submit the flights file only_country_id.xml for processing + And A duckdb pipeline is configured with schema file 'flights_add_reader_checks.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And there are errors with the following details and associated error_count from the file_transformation phase + | FailureType | ErrorCode | error_count | + | submission | AIRPORTEMPTY | 1 | + And the latest audit record for the submission is marked with processing status error_report + When I run the error report phase + Then An error report is produced diff --git a/tests/features/movies.feature b/tests/features/movies.feature index 750975e..c6d5ad6 100644 --- a/tests/features/movies.feature +++ b/tests/features/movies.feature @@ -22,7 +22,7 @@ Feature: Pipeline tests using the movies dataset Then there is 1 submission rejection from the data_contract phase And there are 3 record rejections from the data_contract phase And there are errors with the following details and associated error_count from the data_contract phase - | Entity | ErrorCode | ErrorMessage | RecordIndex | error_count | + | ReportingEntity | ErrorCode | ErrorMessage | RecordIndex | error_count | | movies | BLANKYEAR | year not provided | 2 | 1 | | movies_rename_test | DODGYYEAR | year value (NOT_A_NUMBER) is invalid | 1 | 1 | | movies | DODGYDATE | date_joined value is not valid: daft_date | 1 | 1 | @@ -61,10 +61,10 @@ Feature: Pipeline tests using the movies dataset Then there is 1 submission rejection from the data_contract phase And there are 3 record rejections from the data_contract phase And there are errors with the following details and associated error_count from the data_contract phase - | Entity | ErrorCode | ErrorMessage | RecordIndex | error_count | - | movies | BLANKYEAR | year not provided | 2 | 1 | - | movies_rename_test | DODGYYEAR | year value (NOT_A_NUMBER) is invalid | 1 | 1 | - | movies | DODGYDATE | date_joined value is not valid: daft_date | 1 | 1 | + | ReportingEntity | ErrorCode | ErrorMessage | RecordIndex | error_count | + | movies | BLANKYEAR | year not provided | 2 | 1 | + | movies_rename_test | DODGYYEAR | year value (NOT_A_NUMBER) is invalid | 1 | 1 | + | movies | DODGYDATE | date_joined value is not valid: daft_date | 1 | 1 | | movies | BLANKTITLE | title should not be blank | 4 | 1 | And the movies entity is stored as a parquet after the data_contract phase And the latest audit record for the submission is marked with processing status business_rules diff --git a/tests/features/planets.feature b/tests/features/planets.feature index b37b60b..0c4b21d 100644 --- a/tests/features/planets.feature +++ b/tests/features/planets.feature @@ -43,7 +43,9 @@ Feature: Pipeline tests using the planets dataset And I add initial audit entries for the submission Then the latest audit record for the submission is marked with processing status file_transformation When I run the file transformation phase - Then the latest audit record for the submission is marked with processing status failed + Then the latest audit record for the submission is marked with processing status error_report + When I run the error report phase + Then An error report is produced Scenario: Handle a file with duplicated extension provided (spark) Given I submit the planets file planets.csv.csv for processing diff --git a/tests/features/steps/steps_pipeline.py b/tests/features/steps/steps_pipeline.py index c72c873..2bfcf1a 100644 --- a/tests/features/steps/steps_pipeline.py +++ b/tests/features/steps/steps_pipeline.py @@ -33,8 +33,8 @@ from utilities import ( load_errors_from_service, - get_test_file_path, SERVICE_TO_STORAGE_PATH_MAPPING, + get_test_file_path, get_all_errors_df, ) @@ -183,8 +183,10 @@ def check_error_record_details_from_service(context: Context, service:str): message_df = load_errors_from_service(processing_path, service) for err_details in error_details: filter_expr, error_count = err_details - assert message_df.filter(filter_expr).shape[0] == error_count - + assert message_df.filter(filter_expr).shape[0] == error_count, message_df.select( + *[pl.col(c) for c in table.headings if c not in ["error_count"]] + ) + @given("A {implementation} pipeline is configured") @given("A {implementation} pipeline is configured with schema file '{schema_file_name}'") @@ -282,7 +284,7 @@ def check_rows_removed_with_error_code(context: Context, entity_name: str, error err_df = get_all_errors_df(context) recs_with_err_code = err_df.filter( - (pl.col("Entity").eq(entity_name)) & (pl.col("ErrorCode").eq(error_code)) + (pl.col("ReportingEntity").eq(entity_name)) & (pl.col("ErrorCode").eq(error_code)) ).shape[0] assert recs_with_err_code >= 1 @@ -317,8 +319,3 @@ def create_refdata_tables(context: Context, database: str): pipeline._connection.sql(f"ATTACH '{ref_db_file}' AS {database}") for tbl, source in refdata_tables.items(): pipeline._connection.read_parquet(source).to_table(f"{database}.{tbl}") - - - - - diff --git a/tests/features/steps/steps_post_pipeline.py b/tests/features/steps/steps_post_pipeline.py index 906445e..b284b3c 100644 --- a/tests/features/steps/steps_post_pipeline.py +++ b/tests/features/steps/steps_post_pipeline.py @@ -110,10 +110,24 @@ def check_stats_record(context): stats = ( get_pipeline(context)._audit_tables.get_submission_statistics(sub_info.submission_id).model_dump() ) - assert all([val == stats.get(fld) for fld, val in expected.items()]) + assert all([val == stats.get(fld) for fld, val in expected.items()]), stats @then("the error aggregates are persisted") def check_error_aggregates_persisted(context): processing_location = get_processing_location(context) agg_file = Path(processing_location, "audit", "error_aggregates.parquet") assert agg_file.exists() and agg_file.is_file() + +@then("the final entities have the following row counts") +def check_entity_row_counts(context: Context): + processing_loc = get_processing_location(context) + submission_info = get_submission_info(context) + table: Table = context.table + if table is None: + raise ValueError("No table supplied in step") + for row in table: + record = row.as_dict() + entity_name = record["entity_name"] + expected_count = int(record["row_count"]) + output_df = read_output_parquet(processing_loc, entity_name, "business_rules") + assert expected_count == output_df.shape[0], output_df diff --git a/tests/features/steps/utilities.py b/tests/features/steps/utilities.py index 58edc67..95b73ad 100644 --- a/tests/features/steps/utilities.py +++ b/tests/features/steps/utilities.py @@ -15,7 +15,7 @@ from dve.parser.type_hints import URI ERROR_DF_FIELDS: List[str] = [ - "Entity", + "ReportingEntity", "Key", "ErrorCode", "FailureType", @@ -49,7 +49,7 @@ def load_errors_from_service(processing_folder: Path, service: str) -> pl.DataFr err_location = Path( processing_folder, "errors", - f"{SERVICE_TO_STORAGE_PATH_MAPPING.get(service, service)}_errors.jsonl", + f"{service}_errors.jsonl", ) msgs = [] try: diff --git a/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_data_contract.py b/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_data_contract.py index 2019a66..2e6bf87 100644 --- a/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_data_contract.py +++ b/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_data_contract.py @@ -381,4 +381,4 @@ def test_duckdb_data_contract_custom_error_details(nested_all_string_parquet_w_e assert messages[0].ErrorMessage == "subfield id is invalid: subfield.id - WRONG" assert messages[1].ErrorCode == "TESTIDBAD" assert messages[1].ErrorMessage == "id is invalid: id - WRONG" - assert messages[1].Entity == "test_rename" \ No newline at end of file + assert messages[1].ReportingEntity == "test_rename" \ No newline at end of file diff --git a/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_duckdb_helpers.py b/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_duckdb_helpers.py index 4eefe05..0e81cf4 100644 --- a/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_duckdb_helpers.py +++ b/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_duckdb_helpers.py @@ -76,7 +76,8 @@ def example_data_contract_error_codes(temp_ddb_conn): test_entity = con.sql("SELECT * FROM test_df") error_contract_messages = [ { - "Entity": "test_entity", + "ReportingEntity": "test_entity", + "OriginalEntity": "test_entity", "Key": "", "FailureType": "record", "Status": "error", @@ -90,7 +91,8 @@ def example_data_contract_error_codes(temp_ddb_conn): "Category": "Bad value" }, { - "Entity": "test_entity", + "ReportingEntity": "test_entity", + "OriginalEntity": "test_entity", "Key": "", "FailureType": "record", "Status": "error", diff --git a/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_rules.py b/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_rules.py index 35007a9..ecef834 100644 --- a/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_rules.py +++ b/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_rules.py @@ -1,6 +1,7 @@ """Test DuckDB backend steps.""" # pylint: disable=redefined-outer-name,unused-import,line-too-long +import tempfile from pathlib import Path from typing import Iterator, List, Optional, Set, Tuple, Type @@ -39,6 +40,9 @@ SemiJoin, TableUnion, ) +from dve.core_engine.configuration.v1.hierarchy import ( + EntityHierarchy, HierarchyNode +) from dve.core_engine.type_hints import MultipleExpressions from tests.test_core_engine.test_backends.fixtures import ( duckdb_connection, @@ -581,91 +585,106 @@ def test_header_multi_rows_raises( DUCKDB_STEP_BACKEND.join_header(entities, config=header_join) -def test_orphans_planets_satellites( - planets_rel: DuckDBPyRelation, largest_satellites_rel: DuckDBPyRelation -): - """Test a basic orphan idenfitication from satellites to planets.""" - # Each satellite _must_ have a planet. - join = OrphanIdentification( - entity_name="satellites", - target_name="planets", - join_condition="satellites.planet == planets.planet", - ) - entities = EntityManager( - { - "planets": planets_rel.filter(ColumnExpression("Planet") != ConstantExpression("Mars")), - "satellites": largest_satellites_rel, - } - ) - - DUCKDB_STEP_BACKEND.evaluate(entities, config=join) - actual_rel = ( - entities["satellites"] - .filter(ColumnExpression("IsOrphaned")) - .select(ColumnExpression("name")) - ) - actual_rows = sorted(actual_rel.df().to_dict(orient="records"), key=lambda row: row["name"]) - - expected_rel = largest_satellites_rel.filter( - ColumnExpression("Planet") == ConstantExpression("Mars") - ).select(ColumnExpression("name")) - expected_rows = sorted(expected_rel.df().to_dict(orient="records"), key=lambda row: row["name"]) - - assert actual_rows == expected_rows - - -def test_chained_orphans_planets_satellites( - planets_rel: DuckDBPyRelation, largest_satellites_rel: DuckDBPyRelation -): - """Test a basic chained orphan idenfitication from satellites to planets.""" - join = OrphanIdentification( - entity_name="satellites", - target_name="planets", - join_condition="satellites.planet == planets.planet", - ) - entities = EntityManager( - { - "planets": planets_rel.filter(ColumnExpression("planet") != ConstantExpression("Mars")), - "satellites": largest_satellites_rel, - } - ) - DUCKDB_STEP_BACKEND.evaluate(entities, config=join) - entities["planets"] = planets_rel.filter( - ColumnExpression("planet") != ConstantExpression("Earth") - ) - DUCKDB_STEP_BACKEND.evaluate(entities, config=join) - - actual_rel = ( - entities["satellites"] - .filter(ColumnExpression("IsOrphaned")) - .select(ColumnExpression("name")) - ) - actual_rows = sorted(actual_rel.df().to_dict(orient="records"), key=lambda row: row["name"]) - - expected_rel = largest_satellites_rel.filter( - ColumnExpression("Planet").isin(ConstantExpression("Mars"), ConstantExpression("Earth")) - ).select(ColumnExpression("name")) - expected_rows = sorted(expected_rel.df().to_dict(orient="records"), key=lambda row: row["name"]) - - assert actual_rows == expected_rows - - -def test_orphans_missing_entities_raises( - planets_rel: DuckDBPyRelation, satellites_rel: DuckDBPyRelation -): - """Test that trying to join orphans from missing entities raises correctly.""" - join = OrphanIdentification( - entity_name="planets", - target_name="satellites", - join_condition="planets.planet == satellites.planet", - ) +class TestOrphanRecords: + """ + Check that Orphan records identification and removal is working as expected. - entities = EntityManager({"planets": planets_rel}) - with pytest.raises(MissingEntity): - DUCKDB_STEP_BACKEND.identify_orphans(entities, config=join) - entities = EntityManager({"satellites": satellites_rel}) - with pytest.raises(MissingEntity): - DUCKDB_STEP_BACKEND.identify_orphans(entities, config=join) + Current scenarios are: + Flight ID 1 = Perfect Record - no orphans + Flight ID 2 = Record rejected at the flights entity, therefore, two expected orphans in passengers and food entities. + """ + mod_flights_df = pl.DataFrame([ + {'flight_id': 1, '__record_index__': 1}, + ]) + mod_passengers_df = pl.DataFrame([ + {'flight_id': 1, 'passenger_id': 1, '__record_index__': 1}, + {'flight_id': 1, 'passenger_id': 2, '__record_index__': 2}, + {'flight_id': 2, 'passenger_id': 3, '__record_index__': 3}, + ]) + mod_food_df = pl.DataFrame([ + {'passenger_id': 1, 'food_id': 1, '__record_index__': 1}, + {'passenger_id': 1, 'food_id': 2, '__record_index__': 2}, + {'passenger_id': 3, 'food_id': 3, '__record_index__': 3}, + ]) + + def test_identify_orphan_record_single_entity(self): + """Ensure that a single one-to-one check works to identify orphan records.""" + with duckdb.connect() as cnn: + cnn.register("mod_flights", self.mod_flights_df) + cnn.register("mod_passengers", self.mod_passengers_df) + + mod_entities = EntityManager( + entities={ + "flights": cnn.sql("SELECT * FROM mod_flights"), + "passengers": cnn.sql("SELECT * FROM mod_passengers"), + } + ) + + rules = DuckDBStepImplementations(connection=cnn) + msgs = rules.identify_orphans( + mod_entities.entities, + config=OrphanIdentification( + id="flight_id", + entity_name="passengers", + target_name="flights", + join_condition="passengers.flight_id = flights.flight_id" + ) + ) + assert len(list(msgs)) == 1 + + def test_identify_and_remove_orphans(self): + with duckdb.connect() as cnn: + cnn.register("mod_flights", self.mod_flights_df) + cnn.register("mod_passengers", self.mod_passengers_df) + cnn.register("mod_food", self.mod_food_df) + + mod_entities = EntityManager( + entities={ + "flights": cnn.sql("SELECT * FROM mod_flights"), + "passengers": cnn.sql("SELECT * FROM mod_passengers"), + "food": cnn.sql("SELECT * FROM mod_food"), + } + ) + + rules = DuckDBStepImplementations(connection=cnn) + hierarchy = EntityHierarchy({ + "flights": HierarchyNode( + entity_name="flights", + children=[ + HierarchyNode( + entity_name="passengers", + parent_entity="flights", + children=[ + HierarchyNode( + entity_name="food", + parent_entity="passengers", + children=[], + join_fields={"passenger_id": "passenger_id"}, + mandatory=False + ) + ], + join_fields={"flight_id": "flight_id"}, + mandatory=True + ) + ] + ) + }) + + with tempfile.TemporaryDirectory() as wd: + rules.identify_and_remove_orphans( + wd, + mod_entities.entities, + hierarchy + ) + + flights_rel = mod_entities["flights"] + assert flights_rel.select("__record_index__").unique("*").count("*").fetchone()[0] == 1 # type: ignore + + passenger_rel = mod_entities["passengers"] + assert passenger_rel.select("__record_index__").unique("*").count("*").fetchone()[0] == 2 # type: ignore + + food_rel = mod_entities["food"] + assert food_rel.select("__record_index__").unique("*").count("*").fetchone()[0] == 2 # type: ignore def test_has_match_planets_satellites( @@ -798,6 +817,25 @@ def test_planets_notify(planets_rel: DuckDBPyRelation): assert len(messages[0]) == 4 +def test_notify_null_errors(planets_rel: DuckDBPyRelation): + + config = Notification( + entity_name="planets", + expression="CASE WHEN planet=='Mercury' THEN NULL ELSE False END", + excluded_columns=["mass", "diameter"], + reporting=ReportingConfig( + code="TESTNULLERROR", message="this is a test", location="planet, has_ring_system" + ), + error_if_expression_null=True + ) + entities = EntityManager({"planets": planets_rel}) + messages, success = DUCKDB_STEP_BACKEND.evaluate(entities, config=config) + + assert not success + assert len(messages) == 1 + assert messages[0].is_critical + + def test_read_and_write_simple_parquet(simple_typecast_parquet): parquet_uri, data = simple_typecast_parquet entity: DuckDBPyRelation = DUCKDB_STEP_BACKEND.read_parquet(path=parquet_uri) diff --git a/tests/test_core_engine/test_backends/test_implementations/test_spark/test_data_contract.py b/tests/test_core_engine/test_backends/test_implementations/test_spark/test_data_contract.py index 70c6b9c..ea5fc46 100644 --- a/tests/test_core_engine/test_backends/test_implementations/test_spark/test_data_contract.py +++ b/tests/test_core_engine/test_backends/test_implementations/test_spark/test_data_contract.py @@ -242,6 +242,6 @@ def test_spark_data_contract_custom_error_details(nested_all_string_parquet_w_er assert messages[0].ErrorMessage == "subfield id is invalid: subfield.id - WRONG" assert messages[1].ErrorCode == "TESTIDBAD" assert messages[1].ErrorMessage == "id is invalid: id - WRONG" - assert messages[1].Entity == "test_rename" + assert messages[1].ReportingEntity == "test_rename" \ No newline at end of file diff --git a/tests/test_core_engine/test_backends/test_implementations/test_spark/test_rules.py b/tests/test_core_engine/test_backends/test_implementations/test_spark/test_rules.py index 673e611..6e38c7a 100644 --- a/tests/test_core_engine/test_backends/test_implementations/test_spark/test_rules.py +++ b/tests/test_core_engine/test_backends/test_implementations/test_spark/test_rules.py @@ -21,6 +21,7 @@ from dve.core_engine.backends.base.core import EntityManager from dve.core_engine.backends.exceptions import MissingEntity from dve.core_engine.backends.implementations.spark.rules import SparkStepImplementations +from dve.core_engine.backends.metadata.reporting import ReportingConfig from dve.core_engine.backends.metadata.rules import ( Aggregation, AntiJoin, @@ -32,6 +33,7 @@ HeaderJoin, InnerJoin, LeftJoin, + Notification, OneToOneJoin, OrphanIdentification, RenameEntity, @@ -447,6 +449,24 @@ def test_join_can_take_all_cols( expected_rows = sorted(expected_df.collect(), key=lambda row: row.planet) assert actual_rows == expected_rows + +def test_notify_null_errors(planets_df: DataFrame): + + config = Notification( + entity_name="planets", + expression="CASE WHEN planet=='Mercury' THEN NULL ELSE False END", + excluded_columns=["mass", "diameter"], + reporting=ReportingConfig( + code="TESTNULLERROR", message="this is a test", location="planet, has_ring_system" + ), + error_if_expression_null=True + ) + entities = EntityManager({"planets": planets_df}) + messages, success = SPARK_STEP_BACKEND.evaluate(entities, config=config) + + assert not success + assert len(messages) == 1 + assert messages[0].is_critical def test_one_to_one_join_multi_matches_raises(planets_df: DataFrame, satellites_df: DataFrame): @@ -568,6 +588,7 @@ def test_header_multi_rows_raises(planets_df: DataFrame, value_literal_1_header: SPARK_STEP_BACKEND.join_header(entities, config=header_join) +@pytest.mark.skip(reason="Logic is no longer valid") def test_orphans_planets_satellites(planets_df: DataFrame, largest_satellites_df: DataFrame): """Test a basic orphan idenfitication from satellites to planets.""" # Each satellite _must_ have a planet. @@ -593,6 +614,7 @@ def test_orphans_planets_satellites(planets_df: DataFrame, largest_satellites_df assert actual_rows == expected_rows +@pytest.mark.skip(reason="Logic is no longer valid") def test_chained_orphans_planets_satellites( planets_df: DataFrame, largest_satellites_df: DataFrame ): @@ -623,6 +645,7 @@ def test_chained_orphans_planets_satellites( assert actual_rows == expected_rows +@pytest.mark.skip(reason="Logic is no longer valid") def test_orphans_missing_entities_raises(planets_df: DataFrame, satellites_df: DataFrame): """Test that trying to join orphans from missing entities raises correctly.""" join = OrphanIdentification( diff --git a/tests/test_core_engine/test_backends/test_implementations/test_spark/test_spark_helpers.py b/tests/test_core_engine/test_backends/test_implementations/test_spark/test_spark_helpers.py index e7a37eb..41c9e93 100644 --- a/tests/test_core_engine/test_backends/test_implementations/test_spark/test_spark_helpers.py +++ b/tests/test_core_engine/test_backends/test_implementations/test_spark/test_spark_helpers.py @@ -63,7 +63,7 @@ def example_data_contract_error_codes(spark: SparkSession): ]) error_contract_messages = [ { - "Entity": "test_entity", + "OriginalEntity": "test_entity", "Key": "", "FailureType": "record", "Status": "error", @@ -77,7 +77,7 @@ def example_data_contract_error_codes(spark: SparkSession): "Category": "Bad value" }, { - "Entity": "test_entity", + "OriginalEntity": "test_entity", "Key": "", "FailureType": "record", "Status": "error", diff --git a/tests/test_core_engine/test_hierarchy.py b/tests/test_core_engine/test_hierarchy.py new file mode 100644 index 0000000..bbce74f --- /dev/null +++ b/tests/test_core_engine/test_hierarchy.py @@ -0,0 +1,396 @@ +import json +import pytest +from tempfile import NamedTemporaryFile +from dve.core_engine.configuration.v1 import V1EngineConfig +from dve.core_engine.configuration.v1.hierarchy import EntityHierarchy + +CONFIG_WITHOUT_LINKAGE = """{ + "contract": { + "schemas": {}, + "datasets": { + "animals": { + "fields": { + "name": "str", + "height": "float", + "weight": "float", + "region": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "animal", + "root_tag": "animals" + } + } + }, + "mandatory_fields": [ + "name" + ] + } + } + }, + "transformations": { + "filters": [ + { + "entity": "animals", + "name": "check_valid_region", + "expression": "lower(region) in ('africa', 'asia')", + "error_code": "ANE01", + "failure_message": "Record rejected - `{{ region }}` is not in a valid region." + }, + { + "entity": "animals", + "name": "check_for_pets", + "expression": "lower(name) != 'human'", + "error_code": "ANE02", + "failure_message": "Submission Rejected - 'Human' is not a valid animal.", + "failure_type": "submission" + }, + { + "entity": "animals", + "name": "check_valid_weight", + "expression": "weight > 0", + "error_code": "ANE03", + "failure_message": "Warning - `{{ weight }}` is below zero.", + "is_informational": true + } + ] + } +}""" + +CONFIG_WITH_LINKAGE = """{ + "contract": { + "schemas": {}, + "datasets": { + "ds_001": { + "fields": { + "ds_001_id": "str", + "patient_id": "str", + "address": "str", + "name": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "001", + "root_tag": "header" + } + } + }, + "mandatory_fields": [ + "ds_001_id", + "patient_id", + "address", + "name" + ] + }, + "ds_002": { + "fields": { + "ds_002_id": "str", + "gp_name": "str", + "gp_address": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "002", + "root_tag": "header" + } + } + }, + "mandatory_fields": [ + "ds_002_id", + "gp_name", + "gp_address" + ] + }, + "ds_003": { + "fields": { + "ds_003_id": "str", + "ds_001_id": "str", + "total_income": "int" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "003", + "root_tag": "header" + } + } + }, + "mandatory_fields": [ + "ds_003_id", + "ds_001_id" + ] + }, + "ds_101": { + "fields": { + "ds_001_id": "str", + "referral_id": "int", + "consultant_name": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "101", + "root_tag": "header" + } + } + }, + "mandatory_fields": [ + "referral_id", + "ds_001_id" + ] + }, + "ds_201": { + "fields": { + "ds_201_id": "str", + "ds_101_id": "str", + "contact_date": "date" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "201", + "root_tag": "header" + } + } + }, + "mandatory_fields": [ + "ds_101_id", + "ds_201_id", + "contact_date" + ] + }, + "ds_202": { + "fields": { + "ds_202_id": "str", + "ds_201_id": "str", + "contact_name": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "202", + "root_tag": "header" + } + } + }, + "mandatory_fields": [ + "ds_202_id", + "ds_201_id" + ] + } + } + }, + "transformations": { + "filters": [ + { + "entity": "001", + "name": "check_name", + "expression": "len(name) > 2", + "error_code": "CHECK1", + "failure_message": "Record rejected - `{{ name }}` is not valid." + } + ] + }, + "entity_relationships": { + "ds_003": { + "parent_entity": "ds_001", + "join_fields": {"ds_001_id": "ds_001_id"}, + "mandatory": false, + "missing_parent_id_error_code": "DS003NoParent", + "missing_parent_id_error_message": "record removed as no parent" + }, + "ds_101": { + "parent_entity": "ds_001", + "join_fields": {"ds_001_id": "ds_001_id"}, + "mandatory_entity": true, + "no_valid_records_error_code": "DS101NOVALIDRECS", + "no_valid_records_error_message": "{{ ds_001_id }} removed as no valid ds_101 records", + "missing_parent_id_error_code": "DS101NoParent", + "missing_parent_id_error_message": "record removed as no parent" + }, + "ds_201": { + "parent_entity": "ds_101", + "join_fields": {"referral_id": "ds_101_id"}, + "mandatory": false, + "missing_parent_id_error_code": "DS201NoParent", + "missing_parent_id_error_message": "record removed as no parent" + }, + "ds_202": { + "parent_entity": "ds_201", + "join_fields": {"ds_201_id": "ds_201_id"}, + "mandatory": true + } + } +}""" + +def test_no_linkage_config_load(): + config = V1EngineConfig(location="", + **json.loads(CONFIG_WITHOUT_LINKAGE)) + assert len(config.contract.datasets) == 1 + hierarchy = EntityHierarchy.from_engine_config(config) + assert len(hierarchy.entity_trees) == 1 + assert not hierarchy.entity_trees.get("animals").children + + +def test_linkage_config_load(): + config = V1EngineConfig(location="", + **json.loads(CONFIG_WITH_LINKAGE)) + assert len(config.contract.datasets) == 6 + with NamedTemporaryFile("w") as tmp: + tmp.write(CONFIG_WITH_LINKAGE) + tmp.flush() + hierarchy = EntityHierarchy.from_dischema(tmp.name) + assert len(hierarchy.entity_trees) == 2 + assert not hierarchy.entity_trees.get("ds_002").children + assert len(hierarchy.entity_trees.get("ds_001").get_descendents()) == 4 + children_001 = sorted(hierarchy.entity_trees.get("ds_001").children, key=lambda x: x.entity_name) + dict_rep_001 = hierarchy.entity_trees.get("ds_001").as_dict() + assert len(children_001) == 2 + assert children_001[0].entity_name == "ds_003" + assert not children_001[0].children + assert children_001[1].entity_name == "ds_101" + assert dict_rep_001 == json.loads(""" +{ + "ds_001": { + "parent_entity": null, + "join_fields": {}, + "mandatory": false, + "no_valid_records_error_code": "NoValidRecords", + "no_valid_records_error_message": "parent record removed as no valid child records", + "missing_parent_id_error_code": null, + "missing_parent_id_error_message": null, + "empty_entity_error_code": null, + "empty_entity_error_message": null, + "children": { + "ds_003": { + "parent_entity": "ds_001", + "join_fields": { + "ds_001_id": "ds_001_id" + }, + "mandatory": false, + "no_valid_records_error_code": "NoValidRecords", + "no_valid_records_error_message": "parent record removed as no valid child records", + "missing_parent_id_error_code": "DS003NoParent", + "missing_parent_id_error_message": "record removed as no parent", + "empty_entity_error_code": null, + "empty_entity_error_message": null, + "children": {} + }, + "ds_101": { + "parent_entity": "ds_001", + "join_fields": { + "ds_001_id": "ds_001_id" + }, + "mandatory": false, + "no_valid_records_error_code": "DS101NOVALIDRECS", + "no_valid_records_error_message": "{{ ds_001_id }} removed as no valid ds_101 records", + "missing_parent_id_error_code": "DS101NoParent", + "missing_parent_id_error_message": "record removed as no parent", + "empty_entity_error_code": null, + "empty_entity_error_message": null, + "children": { + "ds_201": { + "parent_entity": "ds_101", + "join_fields": { + "referral_id": "ds_101_id" + }, + "mandatory": false, + "no_valid_records_error_code": "NoValidRecords", + "no_valid_records_error_message": "parent record removed as no valid child records", + "missing_parent_id_error_code": "DS201NoParent", + "missing_parent_id_error_message": "record removed as no parent", + "empty_entity_error_code": null, + "empty_entity_error_message": null, + "children": { + "ds_202": { + "parent_entity": "ds_201", + "join_fields": { + "ds_201_id": "ds_201_id" + }, + "mandatory": true, + "no_valid_records_error_code": "NoValidRecords", + "no_valid_records_error_message": "parent record removed as no valid child records", + "missing_parent_id_error_code": "MissingParentRecord", + "missing_parent_id_error_message": "Records removed due to no valid parent record", + "empty_entity_error_code": "EmptyEntity", + "empty_entity_error_message": "no valid records remaining", + "children": {} + } + } + } + } + } + } + } + }""" + ) + + dict_rep_101 = dict_rep_001["ds_001"]["children"]["ds_101"] + children_101 = children_001[1].children + assert len(children_101) == 1 + assert children_101[0].entity_name == "ds_201" + assert children_101[0].children[0].entity_name == "ds_202" + assert not children_101[0].children[0].children + assert dict_rep_101 == json.loads(""" + { "parent_entity": "ds_001", + "join_fields": { + "ds_001_id": "ds_001_id" + }, + "mandatory": false, + "no_valid_records_error_code": "DS101NOVALIDRECS", + "no_valid_records_error_message": "{{ ds_001_id }} removed as no valid ds_101 records", + "missing_parent_id_error_code": "DS101NoParent", + "missing_parent_id_error_message": "record removed as no parent", + "empty_entity_error_code": null, + "empty_entity_error_message": null, + "children": { + "ds_201": { + "parent_entity": "ds_101", + "join_fields": { + "referral_id": "ds_101_id" + }, + "mandatory": false, + "no_valid_records_error_code": "NoValidRecords", + "no_valid_records_error_message": "parent record removed as no valid child records", + "missing_parent_id_error_code": "DS201NoParent", + "missing_parent_id_error_message": "record removed as no parent", + "empty_entity_error_code": null, + "empty_entity_error_message": null, + "children": { + "ds_202": { + "parent_entity": "ds_201", + "join_fields": { + "ds_201_id": "ds_201_id" + }, + "mandatory": true, + "no_valid_records_error_code": "NoValidRecords", + "no_valid_records_error_message": "parent record removed as no valid child records", + "missing_parent_id_error_code": "MissingParentRecord", + "missing_parent_id_error_message": "Records removed due to no valid parent record", + "empty_entity_error_code": "EmptyEntity", + "empty_entity_error_message": "no valid records remaining", + "children": {} + } + } + } + } + }""") + + +def test_get_all_mandatory_nodes(): + with NamedTemporaryFile("w") as tmp: + tmp.write(CONFIG_WITH_LINKAGE) + tmp.flush() + hierarchy = EntityHierarchy.from_dischema(tmp.name) + + assert len(hierarchy.get_all_mandatory_nodes()) == 1 diff --git a/tests/test_pipeline/pipeline_helpers.py b/tests/test_pipeline/pipeline_helpers.py index b13bef3..efed6e2 100644 --- a/tests/test_pipeline/pipeline_helpers.py +++ b/tests/test_pipeline/pipeline_helpers.py @@ -372,7 +372,7 @@ def error_data_after_business_rules() -> Iterator[Tuple[SubmissionInfo, str]]: error_data = json.loads( """[ { - "Entity": "planets", + "ReportingEntity": "planets", "Key": "", "FailureType": "record", "Status": "error", @@ -386,7 +386,7 @@ def error_data_after_business_rules() -> Iterator[Tuple[SubmissionInfo, str]]: "RecordIndex": "1" }, { - "Entity": "planets", + "ReportingEntity": "planets", "Key": "", "FailureType": "record", "Status": "error", diff --git a/tests/test_pipeline/test_foundry_ddb_pipeline.py b/tests/test_pipeline/test_foundry_ddb_pipeline.py index 9b7b60d..f84073a 100644 --- a/tests/test_pipeline/test_foundry_ddb_pipeline.py +++ b/tests/test_pipeline/test_foundry_ddb_pipeline.py @@ -102,7 +102,7 @@ def test_foundry_runner_validation_success(movies_test_files, temp_ddb_conn): ) output_loc, report_uri, audit_files = dve_pipeline.run_pipeline(sub_info) assert fh.get_resource_exists(report_uri) - assert len(list(fh.iter_prefix(output_loc))) == 2 + assert len(list(fh.iter_prefix(output_loc))) == 1 assert len(list(fh.iter_prefix(audit_files))) == 3 def test_foundry_runner_error(planet_test_files, temp_ddb_conn): @@ -197,7 +197,7 @@ def test_foundry_runner_with_submitted_files_path(movies_test_files, temp_ddb_co assert Path(processing_folder, sub_id, sub_info.file_name_with_ext).exists() assert fh.get_resource_exists(report_uri) - assert len(list(fh.iter_prefix(output_loc))) == 2 + assert len(list(fh.iter_prefix(output_loc))) == 1 assert len(list(fh.iter_prefix(audit_files))) == 3 diff --git a/tests/test_pipeline/test_pipeline_utils.py b/tests/test_pipeline/test_pipeline_utils.py new file mode 100644 index 0000000..fc28306 --- /dev/null +++ b/tests/test_pipeline/test_pipeline_utils.py @@ -0,0 +1,36 @@ +from dve.core_engine.backends.exceptions import MessageBearingError +from dve.core_engine.configuration.v1 import _ModelConfig, _ReaderConfig +from dve.pipeline.utils import load_reader + +import pytest + + +class TestLoadReader: + test_model_config = _ModelConfig( + fields={"test": "str"}, + reporting_fields=["test"], + key_field="test", + reader_config={ + ".csv": _ReaderConfig(reader="TestCsvReader"), + } + ) + + def test_invalid_load_reader_with_file_ext(self): + with pytest.raises(MessageBearingError) as exc_info: + load_reader( + {"test": self.test_model_config}, + "test_model", + "jpeg" + ) + + assert exc_info.value.messages[0].error_message == "The supplied file extension `jpeg` is not a supported file format for test_model." + + def test_invalid_load_reader_missing_file_ext(self): + with pytest.raises(MessageBearingError) as exc_info: + load_reader( + {"test": self.test_model_config}, + "test_model", + "" + ) + + assert exc_info.value.messages[0].error_message == "No supplied file extension. Unable to parse file without a file extension." diff --git a/tests/test_pipeline/test_spark_pipeline.py b/tests/test_pipeline/test_spark_pipeline.py index dd28e26..642a7ea 100644 --- a/tests/test_pipeline/test_spark_pipeline.py +++ b/tests/test_pipeline/test_spark_pipeline.py @@ -162,7 +162,7 @@ def test_apply_data_contract_failed( # pylint: disable=redefined-outer-name expected_errors = [ { - "Entity": "planets", + "ReportingEntity": "planets", "Key": "", "FailureType": "record", "Status": "error", @@ -176,7 +176,7 @@ def test_apply_data_contract_failed( # pylint: disable=redefined-outer-name "Category": "Bad value", }, { - "Entity": "planets", + "ReportingEntity": "planets", "Key": "", "FailureType": "record", "Status": "error", @@ -190,7 +190,7 @@ def test_apply_data_contract_failed( # pylint: disable=redefined-outer-name "Category": "Bad value", }, { - "Entity": "planets", + "ReportingEntity": "planets", "Key": "", "FailureType": "record", "Status": "error", @@ -274,12 +274,6 @@ def test_apply_business_rules_success( assert largest_satellites_entity_path.exists() assert spark.read.parquet(str(largest_satellites_entity_path)).count() == 1 - og_planets_entity_path = Path( - Path(processed_file_path), sub_info.submission_id, "business_rules", "Originalplanets" - ) - assert og_planets_entity_path.exists() - assert spark.read.parquet(str(og_planets_entity_path)).count() == 1 - def test_apply_business_rules_with_data_errors( # pylint: disable=redefined-outer-name spark: SparkSession, @@ -317,16 +311,12 @@ def test_apply_business_rules_with_data_errors( # pylint: disable=redefined-out assert largest_satellites_entity_path.exists() assert spark.read.parquet(str(largest_satellites_entity_path)).count() == 1 - og_planets_entity_path = br_path / "Originalplanets" - assert og_planets_entity_path.exists() - assert spark.read.parquet(str(og_planets_entity_path)).count() == 1 - errors_path = Path(br_path.parent, "errors", "business_rules_errors.jsonl") assert errors_path.exists() expected_errors = [ { - "Entity": "planets", + "ReportingEntity": "planets", "Key": "", "FailureType": "record", "Status": "error", @@ -340,7 +330,7 @@ def test_apply_business_rules_with_data_errors( # pylint: disable=redefined-out "RecordIndex": "1" }, { - "Entity": "planets", + "ReportingEntity": "planets", "Key": "", "FailureType": "record", "Status": "error", diff --git a/tests/testdata/flights/flights.dischema.json b/tests/testdata/flights/flights.dischema.json new file mode 100644 index 0000000..a0342cc --- /dev/null +++ b/tests/testdata/flights/flights.dischema.json @@ -0,0 +1,212 @@ +{ + "contract": { + "schemas": { + "passengers": { + "fields": { + "flight_id": "int", + "passenger_id": "int", + "passenger_name": "str" + } + } + }, + "error_details": "flights_data_contract_error_details.json", + "datasets": { + "country": { + "fields": { + "country_id": "int", + "country_name": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "country", + "root_tag": "country" + } + } + }, + "key_field": "country_id", + "mandatory_fields": [ + "country_id", + "country_name" + ] + }, + "airport": { + "fields": { + "country_id": "int", + "airport_id": "int", + "airport_name": "str", + "postcode": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "airport", + "root_tag": "country" + } + } + }, + "key_field": "airport_id", + "mandatory_fields": [ + "airport_id" + ] + }, + "staff": { + "fields": { + "airport_id": "int", + "staff_id": "int", + "staff_name": "str", + "role": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "staff_member", + "root_tag": "country" + } + } + }, + "key_field": "staff_id" + }, + "flights": { + "fields": { + "airport_id": "int", + "flight_id": "int", + "destination": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "flight", + "root_tag": "country" + } + } + }, + "key_field": "flight_id" + }, + "passengers": { + "fields": { + "flight_id": "int", + "passenger_id": "int", + "passenger_name": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "passenger", + "root_tag": "country" + } + } + }, + "key_field": "passenger_id" + } + } + }, + "transformations": { + "parameters": { + "entity": "country" + }, + "filters": [ + { + "entity": "flights", + "name": "flight_missing_id", + "expression": "flight_id IS NOT NULL", + "failure_type": "record", + "failure_message": "Record Rejected - Flight is missing an id", + "reporting_field": "flight_id", + "reporting_entity": "flights", + "category": "Blank", + "error_code": "FlightIDMissing" + }, + { + "entity": "flights", + "name": "invalid_destination", + "expression": "lower(destination) IN ('paris', 'madrid', 'new york', 'amsterdam', 'rome', 'dubai', 'dublin', 'lisbon', 'toronto')", + "failure_type": "record", + "failure_message": "Record Rejected - {{ destination }} is not a valid destination", + "reporting_field": "flight_id", + "reporting_entity": "flights", + "category": "Bad value", + "error_code": "InvalidFlightDestination" + }, + { + "entity": "passengers", + "name": "passenger_name_is_null", + "expression": "passenger_name IS NOT NULL", + "failure_type": "record", + "failure_message": "Record Rejected - Passenger Name is missing", + "reporting_field": "passenger_name", + "reporting_entity": "passengers", + "category": "Blank", + "error_code": "PassengerNameMissing" + }, + { + "entity": "staff", + "name": "staff_id_is_null", + "expression": "staff_id IS NOT NULL", + "failure_type": "record", + "failure_message": "Record Rejected - staff_id is missing", + "reporting_field": "passenger_name", + "reporting_entity": "passengers", + "category": "Blank", + "error_code": "StaffIDMissing" + } + ] + }, + "entity_relationships": { + "country": { + "is_root_entity": true, + "mandatory": true, + "empty_entity_error_code": "NoValidCountries", + "empty_entity_error_message": "File Rejected - There are no valid country records" + }, + "airport": { + "parent_entity": "country", + "join_fields": { + "country_id": "country_id" + }, + "mandatory": true, + "missing_parent_id_error_code": "AirportHasNoCountry", + "missing_parent_id_error_message": "Record rejected - No valid country id found for airport", + "no_valid_records_error_code": "CountryHasNoAirport", + "no_valid_records_error_message": "Group rejected - Unable to find any valid airports", + "empty_entity_error_code": "NoValidAirports", + "empty_entity_error_message": "File Rejected - There are no valid airport records" + }, + "staff": { + "parent_entity": "airport", + "join_fields": { + "airport_id": "airport_id" + }, + "mandatory": true, + "missing_parent_id_error_code": "StaffHasNoAirport", + "missing_parent_id_error_message": "Record rejected - No valid airport id found for staff. Airport ID = {{ airport_id }}, Staff ID = {{ staff_id }}", + "no_valid_records_error_code": "AirportHasNoStaff", + "no_valid_records_error_message": "Group rejected - Airport has no valid staff. Airport ID = {{ airport_id }}", + "empty_entity_error_code": "NoValidStaff", + "empty_entity_error_message": "File Rejected - There are no valid staff records" + }, + "flights": { + "parent_entity": "airport", + "join_fields": { + "airport_id": "airport_id" + }, + "mandatory": false, + "missing_parent_id_error_code": "FlightHasNoAirport", + "missing_parent_id_error_message": "Record Rejected - No valid airport found for flight" + }, + "passengers": { + "parent_entity": "flights", + "join_fields": { + "flight_id": "flight_id" + }, + "mandatory": false, + "missing_parent_id_error_code": "PassengerHasNoFlight", + "missing_parent_id_error_message": "Record rejected - No valid flight found for passenger" + } + } +} \ No newline at end of file diff --git a/tests/testdata/flights/flights_add_reader_checks.dischema.json b/tests/testdata/flights/flights_add_reader_checks.dischema.json new file mode 100644 index 0000000..5f8e63b --- /dev/null +++ b/tests/testdata/flights/flights_add_reader_checks.dischema.json @@ -0,0 +1,208 @@ +{ + "contract": { + "schemas": { + "passengers": { + "fields": { + "flight_id": "int", + "passenger_id": "int", + "passenger_name": "str" + } + } + }, + "error_details": "flights_data_contract_error_details.json", + "datasets": { + "country": { + "fields": { + "country_id": "int", + "country_name": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "country", + "root_tag": "country" + } + } + }, + "key_field": "country_id", + "mandatory_fields": [ + "country_id", + "country_name" + ] + }, + "airport": { + "fields": { + "country_id": "int", + "airport_id": "int", + "airport_name": "str", + "postcode": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "airport", + "root_tag": "country" + } + } + }, + "reader_additional_checks": { + "check_empty": { + "error_code": "AIRPORTEMPTY", + "error_message": "No airport records included in submission" + } + }, + "key_field": "airport_id", + "mandatory_fields": [ + "airport_id" + ] + }, + "staff": { + "fields": { + "airport_id": "int", + "staff_id": "int", + "staff_name": "str", + "role": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "staff_member", + "root_tag": "country" + } + } + }, + "key_field": "staff_id" + }, + "flights": { + "fields": { + "airport_id": "int", + "flight_id": "int", + "destination": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "flight", + "root_tag": "country" + } + } + }, + "key_field": "flight_id" + }, + "passengers": { + "fields": { + "flight_id": "int", + "passenger_id": "int", + "passenger_name": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "passenger", + "root_tag": "country" + } + } + }, + "key_field": "passenger_id" + } + } + }, + "transformations": { + "parameters": { + "entity": "country" + }, + "filters": [ + { + "entity": "flights", + "name": "flight_missing_id", + "expression": "flight_id IS NOT NULL", + "failure_type": "record", + "failure_message": "Record Rejected - Flight is missing an id", + "reporting_field": "flight_id", + "reporting_entity": "flights", + "category": "Blank", + "error_code": "FlightIDMissing" + }, + { + "entity": "flights", + "name": "invalid_destination", + "expression": "lower(destination) IN ('paris', 'madrid', 'new york', 'amsterdam', 'rome', 'dubai', 'dublin', 'lisbon', 'toronto')", + "failure_type": "record", + "failure_message": "Record Rejected - {{ destination }} is not a valid destination", + "reporting_field": "flight_id", + "reporting_entity": "flights", + "category": "Bad value", + "error_code": "InvalidFlightDestination" + }, + { + "entity": "passengers", + "name": "passenger_name_is_null", + "expression": "passenger_name IS NOT NULL", + "failure_type": "record", + "failure_message": "Record Rejected - Passenger Name is missing", + "reporting_field": "passenger_name", + "reporting_entity": "passengers", + "category": "Blank", + "error_code": "PassengerNameMissing" + }, + { + "entity": "staff", + "name": "staff_id_is_null", + "expression": "staff_id IS NOT NULL", + "failure_type": "record", + "failure_message": "Record Rejected - staff_id is missing", + "reporting_field": "passenger_name", + "reporting_entity": "passengers", + "category": "Blank", + "error_code": "StaffIDMissing" + } + ] + }, + "entity_relationships": { + "airport": { + "parent_entity": "country", + "join_fields": { + "country_id": "country_id" + }, + "mandatory": true, + "missing_parent_id_error_code": "AirportHasNoCountry", + "missing_parent_id_error_message": "Record rejected - No valid country id found for airport", + "no_valid_records_error_code": "CountryHasNoAirport", + "no_valid_records_error_message": "Group rejected - Unable to find any valid airports" + }, + "staff": { + "parent_entity": "airport", + "join_fields": { + "airport_id": "airport_id" + }, + "mandatory": true, + "missing_parent_id_error_code": "StaffHasNoAirport", + "missing_parent_id_error_message": "Record rejected - No valid airport id found for staff. Airport ID = {{ airport_id }}, Staff ID = {{ staff_id }}", + "no_valid_records_error_code": "AirportHasNoStaff", + "no_valid_records_error_message": "Group rejected - Airport has no valid staff. Airport ID = {{ airport_id }}" + }, + "flights": { + "parent_entity": "airport", + "join_fields": { + "airport_id": "airport_id" + }, + "mandatory": false, + "missing_parent_id_error_code": "FlightHasNoAirport", + "missing_parent_id_error_message": "Record Rejected - No valid airport found for flight" + }, + "passengers": { + "parent_entity": "flights", + "join_fields": { + "flight_id": "flight_id" + }, + "mandatory": false, + "missing_parent_id_error_code": "PassengerHasNoFlight", + "missing_parent_id_error_message": "Record rejected - No valid flight found for passenger" + } + } +} \ No newline at end of file diff --git a/tests/testdata/flights/flights_data_contract_error_details.json b/tests/testdata/flights/flights_data_contract_error_details.json new file mode 100644 index 0000000..3cf4327 --- /dev/null +++ b/tests/testdata/flights/flights_data_contract_error_details.json @@ -0,0 +1,18 @@ +{ + "country": { + "country_id": { + "Blank": { + "error_code": "CountryIdIsMissing", + "error_message": "Record Rejected - Country is missing an id" + } + } + }, + "airport": { + "airport_id": { + "Blank": { + "error_code": "AirportIdIsMissing", + "error_message": "Record Rejected - Airport is missing an id" + } + } + } +} \ No newline at end of file diff --git a/tests/testdata/flights/flights_full_regression.xml b/tests/testdata/flights/flights_full_regression.xml new file mode 100644 index 0000000..1ed73cf --- /dev/null +++ b/tests/testdata/flights/flights_full_regression.xml @@ -0,0 +1,171 @@ + + + 1 + England + + + 1 + 1 + Heathrow + TW6 1EW + + + 1 + 1 + Paris + + + 1 + 1 + John + + + + + + + 1 + 1 + Marge + Manager + + + + + 1 + 2 + Manchester + M90 1QX + + + 2 + 2 + Venus + + + 2 + 2 + Jane + + + + + 2 + 3 + Rome + + + 3 + 3 + + + + + + + 2 + 2 + Thomas + Pilot + + + + + 1 + Birmingham + B26 3QJ + + + 3 + 4 + Amsterdam + + + 4 + 4 + Billy + + + + + + + 3 + 3 + Joanne + Security + + + + + 1 + 4 + Leeds & Bradford + LS19 7TU + + + 4 + 5 + Dubai + + + 5 + 5 + Terry + + + + + + + 4 + Rebecca + Ground Crew + + + 4 + Tim + Ground Crew + + + 4 + Julie + Pilot + + + + + 1 + 5 + Newcastle + NE13 8BZ + + + 5 + 6 + Toronto + + + 6 + 6 + Bob + + + + + + + 5 + 8 + Jasmine + Pilot + + + 5 + Rodger + Security + + + + + \ No newline at end of file diff --git a/tests/testdata/flights/missing_country_id.xml b/tests/testdata/flights/missing_country_id.xml new file mode 100644 index 0000000..14ccec2 --- /dev/null +++ b/tests/testdata/flights/missing_country_id.xml @@ -0,0 +1,322 @@ + + + England + + + 1 + 1 + Heathrow + TW6 1EW + + + 1 + 1 + Paris + + + 1 + 1 + John + + + 1 + 2 + Jane + + + 1 + 3 + Peter + + + + + 1 + 2 + Madrid + + + 2 + 4 + Homer + + + 2 + 5 + Marge + + + + + 1 + 3 + New York + + + 3 + 6 + Lisa + + + 3 + 7 + Bart + + + 3 + 8 + Maggie + + + + + + + 1 + 1 + Alice + Manager + + + 1 + 2 + Bob + Pilot + + + 1 + 3 + Charlie + Ground Crew + + + 1 + 4 + Diana + Security + + + + + 1 + 2 + Gatwick + RH6 0NP + + + 2 + 4 + Amsterdam + + + 4 + 9 + Oliver + + + 4 + 10 + Emily + + + + + 2 + 5 + Rome + + + 5 + 11 + George + + + 5 + 12 + Charlotte + + + 5 + 13 + Harry + + + + + + 2 + 6 + Dubai + + + 6 + 14 + William + + + 6 + 15 + Amelia + + + + + + + 2 + 5 + Edward + Manager + + + 2 + 6 + Fiona + Air Traffic Controller + + + 2 + 7 + Graham + Ground Crew + + + 2 + 8 + Hannah + Security + + + 2 + 9 + Ian + Engineer + + + + + 1 + 3 + Manchester + M90 1QX + + + 3 + 7 + Dublin + + + 7 + 16 + Jack + + + 7 + 17 + Isla + + + + + 3 + 8 + Lisbon + + + 8 + 18 + Thomas + + + 8 + 19 + Grace + + + 8 + 20 + Jacob + + + + + 3 + 9 + Toronto + + + 9 + 21 + Leo + + + 9 + 22 + Sophie + + + + + 3 + 10 + New York + + + 10 + 23 + Daniel + + + 10 + 24 + Ella + + + 10 + 25 + Oscar + + + + + + + 3 + 10 + Kevin + Manager + + + 3 + 11 + Laura + Pilot + + + 3 + 12 + Michael + Air Traffic Controller + + + 3 + 13 + Natalie + Ground Crew + + + 3 + 14 + Oliver + Security + + + 3 + 15 + Paula + Engineer + + + + + diff --git a/tests/testdata/flights/missing_flight_id.xml b/tests/testdata/flights/missing_flight_id.xml new file mode 100644 index 0000000..fdd7cf7 --- /dev/null +++ b/tests/testdata/flights/missing_flight_id.xml @@ -0,0 +1,322 @@ + + + 1 + England + + + 1 + 1 + Heathrow + TW6 1EW + + + 1 + Paris + + + 1 + 1 + John + + + 1 + 2 + Jane + + + 1 + 3 + Peter + + + + + 1 + 2 + Madrid + + + 2 + 4 + Homer + + + 2 + 5 + Marge + + + + + 1 + 3 + New York + + + 3 + 6 + Lisa + + + 3 + 7 + Bart + + + 3 + 8 + Maggie + + + + + + + 1 + 1 + Alice + Manager + + + 1 + 2 + Bob + Pilot + + + 1 + 3 + Charlie + Ground Crew + + + 1 + 4 + Diana + Security + + + + + 1 + 2 + Gatwick + RH6 0NP + + + 2 + 4 + Amsterdam + + + 4 + 9 + Oliver + + + 4 + 10 + Emily + + + + + 2 + 5 + Rome + + + 5 + 11 + George + + + 5 + 12 + Charlotte + + + 5 + 13 + Harry + + + + + + 2 + 6 + Dubai + + + 6 + 14 + William + + + 6 + 15 + Amelia + + + + + + + 2 + 5 + Edward + Manager + + + 2 + 6 + Fiona + Air Traffic Controller + + + 2 + 7 + Graham + Ground Crew + + + 2 + 8 + Hannah + Security + + + 2 + 9 + Ian + Engineer + + + + + 1 + 3 + Manchester + M90 1QX + + + 3 + 7 + Dublin + + + 7 + 16 + Jack + + + 7 + 17 + Isla + + + + + 3 + 8 + Lisbon + + + 8 + 18 + Thomas + + + 8 + 19 + Grace + + + 8 + 20 + Jacob + + + + + 3 + 9 + Toronto + + + 9 + 21 + Leo + + + 9 + 22 + Sophie + + + + + 3 + 10 + New York + + + 10 + 23 + Daniel + + + 10 + 24 + Ella + + + 10 + 25 + Oscar + + + + + + + 3 + 10 + Kevin + Manager + + + 3 + 11 + Laura + Pilot + + + 3 + 12 + Michael + Air Traffic Controller + + + 3 + 13 + Natalie + Ground Crew + + + 3 + 14 + Oliver + Security + + + 3 + 15 + Paula + Engineer + + + + + diff --git a/tests/testdata/flights/multi_node_file_rejection.xml b/tests/testdata/flights/multi_node_file_rejection.xml new file mode 100644 index 0000000..466a35b --- /dev/null +++ b/tests/testdata/flights/multi_node_file_rejection.xml @@ -0,0 +1,92 @@ + + + 1 + England + + + 1 + 1 + Heathrow + TW6 1EW + + + 1 + 1 + Paris + + + 1 + 1 + John + + + + + + + 1 + 1 + Alice + Manager + + + 1 + Alice + Manager + + + + + 1 + 2 + Manchester + M90 1QX + + + 2 + 2 + Dublin + + + 2 + 2 + Jack + + + + + + + 3 + Kevin + Manager + + + 3 + Laura + Pilot + + + 3 + Michael + Air Traffic Controller + + + 3 + Natalie + Ground Crew + + + 3 + Oliver + Security + + + 3 + Paula + Engineer + + + + + diff --git a/tests/testdata/flights/only_country_id.xml b/tests/testdata/flights/only_country_id.xml new file mode 100644 index 0000000..b0ebd07 --- /dev/null +++ b/tests/testdata/flights/only_country_id.xml @@ -0,0 +1,5 @@ + + + 1 + England + \ No newline at end of file diff --git a/tests/testdata/flights/perfect_flights.xml b/tests/testdata/flights/perfect_flights.xml new file mode 100644 index 0000000..88346dc --- /dev/null +++ b/tests/testdata/flights/perfect_flights.xml @@ -0,0 +1,323 @@ + + + 1 + England + + + 1 + 1 + Heathrow + TW6 1EW + + + 1 + 1 + Paris + + + 1 + 1 + John + + + 1 + 2 + Jane + + + 1 + 3 + Peter + + + + + 1 + 2 + Madrid + + + 2 + 4 + Homer + + + 2 + 5 + Marge + + + + + 1 + 3 + New York + + + 3 + 6 + Lisa + + + 3 + 7 + Bart + + + 3 + 8 + Maggie + + + + + + + 1 + 1 + Alice + Manager + + + 1 + 2 + Bob + Pilot + + + 1 + 3 + Charlie + Ground Crew + + + 1 + 4 + Diana + Security + + + + + 1 + 2 + Gatwick + RH6 0NP + + + 2 + 4 + Amsterdam + + + 4 + 9 + Oliver + + + 4 + 10 + Emily + + + + + 2 + 5 + Rome + + + 5 + 11 + George + + + 5 + 12 + Charlotte + + + 5 + 13 + Harry + + + + + + 2 + 6 + Dubai + + + 6 + 14 + William + + + 6 + 15 + Amelia + + + + + + + 2 + 5 + Edward + Manager + + + 2 + 6 + Fiona + Air Traffic Controller + + + 2 + 7 + Graham + Ground Crew + + + 2 + 8 + Hannah + Security + + + 2 + 9 + Ian + Engineer + + + + + 1 + 3 + Manchester + M90 1QX + + + 3 + 7 + Dublin + + + 7 + 16 + Jack + + + 7 + 17 + Isla + + + + + 3 + 8 + Lisbon + + + 8 + 18 + Thomas + + + 8 + 19 + Grace + + + 8 + 20 + Jacob + + + + + 3 + 9 + Toronto + + + 9 + 21 + Leo + + + 9 + 22 + Sophie + + + + + 3 + 10 + New York + + + 10 + 23 + Daniel + + + 10 + 24 + Ella + + + 10 + 25 + Oscar + + + + + + + 3 + 10 + Kevin + Manager + + + 3 + 11 + Laura + Pilot + + + 3 + 12 + Michael + Air Traffic Controller + + + 3 + 13 + Natalie + Ground Crew + + + 3 + 14 + Oliver + Security + + + 3 + 15 + Paula + Engineer + + + + + diff --git a/tests/testdata/flights/singular_node_rejections.xml b/tests/testdata/flights/singular_node_rejections.xml new file mode 100644 index 0000000..134ee89 --- /dev/null +++ b/tests/testdata/flights/singular_node_rejections.xml @@ -0,0 +1,49 @@ + + + 1 + England + + + 1 + 1 + Heathrow + TW6 1EW + + + 1 + 1 + Mars + + + 1 + 1 + John + + + 1 + 2 + Jane + + + + + 1 + 2 + Venus + + + 2 + 3 + Homer + + + 2 + 4 + Marge + + + + + + + \ No newline at end of file diff --git a/tests/testdata/movies/movies_contract_error_details.json b/tests/testdata/movies/movies_contract_error_details.json index 260ee96..e457555 100644 --- a/tests/testdata/movies/movies_contract_error_details.json +++ b/tests/testdata/movies/movies_contract_error_details.json @@ -1,27 +1,29 @@ { - "title": { - "Blank": { - "error_code": "BLANKTITLE", - "error_message": "title should not be blank", - "error_level": "submission" - } - }, - "year": { - "Blank": { - "error_code": "BLANKYEAR", - "error_message": "year not provided", - "is_informational": true + "movies": { + "title": { + "Blank": { + "error_code": "BLANKTITLE", + "error_message": "title should not be blank", + "error_level": "submission" + } }, - "Bad value": { - "error_code": "DODGYYEAR", - "error_message": "year value ({{year}}) is invalid", - "reporting_entity": "movies_rename_test" - } - }, - "cast.date_joined": { - "Bad value": { - "error_code": "DODGYDATE", - "error_message": "date_joined value is not valid: {{__error_value}}" + "year": { + "Blank": { + "error_code": "BLANKYEAR", + "error_message": "year not provided", + "is_informational": true + }, + "Bad value": { + "error_code": "DODGYYEAR", + "error_message": "year value ({{year}}) is invalid", + "reporting_entity": "movies_rename_test" + } + }, + "cast.date_joined": { + "Bad value": { + "error_code": "DODGYDATE", + "error_message": "date_joined value is not valid: {{__error_value}}" + } } } } \ No newline at end of file diff --git a/zensical.toml b/zensical.toml index 064cb26..0141b6d 100644 --- a/zensical.toml +++ b/zensical.toml @@ -25,6 +25,7 @@ nav = [ {"File Transformation" = "user_guidance/file_transformation.md"}, {"Data Contract" = "user_guidance/data_contract.md"}, {"Business Rules" = "user_guidance/business_rules.md"}, + {"Entity Relationships" = "user_guidance/entity_relationships.md"} ]}, {"Backend Implementations" = [ {"DuckDB" = "user_guidance/implementations/duckdb.md"}, @@ -56,6 +57,9 @@ nav = [ {"Refdata" = [ {"Refdata Types" = "advanced_guidance/package_documentation/refence_data_types.md"}, {"Refdata Loaders" = "advanced_guidance/package_documentation/refdata_loaders.md"}, + ]}, + {"Entity Relationships" = [ + {"Entity Hierarchy" = "advanced_guidance/package_documentation/entity_hierarchy.md"} ]} ]}, {"Feedback" = [ @@ -193,6 +197,9 @@ options.custom_icons = ["overrides/.icons"] auto_append = ["includes/jargon_and_acronyms.md"] [project.markdown_extensions.pymdownx.superfences] +custom_fences = [ + { name = "mermaid", class = "mermaid", format = "pymdownx.superfences.fence_code_format" }, +] [project.markdown_extensions.pymdownx.tabbed] alternate_style = true