diff --git a/.mise.toml b/.mise.toml index be3149da..0213a692 100644 --- a/.mise.toml +++ b/.mise.toml @@ -1,4 +1,4 @@ [tools] python="3.12" poetry="2.4.1" -java="liberica-1.8.0" +java="zulu-17.60.17" diff --git a/.tool-versions b/.tool-versions index e111eee1..eb68911b 100644 --- a/.tool-versions +++ b/.tool-versions @@ -1,3 +1,3 @@ python 3.12.12 poetry 2.4.1 -java liberica-1.8.0 +java zulu-17.60.17 diff --git a/README.md b/README.md index e94b29e1..f91ce2e9 100644 --- a/README.md +++ b/README.md @@ -17,6 +17,7 @@ __The DVE offers__: - Format normalization to Parquet for a unified data representation - Data modelling and typecasting - Business-rule validations executed on supported backends such as Spark and DuckDB, with the option to add custom backends +- Validate and enforce referential integrity checks with minimal configuration - Deriving new fields and entities - Clear validation reporting, including summary insights and record-level error messages @@ -41,7 +42,8 @@ Below is a list of features that we would like to implement or have been request | Uplift to Python 3.11 | 0.2.0 | Yes | | Uplift Pyspark to 3.5 | 0.8.0 | Yes | | Allow DVE to run on Python 3.12+ | 0.8.0 | Yes | -| Upgrade to Pydantic 2.0 | 0.9.0 | Yes | +| Upgrade to Pydantic 2.0 | 0.9.0 | Yes | +| Upgrade DuckDB to v1.4 | 0.10.0 | Yes | | Uplift Pyspark to 4.0+ | TBA | No | | Polars upgrade to v1+ | TBA | No | | DuckDB upgrade to v1.5+ | TBA | No | diff --git a/docs/advanced_guidance/json_schemas/contract/components/base_entity.schema.json b/docs/advanced_guidance/json_schemas/contract/components/base_entity.schema.json index b89291ad..06607cbb 100644 --- a/docs/advanced_guidance/json_schemas/contract/components/base_entity.schema.json +++ b/docs/advanced_guidance/json_schemas/contract/components/base_entity.schema.json @@ -26,8 +26,15 @@ "items": { "type": "string" } + }, + "reader_additional_checks": { + "description": "A mapping of additional checks to perform on entities after initial read", + "type": "object", + "additionalProperties": { + "$ref": "reader_additional_checks.json" } - }, + } +}, "required": ["fields"] } diff --git a/docs/advanced_guidance/json_schemas/contract/components/reader_additional_checks.schema.json b/docs/advanced_guidance/json_schemas/contract/components/reader_additional_checks.schema.json new file mode 100644 index 00000000..6c01eb70 --- /dev/null +++ b/docs/advanced_guidance/json_schemas/contract/components/reader_additional_checks.schema.json @@ -0,0 +1,22 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "data-ingest:contract/components/reader_additional_checks.schema.json", + "title": "reader_additional_checks", + "description": "Additional checks to perform on initially read entities", + "type": "object", + "properties": { + "error_code": { + "description": "The code to be used for the additional check specified", + "type": "string" + }, + "error_message": { + "description": "The message to be displayed for the additional check specified.", + "type": "string" + } + }, + "required": [ + "error_code", + "error_message" + ], + "additionalProperties": false +} \ No newline at end of file diff --git a/docs/advanced_guidance/json_schemas/contract/components/reader_constraints/global_csv_reader_args.schema.json b/docs/advanced_guidance/json_schemas/contract/components/reader_constraints/global_csv_reader_args.schema.json index 56e2acc6..ea049216 100644 --- a/docs/advanced_guidance/json_schemas/contract/components/reader_constraints/global_csv_reader_args.schema.json +++ b/docs/advanced_guidance/json_schemas/contract/components/reader_constraints/global_csv_reader_args.schema.json @@ -4,6 +4,11 @@ "title": "Keyword Arguments used across all CSV readers", "description": "Arguments present in all CSV readers available.", "type": "object", + "anyOf": [ + { + "$ref": "global_reader_args.schema.json" + } + ], "properties": { "field_check": { "type": "string", @@ -13,14 +18,6 @@ "false" ] }, - "field_check_error_code": { - "type": "string", - "description": "Allows you to customise the error code generated in the error report when the field check fails." - }, - "field_check_error_message": { - "type": "string", - "description": "Allows you to customise the error message generated in the error report when the field check fails." - }, "null_empty_strings": { "type": "boolean", "description": "Converts empty string values into 'Null' values." diff --git a/docs/advanced_guidance/json_schemas/contract/components/reader_constraints/global_reader_args.schema.json b/docs/advanced_guidance/json_schemas/contract/components/reader_constraints/global_reader_args.schema.json new file mode 100644 index 00000000..3da16c23 --- /dev/null +++ b/docs/advanced_guidance/json_schemas/contract/components/reader_constraints/global_reader_args.schema.json @@ -0,0 +1,19 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "data-ingest:contract/components/reader_constraints/global_reader_args.schema.json", + "title": "Keyword Arguments used across all readers", + "description": "Arguments present in all readers available.", + "type": "object", + "properties": { + "ft_error_code": { + "type": "string", + "description": "Error code to raise when a reader is unable to read the contents of a file.", + "minLength": 1 + }, + "ft_error_message": { + "type": "string", + "description": "Error message to raise when a reader is unable to read the contents of a file.", + "minLength": 1 + } + } +} \ No newline at end of file diff --git a/docs/advanced_guidance/json_schemas/contract/components/reader_constraints/global_xml_reader_args.schema.json b/docs/advanced_guidance/json_schemas/contract/components/reader_constraints/global_xml_reader_args.schema.json index d82ce720..40f893c9 100644 --- a/docs/advanced_guidance/json_schemas/contract/components/reader_constraints/global_xml_reader_args.schema.json +++ b/docs/advanced_guidance/json_schemas/contract/components/reader_constraints/global_xml_reader_args.schema.json @@ -4,6 +4,11 @@ "title": "Keyword Arguments used across all XML readers", "description": "Arguments present in all XML readers.", "type": "object", + "anyOf": [ + { + "$ref": "global_reader_args.schema.json" + } + ], "properties": { "record_tag": { "type": "string", @@ -40,14 +45,6 @@ "type": "string", "description": "The Relative path (to the dischema) for the XSD document." }, - "xsd_error_code": { - "type": "string", - "description": "Allows you to customise the error code generated in the error report when the XSD check fails." - }, - "xsd_error_message": { - "type": "string", - "description": "Allows you to customise the error message generated in the error report when the XSD check fails." - }, "rules_location": { "type": "string", "description": "Allows you to prefix the directory which contains the XSD document." diff --git a/docs/advanced_guidance/json_schemas/dataset.schema.json b/docs/advanced_guidance/json_schemas/dataset.schema.json index 4e85011d..af8b620f 100644 --- a/docs/advanced_guidance/json_schemas/dataset.schema.json +++ b/docs/advanced_guidance/json_schemas/dataset.schema.json @@ -10,6 +10,9 @@ }, "transformations": { "$ref": "transformations/transformations.schema.json" + }, + "entity_relationships": { + "$ref": "entity_relationships.schema.json" } }, "required": [ diff --git a/docs/advanced_guidance/json_schemas/entity_relationships.schema.json b/docs/advanced_guidance/json_schemas/entity_relationships.schema.json new file mode 100644 index 00000000..0c2abbb2 --- /dev/null +++ b/docs/advanced_guidance/json_schemas/entity_relationships.schema.json @@ -0,0 +1,50 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "data-ingest:entity_relationships.schema.json", + "title": "entity_relationships", + "description": "Description of relationships to link normalised entities back to parent entities.", + "type": "object", + "patternProperties": { + "^[A-Za-z0-9_]+.$": { + "type": "object", + "properties": { + "parent_entity": { + "type": "string" + }, + "join_fields": { + "type": "object", + "additionalProperties": { + "type": "string" + } + }, + "is_root_entity": { + "type": "boolean" + }, + "mandatory": { + "type": "boolean" + }, + "missing_parent_id_error_code": { + "type": "string" + }, + "missing_parent_id_error_message": { + "type": "string" + }, + "no_valid_records_error_code": { + "type": "string" + }, + "no_valid_records_error_message": { + "type": "string" + }, + "empty_entity_error_code": { + "type": "string", + "minLength": 1 + }, + "empty_entity_error_message": { + "type": "string", + "minLength": 1 + } + }, + "additionalProperties": false + } + } +} \ No newline at end of file diff --git a/docs/advanced_guidance/package_documentation/entity_hierarchy.md b/docs/advanced_guidance/package_documentation/entity_hierarchy.md new file mode 100644 index 00000000..4080dcbe --- /dev/null +++ b/docs/advanced_guidance/package_documentation/entity_hierarchy.md @@ -0,0 +1,5 @@ +::: dve.core_engine.configuration.v1.hierarchy + handler: python + options: + show_root_heading: true + heading_level: 2 diff --git a/docs/user_guidance/entity_relationships.md b/docs/user_guidance/entity_relationships.md new file mode 100644 index 00000000..bf0f6c19 --- /dev/null +++ b/docs/user_guidance/entity_relationships.md @@ -0,0 +1,63 @@ +--- +title: Entity Relationships +tags: + - Linkage + - Relationships + - Missing + - Parent + - Group + - Rejections +--- + +Sometimes a user may choose to use the file transformation stage to `normalise` a heavily nested dataset into separate entities during the initial reading of data. This would be done by specifying different entities in the dataset section of the contract configuration in the `dischema` file. This allows for easier interaction when customising errors in the data contract or writing transformations in the business rules. + +`Normalising` assets can lead to more complex validations being required. For example in the flights dataset: + +```mermaid +erDiagram + COUNTRY ||--|{ AIRPORT : "" + AIRPORT ||--o{ FLIGHT : "" + FLIGHT ||--o{ PASSENGER : "" + AIRPORT ||--|{ STAFF_MEMBER : "" +``` + +### Missing Parent Records + +It could be that an airport record is deemed invalid and removed. Due to this, any flight records that linked to the now removed airport record are themselves invalid - a situation we refer to as a `missing_parent` issue, but are now existing in an entirely different entity. + +### No Valid Mandatory Records + +It could also be the case that staff records are a mandatory field for airport records. If all staff records for a particular airport record are removed during validation, this itself would invalidate the airport record - a situation we refer to as `no_valid_records` issue - but again the invalid airport record is in a different entity. + +### Dischema + +In order to perform these validations, how to link normalised entities needs to be provided. This can be specified in the `entity_relationships` section of the `dischema`. + +## Entity Relationships Content + +To allow the DVE to link between normalised assets, the following information should be provided (per linkable entity): + +- `parent_entity`: the immediate parent of the entity +- `join_fields`: how to join the entity with its parent in dictionary form (parent_field_name: child_field_name) +- `is_root_entity`: indicates that the entity is a root node in a hierarchical model +- `mandatory`: whether the child entity is a mandatory field in the immediate parent + +There is also the functionality to customise errors related to either missing parent or group rejections: + +- `missing_parent_id_error_code`: the error code to display if a record is rejected as it has no valid parent record +- `missing_parent_id_error_message`: the error message to display if a record is rejected as it has no valid parent record +- `no_valid_records_error_code`: the error code to display if parent records are removed due to no valid children in a mandatory field +- `no_valid_records_error_message`: the error message to display if parent records are removed due to no valid children in a mandatory field +- `empty_entity_error_code`: the error code to display if a __mandatory__ entity has any records post filtering +- `empty_entity_error_message`: the error message to display if a __mandatory__ entity has any records post filtering + +!!! note + Specifying root entities is __optional__. Root entities will be inferred based on their absence. + You may wish to specify root entities so that error codes and messages can be customised (e.g. empty entity). + + When specifying root entities you should ensure that parent_entity and join_fields values are left blank. + +## Entity Hierarchy Object + +The details provided in the entity_relationships section of the dischema are used to create an EntityHierarchy object. +Please refer to [Advanced User Guidance: Entity Hierarchy](../advanced_guidance/package_documentation/entity_hierarchy.md). diff --git a/docs/user_guidance/getting_started.md b/docs/user_guidance/getting_started.md index e938c2df..fbda5ae6 100644 --- a/docs/user_guidance/getting_started.md +++ b/docs/user_guidance/getting_started.md @@ -68,6 +68,8 @@ Within the example above, there are two parent keys - `schemas` and `datasets`. !!! note The "splitting" of entities is considerably more useful in situtations where you want to normalise/de-normalise your data. If you're unfamiliar with this concept, you can read more about it [here](https://en.wikipedia.org/wiki/Database_normalization). However, you should keep in mind potential performance impacts of doing this. If you have rules that requires fields from different entities, you will have to perform a `join` between the split entities to be able to perform the rule. +To support with the application of more complex validation relating to parent and child records within normalised data, the [entity_relationships](entity_relationships.md) section of the `dischema` enables users to specify parent-child relationships and to customise error codes related to missing parent and group level validation issues. + For each dataset definition, you will need to provide a `reader_config` which describes how to load the data during the [File Transformation](file_transformation.md) stage. So, in the example above, we expect `movies` to come in as a `JSON` file. However, you can add more readers if you have the same data in different data formats (e.g. `csv`, `xml`, `json`). Regardless of what file format, the [File Transformation](file_transformation.md) stage will convert the submitted data into a "stringified" parquet format which is a requirement for the subsequent stages. To learn more about how you can construct your Data Contract please read [here](data_contract.md). diff --git a/docs/user_guidance/install.md b/docs/user_guidance/install.md index 85186cd7..2c86b122 100644 --- a/docs/user_guidance/install.md +++ b/docs/user_guidance/install.md @@ -78,11 +78,12 @@ Once you have installed the DVE you are almost ready to use it. To be able to ru ## DVE Version Compatability Matrix -| DVE Version | Python Version | DuckDB Version | Spark Version | Pydantic Version | -| ------------ | -------------- | -------------- | --------------- | ---------------- | -| >=0.9.0 | >=3.10,<3.13 | 1.1.3 | >=3.5.0,<=3.5.5 | 2.13.4 | -| >=0.8.0 | >=3.10,<3.13 | 1.1.3 | 3.5.2 | 1.10.19 | -| >=0.7.2 | >=3.10,<3.12 | 1.1.* | 3.4.* | 1.10.16 | -| >=0.6 | >=3.10,<3.12 | 1.1.* | 3.4.* | 1.10.15 | -| >=0.2,<0.6 | >=3.10,<3.12 | 1.1.0 | 3.4.4 | 1.10.15 | -| 0.1 | >=3.7.2,<3.8 | 1.1.0 | 3.2.1 | 1.10.15 | +| DVE Version | Python Version | DuckDB Version | Spark Version | Pydantic Version | +| ------------ | -------------- | ---------------- | --------------- | ---------------- | +| >=0.10.0 | >=3.10,<1.13 | __>=1.4,<1.4.5__ | >=3.5.0,<=3.5.5 | 2.13.4 | +| >=0.9.0 | >=3.10,<3.13 | 1.1.3 | >=3.5.0,<=3.5.5 | __2.13.4__ | +| >=0.8.0 | >=3.10,<3.13 | __1.1.3__ | __3.5.2__ | 1.10.19 | +| >=0.7.2 | >=3.10,<3.12 | 1.1.* | 3.4.* | __1.10.16__ | +| >=0.6 | >=3.10,<3.12 | __1.1.*__ | __3.4.*__ | 1.10.15 | +| >=0.2,<0.6 | __>=3.10,<3.12__ | 1.1.0 | 3.4.4 | 1.10.15 | +| 0.1 | >=3.7.2,<3.8 | 1.1.0 | 3.2.1 | 1.10.15 | diff --git a/poetry.lock b/poetry.lock index 0fedfd49..18f789ec 100644 --- a/poetry.lock +++ b/poetry.lock @@ -1144,66 +1144,58 @@ files = [ [[package]] name = "duckdb" -version = "1.1.3" +version = "1.4.4" description = "DuckDB in-process database" optional = false -python-versions = ">=3.7.0" +python-versions = ">=3.9.0" groups = ["main"] files = [ - {file = "duckdb-1.1.3-cp310-cp310-macosx_12_0_arm64.whl", hash = "sha256:1c0226dc43e2ee4cc3a5a4672fddb2d76fd2cf2694443f395c02dd1bea0b7fce"}, - {file = "duckdb-1.1.3-cp310-cp310-macosx_12_0_universal2.whl", hash = "sha256:7c71169fa804c0b65e49afe423ddc2dc83e198640e3b041028da8110f7cd16f7"}, - {file = "duckdb-1.1.3-cp310-cp310-macosx_12_0_x86_64.whl", hash = "sha256:872d38b65b66e3219d2400c732585c5b4d11b13d7a36cd97908d7981526e9898"}, - {file = "duckdb-1.1.3-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:25fb02629418c0d4d94a2bc1776edaa33f6f6ccaa00bd84eb96ecb97ae4b50e9"}, - {file = "duckdb-1.1.3-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9e3f5cd604e7c39527e6060f430769b72234345baaa0987f9500988b2814f5e4"}, - {file = "duckdb-1.1.3-cp310-cp310-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:08935700e49c187fe0e9b2b86b5aad8a2ccd661069053e38bfaed3b9ff795efd"}, - {file = "duckdb-1.1.3-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:f9b47036945e1db32d70e414a10b1593aec641bd4c5e2056873d971cc21e978b"}, - {file = "duckdb-1.1.3-cp310-cp310-win_amd64.whl", hash = "sha256:35c420f58abc79a68a286a20fd6265636175fadeca1ce964fc8ef159f3acc289"}, - {file = "duckdb-1.1.3-cp311-cp311-macosx_12_0_arm64.whl", hash = "sha256:4f0e2e5a6f5a53b79aee20856c027046fba1d73ada6178ed8467f53c3877d5e0"}, - {file = "duckdb-1.1.3-cp311-cp311-macosx_12_0_universal2.whl", hash = "sha256:911d58c22645bfca4a5a049ff53a0afd1537bc18fedb13bc440b2e5af3c46148"}, - {file = "duckdb-1.1.3-cp311-cp311-macosx_12_0_x86_64.whl", hash = "sha256:c443d3d502335e69fc1e35295fcfd1108f72cb984af54c536adfd7875e79cee5"}, - {file = "duckdb-1.1.3-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:0a55169d2d2e2e88077d91d4875104b58de45eff6a17a59c7dc41562c73df4be"}, - {file = "duckdb-1.1.3-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9d0767ada9f06faa5afcf63eb7ba1befaccfbcfdac5ff86f0168c673dd1f47aa"}, - {file = "duckdb-1.1.3-cp311-cp311-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:51c6d79e05b4a0933672b1cacd6338f882158f45ef9903aef350c4427d9fc898"}, - {file = "duckdb-1.1.3-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:183ac743f21c6a4d6adfd02b69013d5fd78e5e2cd2b4db023bc8a95457d4bc5d"}, - {file = "duckdb-1.1.3-cp311-cp311-win_amd64.whl", hash = "sha256:a30dd599b8090ea6eafdfb5a9f1b872d78bac318b6914ada2d35c7974d643640"}, - {file = "duckdb-1.1.3-cp312-cp312-macosx_12_0_arm64.whl", hash = "sha256:a433ae9e72c5f397c44abdaa3c781d94f94f4065bcbf99ecd39433058c64cb38"}, - {file = "duckdb-1.1.3-cp312-cp312-macosx_12_0_universal2.whl", hash = "sha256:d08308e0a46c748d9c30f1d67ee1143e9c5ea3fbcccc27a47e115b19e7e78aa9"}, - {file = "duckdb-1.1.3-cp312-cp312-macosx_12_0_x86_64.whl", hash = "sha256:5d57776539211e79b11e94f2f6d63de77885f23f14982e0fac066f2885fcf3ff"}, - {file = "duckdb-1.1.3-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:e59087dbbb63705f2483544e01cccf07d5b35afa58be8931b224f3221361d537"}, - {file = "duckdb-1.1.3-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:4ebf5f60ddbd65c13e77cddb85fe4af671d31b851f125a4d002a313696af43f1"}, - {file = "duckdb-1.1.3-cp312-cp312-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:e4ef7ba97a65bd39d66f2a7080e6fb60e7c3e41d4c1e19245f90f53b98e3ac32"}, - {file = "duckdb-1.1.3-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:f58db1b65593ff796c8ea6e63e2e144c944dd3d51c8d8e40dffa7f41693d35d3"}, - {file = "duckdb-1.1.3-cp312-cp312-win_amd64.whl", hash = "sha256:e86006958e84c5c02f08f9b96f4bc26990514eab329b1b4f71049b3727ce5989"}, - {file = "duckdb-1.1.3-cp313-cp313-macosx_12_0_arm64.whl", hash = "sha256:0897f83c09356206ce462f62157ce064961a5348e31ccb2a557a7531d814e70e"}, - {file = "duckdb-1.1.3-cp313-cp313-macosx_12_0_universal2.whl", hash = "sha256:cddc6c1a3b91dcc5f32493231b3ba98f51e6d3a44fe02839556db2b928087378"}, - {file = "duckdb-1.1.3-cp313-cp313-macosx_12_0_x86_64.whl", hash = "sha256:1d9ab6143e73bcf17d62566e368c23f28aa544feddfd2d8eb50ef21034286f24"}, - {file = "duckdb-1.1.3-cp313-cp313-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:2f073d15d11a328f2e6d5964a704517e818e930800b7f3fa83adea47f23720d3"}, - {file = "duckdb-1.1.3-cp313-cp313-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:d5724fd8a49e24d730be34846b814b98ba7c304ca904fbdc98b47fa95c0b0cee"}, - {file = "duckdb-1.1.3-cp313-cp313-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:51e7dbd968b393343b226ab3f3a7b5a68dee6d3fe59be9d802383bf916775cb8"}, - {file = "duckdb-1.1.3-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:00cca22df96aa3473fe4584f84888e2cf1c516e8c2dd837210daec44eadba586"}, - {file = "duckdb-1.1.3-cp313-cp313-win_amd64.whl", hash = "sha256:77f26884c7b807c7edd07f95cf0b00e6d47f0de4a534ac1706a58f8bc70d0d31"}, - {file = "duckdb-1.1.3-cp37-cp37m-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:a4748635875fc3c19a7320a6ae7410f9295557450c0ebab6d6712de12640929a"}, - {file = "duckdb-1.1.3-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:b74e121ab65dbec5290f33ca92301e3a4e81797966c8d9feef6efdf05fc6dafd"}, - {file = "duckdb-1.1.3-cp37-cp37m-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:9c619e4849837c8c83666f2cd5c6c031300cd2601e9564b47aa5de458ff6e69d"}, - {file = "duckdb-1.1.3-cp37-cp37m-win_amd64.whl", hash = "sha256:0ba6baa0af33ded836b388b09433a69b8bec00263247f6bf0a05c65c897108d3"}, - {file = "duckdb-1.1.3-cp38-cp38-macosx_12_0_arm64.whl", hash = "sha256:ecb1dc9062c1cc4d2d88a5e5cd8cc72af7818ab5a3c0f796ef0ffd60cfd3efb4"}, - {file = "duckdb-1.1.3-cp38-cp38-macosx_12_0_universal2.whl", hash = "sha256:5ace6e4b1873afdd38bd6cc8fcf90310fb2d454f29c39a61d0c0cf1a24ad6c8d"}, - {file = "duckdb-1.1.3-cp38-cp38-macosx_12_0_x86_64.whl", hash = "sha256:a1fa0c502f257fa9caca60b8b1478ec0f3295f34bb2efdc10776fc731b8a6c5f"}, - {file = "duckdb-1.1.3-cp38-cp38-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:6411e21a2128d478efbd023f2bdff12464d146f92bc3e9c49247240448ace5a6"}, - {file = "duckdb-1.1.3-cp38-cp38-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:c5336939d83837af52731e02b6a78a446794078590aa71fd400eb17f083dda3e"}, - {file = "duckdb-1.1.3-cp38-cp38-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f549af9f7416573ee48db1cf8c9d27aeed245cb015f4b4f975289418c6cf7320"}, - {file = "duckdb-1.1.3-cp38-cp38-win_amd64.whl", hash = "sha256:2141c6b28162199999075d6031b5d63efeb97c1e68fb3d797279d31c65676269"}, - {file = "duckdb-1.1.3-cp39-cp39-macosx_12_0_arm64.whl", hash = "sha256:09c68522c30fc38fc972b8a75e9201616b96ae6da3444585f14cf0d116008c95"}, - {file = "duckdb-1.1.3-cp39-cp39-macosx_12_0_universal2.whl", hash = "sha256:8ee97ec337794c162c0638dda3b4a30a483d0587deda22d45e1909036ff0b739"}, - {file = "duckdb-1.1.3-cp39-cp39-macosx_12_0_x86_64.whl", hash = "sha256:a1f83c7217c188b7ab42e6a0963f42070d9aed114f6200e3c923c8899c090f16"}, - {file = "duckdb-1.1.3-cp39-cp39-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:1aa3abec8e8995a03ff1a904b0e66282d19919f562dd0a1de02f23169eeec461"}, - {file = "duckdb-1.1.3-cp39-cp39-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:80158f4c7c7ada46245837d5b6869a336bbaa28436fbb0537663fa324a2750cd"}, - {file = "duckdb-1.1.3-cp39-cp39-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:647f17bd126170d96a38a9a6f25fca47ebb0261e5e44881e3782989033c94686"}, - {file = "duckdb-1.1.3-cp39-cp39-musllinux_1_2_x86_64.whl", hash = "sha256:252d9b17d354beb9057098d4e5d5698e091a4f4a0d38157daeea5fc0ec161670"}, - {file = "duckdb-1.1.3-cp39-cp39-win_amd64.whl", hash = "sha256:eeacb598120040e9591f5a4edecad7080853aa8ac27e62d280f151f8c862afa3"}, - {file = "duckdb-1.1.3.tar.gz", hash = "sha256:68c3a46ab08836fe041d15dcbf838f74a990d551db47cb24ab1c4576fc19351c"}, + {file = "duckdb-1.4.4-cp310-cp310-macosx_10_9_universal2.whl", hash = "sha256:e870a441cb1c41d556205deb665749f26347ed13b3a247b53714f5d589596977"}, + {file = "duckdb-1.4.4-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:49123b579e4a6323e65139210cd72dddc593a72d840211556b60f9703bda8526"}, + {file = "duckdb-1.4.4-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:5e1933fac5293fea5926b0ee75a55b8cfe7f516d867310a5b251831ab61fe62b"}, + {file = "duckdb-1.4.4-cp310-cp310-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:707530f6637e91dc4b8125260595299ec9dd157c09f5d16c4186c5988bfbd09a"}, + {file = "duckdb-1.4.4-cp310-cp310-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:453b115f4777467f35103d8081770ac2f223fb5799178db5b06186e3ab51d1f2"}, + {file = "duckdb-1.4.4-cp310-cp310-win_amd64.whl", hash = "sha256:a3c8542db7ffb128aceb7f3b35502ebaddcd4f73f1227569306cc34bad06680c"}, + {file = "duckdb-1.4.4-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:5ba684f498d4e924c7e8f30dd157da8da34c8479746c5011b6c0e037e9c60ad2"}, + {file = "duckdb-1.4.4-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:5536eb952a8aa6ae56469362e344d4e6403cc945a80bc8c5c2ebdd85d85eb64b"}, + {file = "duckdb-1.4.4-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:47dd4162da6a2be59a0aef640eb08d6360df1cf83c317dcc127836daaf3b7f7c"}, + {file = "duckdb-1.4.4-cp311-cp311-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:6cb357cfa3403910e79e2eb46c8e445bb1ee2fd62e9e9588c6b999df4256abc1"}, + {file = "duckdb-1.4.4-cp311-cp311-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:4c25d5b0febda02b7944e94fdae95aecf952797afc8cb920f677b46a7c251955"}, + {file = "duckdb-1.4.4-cp311-cp311-win_amd64.whl", hash = "sha256:6703dd1bb650025b3771552333d305d62ddd7ff182de121483d4e042ea6e2e00"}, + {file = "duckdb-1.4.4-cp311-cp311-win_arm64.whl", hash = "sha256:bf138201f56e5d6fc276a25138341b3523e2f84733613fc43f02c54465619a95"}, + {file = "duckdb-1.4.4-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:ddcfd9c6ff234da603a1edd5fd8ae6107f4d042f74951b65f91bc5e2643856b3"}, + {file = "duckdb-1.4.4-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:6792ca647216bd5c4ff16396e4591cfa9b4a72e5ad7cdd312cec6d67e8431a7c"}, + {file = "duckdb-1.4.4-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:1f8d55843cc940e36261689054f7dfb6ce35b1f5b0953b0d355b6adb654b0d52"}, + {file = "duckdb-1.4.4-cp312-cp312-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:c65d15c440c31e06baaebfd2c06d71ce877e132779d309f1edf0a85d23c07e92"}, + {file = "duckdb-1.4.4-cp312-cp312-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:b297eff642503fd435a9de5a9cb7db4eccb6f61d61a55b30d2636023f149855f"}, + {file = "duckdb-1.4.4-cp312-cp312-win_amd64.whl", hash = "sha256:d525de5f282b03aa8be6db86b1abffdceae5f1055113a03d5b50cd2fb8cf2ef8"}, + {file = "duckdb-1.4.4-cp312-cp312-win_arm64.whl", hash = "sha256:50f2eb173c573811b44aba51176da7a4e5c487113982be6a6a1c37337ec5fa57"}, + {file = "duckdb-1.4.4-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:337f8b24e89bc2e12dadcfe87b4eb1c00fd920f68ab07bc9b70960d6523b8bc3"}, + {file = "duckdb-1.4.4-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:0509b39ea7af8cff0198a99d206dca753c62844adab54e545984c2e2c1381616"}, + {file = "duckdb-1.4.4-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:fb94de6d023de9d79b7edc1ae07ee1d0b4f5fa8a9dcec799650b5befdf7aafec"}, + {file = "duckdb-1.4.4-cp313-cp313-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:0d636ceda422e7babd5e2f7275f6a0d1a3405e6a01873f00d38b72118d30c10b"}, + {file = "duckdb-1.4.4-cp313-cp313-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:7df7351328ffb812a4a289732f500d621e7de9942a3a2c9b6d4afcf4c0e72526"}, + {file = "duckdb-1.4.4-cp313-cp313-win_amd64.whl", hash = "sha256:6fb1225a9ea5877421481d59a6c556a9532c32c16c7ae6ca8d127e2b878c9389"}, + {file = "duckdb-1.4.4-cp313-cp313-win_arm64.whl", hash = "sha256:f28a18cc790217e5b347bb91b2cab27aafc557c58d3d8382e04b4fe55d0c3f66"}, + {file = "duckdb-1.4.4-cp314-cp314-macosx_10_15_universal2.whl", hash = "sha256:25874f8b1355e96178079e37312c3ba6d61a2354f51319dae860cf21335c3a20"}, + {file = "duckdb-1.4.4-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:452c5b5d6c349dc5d1154eb2062ee547296fcbd0c20e9df1ed00b5e1809089da"}, + {file = "duckdb-1.4.4-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:8e5c2d8a0452df55e092959c0bfc8ab8897ac3ea0f754cb3b0ab3e165cd79aff"}, + {file = "duckdb-1.4.4-cp314-cp314-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:1af6e76fe8bd24875dc56dd8e38300d64dc708cd2e772f67b9fbc635cc3066a3"}, + {file = "duckdb-1.4.4-cp314-cp314-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:d0440f59e0cd9936a9ebfcf7a13312eda480c79214ffed3878d75947fc3b7d6d"}, + {file = "duckdb-1.4.4-cp314-cp314-win_amd64.whl", hash = "sha256:59c8d76016dde854beab844935b1ec31de358d4053e792988108e995b18c08e7"}, + {file = "duckdb-1.4.4-cp314-cp314-win_arm64.whl", hash = "sha256:53cd6423136ab44383ec9955aefe7599b3fb3dd1fe006161e6396d8167e0e0d4"}, + {file = "duckdb-1.4.4-cp39-cp39-macosx_10_9_universal2.whl", hash = "sha256:8097201bc5fd0779d7fcc2f3f4736c349197235f4cb7171622936343a1aa8dbf"}, + {file = "duckdb-1.4.4-cp39-cp39-macosx_10_9_x86_64.whl", hash = "sha256:cd1be3d48577f5b40eb9706c6b2ae10edfe18e78eb28e31a3b922dcff1183597"}, + {file = "duckdb-1.4.4-cp39-cp39-macosx_11_0_arm64.whl", hash = "sha256:e041f2fbd6888da090eca96ac167a7eb62d02f778385dd9155ed859f1c6b6dc8"}, + {file = "duckdb-1.4.4-cp39-cp39-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:7eec0bf271ac622e57b7f6554a27a6e7d1dd2f43d1871f7962c74bcbbede15ba"}, + {file = "duckdb-1.4.4-cp39-cp39-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:5cdc4126ec925edf3112bc656ac9ed23745294b854935fa7a643a216e4455af6"}, + {file = "duckdb-1.4.4-cp39-cp39-win_amd64.whl", hash = "sha256:c9566a4ed834ec7999db5849f53da0a7ee83d86830c33f471bf0211a1148ca12"}, + {file = "duckdb-1.4.4.tar.gz", hash = "sha256:8bba52fd2acb67668a4615ee17ee51814124223de836d9e2fdcbc4c9021b3d3c"}, ] +[package.extras] +all = ["adbc-driver-manager", "fsspec", "ipython", "numpy", "pandas", "pyarrow"] + [[package]] name = "et-xmlfile" version = "2.0.0" @@ -3595,25 +3587,22 @@ test = ["pytest", "pytest-cov"] [[package]] name = "zensical" -version = "0.0.46" +version = "0.0.63" description = "A modern static site generator built by the creators of Material for MkDocs" optional = false python-versions = ">=3.10" groups = ["docs"] files = [ - {file = "zensical-0.0.46-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:d91af81ab058c8693dfd75f2f77b4c73bcba4125681d1d276f38624291820bd2"}, - {file = "zensical-0.0.46-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:d9221264a9a87409900a47e29985607b0c9245dacb89077e87c8e16e31edc167"}, - {file = "zensical-0.0.46-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:ec43018d5343ca2e1d71aa352eeddd560fef504effd03025840a5a783abefa4f"}, - {file = "zensical-0.0.46-cp310-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:26e98fb8ab7ab50cdd20a73e2c7d4d9aae0b46cf2d8691e6bb22f9c261b8a60a"}, - {file = "zensical-0.0.46-cp310-abi3-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:46fe578f26963f8ee89567983e62737b6fadc9197d4742e1020b522e092d7baa"}, - {file = "zensical-0.0.46-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:aef03fa186a5589148e10b62610500989c6b075a2c08e1554233adbf91b2a3dc"}, - {file = "zensical-0.0.46-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:bc7446cdf97a8dea390f20ed2bd6b030cddc1bd36a8ce113ea3efef6fa61c573"}, - {file = "zensical-0.0.46-cp310-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:bbee37801f1ed500f158dc0992c569282950f780ae353c37fe6969f99983d701"}, - {file = "zensical-0.0.46-cp310-abi3-musllinux_1_2_i686.whl", hash = "sha256:9487c147c9cceb50c04d0ad70b024821a6eab1629dafd70ab6d1e86ec841e623"}, - {file = "zensical-0.0.46-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:f42a4683c762f026878d19ede4bcf7bfbb84dbecb5ad923949abb77806ed88a5"}, - {file = "zensical-0.0.46-cp310-abi3-win32.whl", hash = "sha256:85f018f2a7ee76a83915c87ddb12b58cf343fd6154081d33ac95b6751b011dd7"}, - {file = "zensical-0.0.46-cp310-abi3-win_amd64.whl", hash = "sha256:1543a693a160de60e86ca589592401b584670e7e12c5ae30e3c2ba76786f7ec3"}, - {file = "zensical-0.0.46.tar.gz", hash = "sha256:3ec21f4fb1e78cd7c0d6b07ae336b04770e27ba020dabc457b2790e5d34f1978"}, + {file = "zensical-0.0.63-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:069af2ee0254eaa7099faad908f0ae0017c151ef0160b21a039d08efaa61b38a"}, + {file = "zensical-0.0.63-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:6ec02283d946569f68fdd1ab4a1dd2de560c0ad540874d5ecd4a4f1e96bdbe1d"}, + {file = "zensical-0.0.63-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:af2711ff1397338c730cc9232f687066da7181751257f317877501b1b77a44ea"}, + {file = "zensical-0.0.63-cp310-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:f1afb0e2819a588d956c1684c90e23fc0c7efc89e1ec141403a9fb41b75b1993"}, + {file = "zensical-0.0.63-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:925b3dd8bb6812b780f5e00828c42bd1eebd675b1eb6b4a58135926bb716bdde"}, + {file = "zensical-0.0.63-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:6365bce465a7f6755533eaa80aa20c8a48d1b39fa7b12cd394eece3921e2c530"}, + {file = "zensical-0.0.63-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:63706419125f4c46461b322c8aaed1bbc3b917dd389ecf29601d713fc962bfb6"}, + {file = "zensical-0.0.63-cp310-abi3-win_amd64.whl", hash = "sha256:43e223b8d6ad1f08a772232926c123a000abd30ffe98d9db63a2731ec6275329"}, + {file = "zensical-0.0.63-cp310-abi3-win_arm64.whl", hash = "sha256:73c29ca1cd00243384fdc69414fac1c4367814203116acc9d198121b7a91012a"}, + {file = "zensical-0.0.63.tar.gz", hash = "sha256:95f68b494fa6a11a59065f7965d27acfca404dd867f81d31295b5ab7172f2f8a"}, ] [package.dependencies] @@ -3622,7 +3611,7 @@ deepmerge = ">=2.0" jinja2 = ">=3.1" markdown = ">=3.7" pygments = ">=2.20" -pymdown-extensions = ">=10.21.3" +pymdown-extensions = ">=11.0" pyyaml = ">=6.0.2" tomli = ">=2.4.0" @@ -3649,4 +3638,4 @@ type = ["pytest-mypy (>=1.0.1) ; platform_python_implementation != \"PyPy\""] [metadata] lock-version = "2.1" python-versions = ">=3.10,<3.13" -content-hash = "3c6b964ad86fe375ec1862480b207189085a8d1da8e5017a8545b78dc0ff469b" +content-hash = "6d996888c9d149407fb885b454745bf763f9cba56404a29148ffe0b9fd57b0de" diff --git a/pyproject.toml b/pyproject.toml index 0ad62d32..5cac0282 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -7,6 +7,7 @@ authors = [ ] readme = "README.md" classifiers = [ + "Development Status :: 4 - Beta", "Programming Language :: Python :: 3", "Programming Language :: Python :: 3.10", "Programming Language :: Python :: 3.11", @@ -35,7 +36,7 @@ python = ">=3.10,<3.13" # breaking changes beyond 3.12 boto3 = ">=1.34.162,<1.36" # breaking change beyond 1.36 botocore = ">=1.34.162,<1.36" # breaking change beyond 1.36 delta-spark = ">=3.0.0,<=3.2.0" -duckdb = "1.1.3" # breaking changes beyond 1.1 +duckdb = ">=1.4,<1.4.5" Jinja2 = "3.1.6" lxml = "6.1.1" numpy = "1.26.4" @@ -108,7 +109,7 @@ mkdocs = "1.6.1" mkdocstrings = { version = "1.0.3", extras = ["python"] } griffelib = "2.0.1" pymdown-extensions = "11.0.1" -zensical = "0.0.46" +zensical = "0.0.63" [tool.ruff] line-length = 100 diff --git a/src/dve/common/error_utils.py b/src/dve/common/error_utils.py index 120c902d..23ad714f 100644 --- a/src/dve/common/error_utils.py +++ b/src/dve/common/error_utils.py @@ -5,7 +5,7 @@ import logging from collections.abc import Iterable from itertools import chain -from multiprocessing import Queue +from queue import Queue from threading import Thread from typing import Optional, Union diff --git a/src/dve/core_engine/backends/base/contract.py b/src/dve/core_engine/backends/base/contract.py index 948ff775..003f0b40 100644 --- a/src/dve/core_engine/backends/base/contract.py +++ b/src/dve/core_engine/backends/base/contract.py @@ -443,10 +443,6 @@ def apply( ], ) - if contract_metadata.cache_originals: - for entity_name in list(entities): - entities[f"Original{entity_name}"] = entities[entity_name] - return entities, feedback_errors_uri, successful, processing_errors_uri def read_parquet(self, path: URI, **kwargs) -> EntityType: diff --git a/src/dve/core_engine/backends/base/reader.py b/src/dve/core_engine/backends/base/reader.py index ae0e99f1..eeeca7f4 100644 --- a/src/dve/core_engine/backends/base/reader.py +++ b/src/dve/core_engine/backends/base/reader.py @@ -8,8 +8,16 @@ from pydantic import BaseModel from typing_extensions import Protocol -from dve.core_engine.backends.exceptions import MessageBearingError, ReaderLacksEntityTypeSupport +from dve.core_engine.backends.exceptions import ( + CriticalMessageBearingError, + MessageBearingError, + ReaderLacksEntityTypeSupport, +) from dve.core_engine.backends.types import EntityName, EntityType +from dve.core_engine.configuration.v1 import ( + AllowedAdditionalReaderChecks, + _ReaderAdditionalChecksConfig, +) from dve.core_engine.message import FeedbackMessage from dve.core_engine.type_hints import URI, ArbitraryFunction, WrapDecorator from dve.parser.file_handling.service import open_stream @@ -65,6 +73,10 @@ class BaseFileReader(ABC): decorated with the '@read_function' decorator, and is used in `read_entity_type`. """ + ft_error_code: Optional[str] = "MalformedFile" + """Default error code for when a file/submission cannot be parsed succesfully.""" + ft_error_message: Optional[str] = "The resource doesn't seem to be a valid text file" + """Default error message for when a file/submission cannot be parsed succesfully.""" def __init_subclass__(cls, *_, **__) -> None: """When this class is subclassed, create and populate the `__read_methods__` @@ -109,7 +121,10 @@ def read_to_entity_type( entity_name: EntityName, schema: type[BaseModel], all_model_fields: Optional[set[str]] = None, - ) -> EntityType: + additional_checks: Optional[ + dict[AllowedAdditionalReaderChecks, _ReaderAdditionalChecksConfig] + ] = None, + ): """Read to the specified entity type, if supported. NOTE: Simple types should either be returned as strings (if present) or @@ -117,21 +132,47 @@ def read_to_entity_type( data contract. """ - if entity_name == Iterator[dict[str, Any]]: - return self.read_to_py_iterator( + additional_checks = additional_checks or {} + + self.raise_if_not_sensible_file(resource, entity_name) + + if entity_type == Iterator[dict[str, Any]]: + entity = self.read_to_py_iterator( resource, entity_name, schema, all_model_fields # type: ignore ) - self.raise_if_not_sensible_file(resource, entity_name) + else: - try: - reader_func = self.__read_methods__[entity_type] - except KeyError as err: - raise ReaderLacksEntityTypeSupport(entity_type=entity_type) from err + try: + reader_func = self.__read_methods__[entity_type] + except KeyError as err: + raise ReaderLacksEntityTypeSupport(entity_type=entity_type) from err - return reader_func( - self, resource, entity_name, schema, all_model_fields=all_model_fields # type: ignore - ) + entity = reader_func( + self, + resource, + entity_name, + schema, + all_model_fields=all_model_fields, # type: ignore + ) + + if config := additional_checks.get("check_empty"): + if self.check_entity_empty(entity): + raise MessageBearingError( + f"The mandatory entity {entity_name} is empty", + messages=[ + FeedbackMessage( + entity=entity_name, + record=None, + failure_type="submission", + error_location=entity_name, + error_code=config.error_code, + error_message=config.error_message, + ) + ], + ) + + return entity def add_record_index(self, entity: EntityType, **kwargs) -> EntityType: """Add a record index to the entity""" @@ -141,6 +182,10 @@ def drop_record_index(self, entity: EntityType, **kwargs) -> EntityType: """Drop a record index to the entity""" raise NotImplementedError(f"drop_record_index not implemented in {self.__class__}") + def check_entity_empty(self, entity: EntityType) -> bool: + """Determine if the entity supplied is empty""" + raise NotImplementedError(f"check_entity_empty not implemented in {self.__class__}") + def write_parquet( self, entity: EntityType, @@ -171,20 +216,22 @@ def _check_likely_text_file(resource: URI) -> bool: return False return True - def raise_if_not_sensible_file(self, resource: URI, entity_name: str): + def raise_if_not_sensible_file( + self, + resource: URI, + entity_name: str, + ): """Sense check that the file is a text file. Raise error if doesn't appear to be the case.""" if not self._check_likely_text_file(resource): - raise MessageBearingError( + raise CriticalMessageBearingError( "The submitted file doesn't appear to be text", - messages=[ - FeedbackMessage( - entity=entity_name, - record=None, - failure_type="submission", - error_location="Whole File", - error_code="MalformedFile", - error_message="The resource doesn't seem to be a valid text file", - ) - ], + message=FeedbackMessage( + entity=entity_name, + record=None, + failure_type="submission", + error_location="Whole File", + error_code=self.ft_error_code, + error_message=self.ft_error_message, + ), ) diff --git a/src/dve/core_engine/backends/base/rules.py b/src/dve/core_engine/backends/base/rules.py index 9b6b4fe9..9f19c024 100644 --- a/src/dve/core_engine/backends/base/rules.py +++ b/src/dve/core_engine/backends/base/rules.py @@ -3,8 +3,8 @@ import logging from abc import ABC, abstractmethod from collections import defaultdict -from collections.abc import Iterable -from typing import Any, ClassVar, Generic, NoReturn, Optional, TypeVar +from collections.abc import Iterable, Iterator +from typing import Any, ClassVar, Generic, MutableMapping, NoReturn, Optional, TypeVar from uuid import uuid4 from typing_extensions import Literal, Protocol, get_type_hints @@ -27,6 +27,7 @@ CopyEntity, DeferredFilter, EntityRemoval, + GroupIdentification, HeaderJoin, ImmediateFilter, InnerJoin, @@ -43,8 +44,11 @@ TableUnion, ) from dve.core_engine.backends.types import Entities, EntityType, StageSuccessful +from dve.core_engine.configuration.v1.hierarchy import EntityHierarchy, HierarchyNode from dve.core_engine.exceptions import CriticalProcessingError from dve.core_engine.loggers import get_logger +from dve.core_engine.message import FeedbackMessage +from dve.core_engine.templating import template_object from dve.core_engine.type_hints import URI, DVEStageName, EntityName, Messages, TemplateVariables T_contra = TypeVar("T_contra", bound=AbstractStep, contravariant=True) @@ -55,6 +59,8 @@ """A convenience type indicating a mapping from config type to step method.""" Stage = Literal["Pre-filter", "Filter", "Post-filter"] """The name of a stage within a rule.""" +TempTableName = str +"""temp tables to cache intermediate results""" class _UnboundStepFunction(Generic[T_contra], Protocol): # pylint: disable=too-few-public-methods @@ -128,6 +134,7 @@ def __init__( # pylint: disable=unused-argument ): self.logger = logger or get_logger(type(self).__name__) """The `logging.Logger instance for the data contract config.""" + self.entity_cache_tracker: MutableMapping[EntityName, TempTableName] = {} @classmethod @abstractmethod @@ -307,7 +314,8 @@ def join_header(self, entities: Entities, *, config: HeaderJoin) -> Messages: """ raise NotImplementedError - def identify_orphans(self, entities: Entities, *, config: OrphanIdentification) -> Messages: + @abstractmethod + def identify_orphans(self, entities: Entities, *, config: OrphanIdentification) -> Iterable: """Identify records in an entity which don't have at least one corresponding match in the target. A new boolean column will be added to `entity` ('IsOrphaned') indicating whether the condition matched. @@ -320,6 +328,14 @@ def identify_orphans(self, entities: Entities, *, config: OrphanIdentification) """ raise NotImplementedError + @abstractmethod + def check_mandatory_group(self, entities: Entities, *, config: GroupIdentification) -> Iterator: + """ + Check that a mandatory key in an entity has at least one valid entry in the all the child + entities. + """ + raise NotImplementedError + @abstractmethod def union(self, entities: Entities, *, config: TableUnion) -> Messages: """Union two entities together, taking the columns from each by name. @@ -352,6 +368,157 @@ def notify(self, entities: Entities, *, config: Notification) -> Messages: """ + def identify_and_remove_orphans( + self, + working_directory: URI, + entities: Entities, + entity_hierarchy: EntityHierarchy, + key_fields: Optional[dict[str, list[str]]] = None, + ) -> tuple[Messages, dict[EntityName, bool]]: + """ + Identifies and removes orphan records by traversing the EntityHierarchy object. + An orphan is a child record whose parent FK does not exist in the parent entity. + Processes recursively: removes orphans at each level, then processes children. + """ + + def process_node(node: HierarchyNode) -> bool: + """Identify orphans and remove in a given node""" + issues_found: bool = False + if node.parent_entity is None: + return issues_found + + self.logger.info(f"Checking for orphan records in {node.entity_name}") + + join_expr = " AND ".join( + f"{node.parent_entity}.{k} = {node.entity_name}.{v}" + for k, v in node.join_fields.items() + ) + location = list(node.join_fields.values())[0] + with BackgroundMessageWriter( + working_directory=working_directory, + dve_stage=self.__stage_name__, + key_fields=key_fields, + logger=self.logger, + ) as msg_writer: + _orph_records = self.identify_orphans( + entities=entities, + config=OrphanIdentification( + id=list(node.join_fields.values())[0], + entity_name=node.entity_name, + target_name=node.parent_entity, + join_condition=join_expr, + ), + ) + _messages = [ + FeedbackMessage( + entity=node.entity_name, + record=record, # type: ignore + error_location=location, + error_message=template_object(node.missing_parent_id_error_message, record), + failure_type="record", + error_type="record", + error_code=node.missing_parent_id_error_code, + reporting_field=location, + category="Parent Missing", + ) + for record in _orph_records + ] + msg_writer.write_queue.put(_messages) + + self.cache_entity(node.entity_name, entities) + + _orph_count = len(_messages) + + self.logger.info( + f"Found {_orph_count} orphan records between {node.entity_name} and {node.parent_entity}" # pylint: disable=C0301 + ) + + return _orph_count > 0 + + entity_issues_found: dict[EntityName, bool] = {} + + for tree in entity_hierarchy.entity_trees.values(): + for node in tree.iterate_root_down(): + entity_issues_found[node.entity_name] = process_node(node) + + return [], entity_issues_found + + def identify_and_remove_missing_mandatory_groups( + self, + working_directory: URI, + entities: Entities, + entity_hierarchy: EntityHierarchy, + key_fields: Optional[dict[str, list[str]]] = None, + ) -> tuple[Messages, dict[EntityName, bool]]: + """ + Identify that an entity with a mandatory key has at least one valid child record. + """ + + def process_node(node: HierarchyNode) -> bool: + """Identify at least one valid child for a mandatory entity at a given node.""" + if node.parent_entity is None or not node.mandatory: + return False + + self.logger.info( + f"Identifying that mandatory entity `{node.parent_entity}` has at least 1 valid child record in {node.entity_name}" # pylint: disable=C0301 + ) + + join_expr = " AND ".join( + f"{node.parent_entity}.{k} = {node.entity_name}.{v}" + for k, v in node.join_fields.items() + ) + + with BackgroundMessageWriter( + working_directory=working_directory, + dve_stage=self.__stage_name__, + key_fields=key_fields, + logger=self.logger, + ) as msg_writer: + location = next(iter(node.join_fields.values())) + missing_children_records = self.check_mandatory_group( + entities=entities, + config=GroupIdentification( + entity_name=node.parent_entity, + target_name=node.entity_name, + join_condition=join_expr, + ), + ) + _messages = [ + FeedbackMessage( + entity=node.parent_entity, + record=record, # type: ignore + error_location=location, + error_message=template_object(node.no_valid_records_error_message, record), + failure_type="record", + error_type="record", + error_code=node.no_valid_records_error_code, + reporting_field=location, + category="Children missing", + ) + for record in missing_children_records + ] + msg_writer.write_queue.put(_messages) + self.cache_entity(node.parent_entity, entities) + + _no_valid_child_records: int = len(_messages) + + self.logger.info( + f"Found {_no_valid_child_records} records with no valid children in {node.parent_entity}." # pylint: disable=C0301 + ) + + return _no_valid_child_records > 0 + + entity_issues_found: dict[EntityName, bool] = {} + + for tree in entity_hierarchy.entity_trees.values(): + for node in tree.iterate_lowest_descendent_up(): + if node.parent_entity and node.mandatory: + result = process_node(node) + if result: + entity_issues_found[node.parent_entity] = result + + return [], entity_issues_found + # pylint: disable=R0912,R0914 def apply_sync_filters( self, @@ -428,6 +595,7 @@ def apply_sync_filters( excluded_columns=filter_column_names, reporting=rule.reporting, parent=rule.parent, + error_on_null=True, ), ) if not success: @@ -459,6 +627,7 @@ def apply_sync_filters( expression=f"NOT ({rule.expression})", reporting=rule.reporting, parent=rule.parent, + error_on_null=True, ), ) if not success: @@ -691,3 +860,25 @@ def filter_data_contract_record_rejections( ): """Method to filter out record rejection errors from the data contract for a given entity""" raise NotImplementedError() + + @staticmethod + def get_entity_count(entity: EntityType) -> int: + """Method to get count of records in entity""" + raise NotImplementedError() + + def cache_entity(self, entity_name: EntityName, entities: Entities): + """Store the materialised query in memory and update entity to query directly. + If the entity is already cached, the new cache should be created first, then the old one + removed as part of the function (in case the newer cache depends on the older one).""" + raise NotImplementedError() + + def _remove_cached_artifact(self, entity_name: EntityName): + """Delete artifact in memory and clear from the entity cache keeping track. + This should not be used directly as removing artifacts from memory, may lead to some + entities being unable to be processed as their execution plans depend on these artifacts.""" + raise NotImplementedError() + + def clear_entity_cache(self): + """Helper method to remove all artifacts and cache trackers at end of processing.""" + for entity_name in list(self.entity_cache_tracker): + self._remove_cached_artifact(entity_name) diff --git a/src/dve/core_engine/backends/exceptions.py b/src/dve/core_engine/backends/exceptions.py index bb585168..99808c8a 100644 --- a/src/dve/core_engine/backends/exceptions.py +++ b/src/dve/core_engine/backends/exceptions.py @@ -33,27 +33,39 @@ def __init__(self, *args: object, messages: Messages) -> None: """The messages to be returned as part of the error.""" -class UnableToParseCSVError(MessageBearingError): +class CriticalMessageBearingError(BackendError): + """ + A backend error that comes with a pre-created message. + The intention of this exception vs MessageBearingError is that + this should be used to halt the processing of a submission. + """ + + def __init__(self, *args: object, message: FeedbackMessage) -> None: + super().__init__(*args) + self.message = message + """The message to be returned as part of the error.""" + + +class UnableToParseCSVError(CriticalMessageBearingError): """An error raised when unable to parse a CSV file""" def __init__( - self, entity_name: str, field_check_error_message: str, field_check_error_code: str + self, + entity_name: Optional[str], + error_message: Optional[str], + error_code: Optional[str], ): super().__init__( - messages=[ - FeedbackMessage( - entity="csv_structure", - record={ - entity_name: "Unable to parse file. Please check the structure of the file." - }, - failure_type="submission", - is_informational=False, - error_type="csv read", - error_location=entity_name, - error_message=field_check_error_message, - error_code=field_check_error_code, - ) - ] + message=FeedbackMessage( + entity=entity_name, + record=None, + failure_type="submission", + is_informational=False, + error_type="csv read", + error_message=error_message + or "Unable to parse the CSV file. Please check the structure of your CSV.", # pylint: disable=C0301 + error_code=error_code or "MalformedCSV", + ) ) diff --git a/src/dve/core_engine/backends/implementations/duckdb/duckdb_helpers.py b/src/dve/core_engine/backends/implementations/duckdb/duckdb_helpers.py index 588cd7e0..55db0945 100644 --- a/src/dve/core_engine/backends/implementations/duckdb/duckdb_helpers.py +++ b/src/dve/core_engine/backends/implementations/duckdb/duckdb_helpers.py @@ -3,6 +3,7 @@ """Helper objects for duckdb data contract implementation""" +import itertools from collections.abc import Generator, Iterator from dataclasses import is_dataclass from datetime import date, datetime, time @@ -99,7 +100,12 @@ def __call__(self): def table_exists(connection: DuckDBPyConnection, table_name: str) -> bool: """check if a table exists in a given DuckDBPyConnection""" - return table_name in map(lambda x: x[0], connection.sql("SHOW TABLES").fetchall()) + return table_name in get_all_existing_ddb_tables(connection) + + +def get_all_existing_ddb_tables(connection: DuckDBPyConnection) -> tuple[str]: + """Fetch all tables available ina given duckdb connection""" + return tuple(itertools.chain.from_iterable(connection.sql("SHOW TABLES").fetchall())) def relation_is_empty(relation: DuckDBPyRelation) -> bool: @@ -284,11 +290,11 @@ def _ddb_filter_contract_errors( "RecordIndex": "INTEGER", "FailureType": "STRING", "Status": "STRING", - "Entity": "STRING", + "OriginalEntity": "STRING", }, ) .filter( - f"FailureType == 'record' AND Status != 'informational' AND Entity = '{entity_name}'" + f"FailureType == 'record' AND Status != 'informational' AND OriginalEntity = '{entity_name}'" # pylint: disable=C0301 ) # pylint: disable=C0301 .select("RecordIndex") .distinct() @@ -322,6 +328,16 @@ def duckdb_get_entity_count(cls): return cls +def _duckdb_check_entity_empty(self, entity: DuckDBPyRelation) -> bool: # pylint: disable=W0613 + return entity.shape[0] == 0 + + +def duckdb_check_entity_empty(cls): + """Class decorator to check whether a supplied entity is empty""" + cls.check_entity_empty = _duckdb_check_entity_empty + return cls + + def get_all_registered_udfs(connection: DuckDBPyConnection) -> set[str]: """Function to supply the names of a registered functions stored in the supplied duckdb connection. Creates the temp table used to store registered functions (if not exists). diff --git a/src/dve/core_engine/backends/implementations/duckdb/readers/csv.py b/src/dve/core_engine/backends/implementations/duckdb/readers/csv.py index 723e5e37..488f8c85 100644 --- a/src/dve/core_engine/backends/implementations/duckdb/readers/csv.py +++ b/src/dve/core_engine/backends/implementations/duckdb/readers/csv.py @@ -22,19 +22,22 @@ UnableToParseCSVError, ) from dve.core_engine.backends.implementations.duckdb.duckdb_helpers import ( + duckdb_check_entity_empty, duckdb_record_index, duckdb_write_parquet, get_duckdb_type_from_annotation, + relation_is_empty, ) from dve.core_engine.backends.implementations.duckdb.types import SQLType from dve.core_engine.backends.readers.csv import CSVFileReader from dve.core_engine.backends.utilities import get_polars_type_from_annotation, polars_record_index -from dve.core_engine.constants import RECORD_INDEX_COLUMN_NAME +from dve.core_engine.constants import PRE_VALIDATION_ENTITY, RECORD_INDEX_COLUMN_NAME from dve.core_engine.message import FeedbackMessage from dve.core_engine.type_hints import URI, EntityName from dve.parser.file_handling import get_content_length +@duckdb_check_entity_empty @duckdb_record_index @duckdb_write_parquet class DuckDBCSVReader(CSVFileReader): @@ -42,10 +45,7 @@ class DuckDBCSVReader(CSVFileReader): to the file header, if it exists. field_check: flag to compare submitted file header to the accompanying pydantic model - field_check_error_code: The error code to provide if the file header doesn't contain - the expected fields - field_check_error_message: The error message to provide if the file header doesn't contain - the expected fields""" + """ # TODO - the read_to_relation should include the schema and determine whether to # TODO - stringify or not @@ -57,8 +57,8 @@ def __init__( quotechar: str = '"', connection: Optional[DuckDBPyConnection] = None, field_check: bool = False, - field_check_error_code: str = "ExpectedVsActualFieldMismatch", - field_check_error_message: str = "The submitted header is missing fields", + ft_error_code: str = "ExpectedVsActualFieldMismatch", + ft_error_message: str = "The submitted header is missing fields", null_empty_strings: bool = False, **_, ): @@ -70,8 +70,8 @@ def __init__( delimiter=delim, quote_char=quotechar, field_check=field_check, - field_check_error_code=field_check_error_code, - field_check_error_message=field_check_error_message, + ft_error_code=ft_error_code, + ft_error_message=ft_error_message, ) def read_to_py_iterator( @@ -122,8 +122,9 @@ def read_to_relation( # pylint: disable=unused-argument except InvalidInputException as exc: raise UnableToParseCSVError( entity_name="csv_structure", - field_check_error_message=self.field_check_error_message, - field_check_error_code=self.field_check_error_code, + error_code=self.ft_error_code, + error_message=self.ft_error_message + or "Unable to parse CSV file. Structure is likely malformed.", # pylint: disable=C0301 ) from exc if self.null_empty_strings: @@ -182,8 +183,9 @@ def read_to_relation( # pylint: disable=unused-argument except pl.exceptions.PolarsError as exc: raise UnableToParseCSVError( entity_name="csv_structure", - field_check_error_message=self.field_check_error_message, - field_check_error_code=self.field_check_error_code, + error_code=self.ft_error_code, + error_message=self.ft_error_message + or "Unable to parse CSV file. Structure is likely malformed.", # pylint: disable=C0301 ) from exc if self.null_empty_strings: @@ -196,11 +198,12 @@ def read_to_relation( # pylint: disable=unused-argument entity = self._connection.sql("SELECT * FROM df") - if entity.pl().shape[0] == 0: + if relation_is_empty(entity): raise UnableToParseCSVError( entity_name="csv_structure", - field_check_error_message=self.field_check_error_message, - field_check_error_code=self.field_check_error_code, + error_code=self.ft_error_code, + error_message=self.ft_error_message + or "Found zero records after loading CSV. File is likely malformed.", # pylint: disable=C0301 ) return entity @@ -271,7 +274,7 @@ def read_to_relation( # pylint: disable=unused-argument messages=[ FeedbackMessage( record={entity_name: differing_values}, - entity="Pre-validation", + entity=PRE_VALIDATION_ENTITY, failure_type="submission", error_message=( f"Found {no_records} distinct combination of header values." diff --git a/src/dve/core_engine/backends/implementations/duckdb/readers/json.py b/src/dve/core_engine/backends/implementations/duckdb/readers/json.py index 79d74c62..84b601de 100644 --- a/src/dve/core_engine/backends/implementations/duckdb/readers/json.py +++ b/src/dve/core_engine/backends/implementations/duckdb/readers/json.py @@ -10,6 +10,7 @@ from dve.core_engine.backends.base.reader import BaseFileReader, read_function from dve.core_engine.backends.implementations.duckdb.duckdb_helpers import ( + duckdb_check_entity_empty, duckdb_record_index, duckdb_write_parquet, get_duckdb_type_from_annotation, @@ -18,6 +19,7 @@ from dve.core_engine.type_hints import URI, EntityName +@duckdb_check_entity_empty @duckdb_record_index @duckdb_write_parquet class DuckDBJSONReader(BaseFileReader): diff --git a/src/dve/core_engine/backends/implementations/duckdb/readers/xml.py b/src/dve/core_engine/backends/implementations/duckdb/readers/xml.py index 42e281a0..7e591e58 100644 --- a/src/dve/core_engine/backends/implementations/duckdb/readers/xml.py +++ b/src/dve/core_engine/backends/implementations/duckdb/readers/xml.py @@ -9,8 +9,11 @@ from pydantic import BaseModel from dve.core_engine.backends.base.reader import read_function -from dve.core_engine.backends.exceptions import MessageBearingError -from dve.core_engine.backends.implementations.duckdb.duckdb_helpers import duckdb_write_parquet +from dve.core_engine.backends.exceptions import CriticalMessageBearingError +from dve.core_engine.backends.implementations.duckdb.duckdb_helpers import ( + duckdb_check_entity_empty, + duckdb_write_parquet, +) from dve.core_engine.backends.readers.xml import XMLStreamReader from dve.core_engine.backends.utilities import ( get_polars_type_from_annotation, @@ -20,6 +23,7 @@ from dve.core_engine.type_hints import URI +@duckdb_check_entity_empty @polars_record_index @duckdb_write_parquet class DuckDBXMLStreamReader(XMLStreamReader): @@ -41,9 +45,9 @@ def read_to_relation( if self.xsd_location: msg = self._run_xmllint(file_uri=resource) if msg: - raise MessageBearingError( + raise CriticalMessageBearingError( "Submitted file failed XSD validation.", - messages=[msg], + message=msg, ) polars_schema: dict[str, pl.DataType] = { # type: ignore diff --git a/src/dve/core_engine/backends/implementations/duckdb/rules.py b/src/dve/core_engine/backends/implementations/duckdb/rules.py index dc73dad2..766142aa 100644 --- a/src/dve/core_engine/backends/implementations/duckdb/rules.py +++ b/src/dve/core_engine/backends/implementations/duckdb/rules.py @@ -1,6 +1,7 @@ """Business rule definitions for duckdb backend""" -from collections.abc import Callable +# pylint: disable=R0801 +from collections.abc import Callable, Iterable, Iterator from typing import get_type_hints from uuid import uuid4 @@ -23,12 +24,14 @@ from dve.core_engine.backends.implementations.duckdb.duckdb_helpers import ( DDBStruct, ddb_filter_contract_errors, + duckdb_get_entity_count, duckdb_read_parquet, duckdb_record_index, duckdb_rel_to_dictionaries, duckdb_write_parquet, get_all_registered_udfs, get_duckdb_type_from_annotation, + relation_is_empty, ) from dve.core_engine.backends.implementations.duckdb.types import ( DuckDBEntities, @@ -43,6 +46,7 @@ Aggregation, AntiJoin, ConfirmJoinHasMatch, + GroupIdentification, HeaderJoin, ImmediateFilter, InnerJoin, @@ -53,12 +57,17 @@ SemiJoin, TableUnion, ) +from dve.core_engine.constants import RECORD_INDEX_COLUMN_NAME from dve.core_engine.functions import implementations as functions from dve.core_engine.message import FeedbackMessage from dve.core_engine.templating import template_object -from dve.core_engine.type_hints import Messages +from dve.core_engine.type_hints import EntityName, Messages +TempTableName = str +"""temp tables to cache intermediate results""" + +@duckdb_get_entity_count @duckdb_record_index @duckdb_write_parquet @duckdb_read_parquet @@ -364,7 +373,7 @@ def join_header(self, entities: DuckDBEntities, *, config: HeaderJoin) -> Messag ), ) - target_schema = DDBStruct(dict(zip(target_rel.columns, target_rel.dtypes)))() + target_schema = DDBStruct(dict(zip(target_rel.columns, target_rel.dtypes)))() # type: ignore # pylint:disable=C0301 joined_rel = source_rel.select( StarExpression(exclude=[]), @@ -375,8 +384,11 @@ def join_header(self, entities: DuckDBEntities, *, config: HeaderJoin) -> Messag return [] def identify_orphans( - self, entities: DuckDBEntities, *, config: OrphanIdentification - ) -> Messages: + self, + entities: DuckDBEntities, + *, + config: OrphanIdentification, + ) -> Iterable: """Identify records in an entity which don't have at least one corresponding match in the target. A new boolean column will be added to `entity` ('IsOrphaned') indicating whether the condition matched. @@ -390,41 +402,75 @@ def identify_orphans( target_rel: DuckDBPyRelation = entities[config.target_name] target_rel = target_rel.set_alias(config.target_name) - key_name = f"key_{uuid4().hex}" - source_rel = source_rel.select(f"*, row_number() over () as {key_name}").set_alias( - config.entity_name - ) + if relation_is_empty(source_rel): + self.logger.info(f"{config.entity_name} is empty. Skipping orphan check.") + return [] + match_name = f"matched_{uuid4().hex}" target_rel = target_rel.select( StarExpression(exclude=[]), ConstantExpression(1).alias(match_name) ).set_alias(config.target_name) - joined_rel: DuckDBPyRelation = source_rel.join( - target_rel, condition=config.join_condition, how="left" - ).aggregate(f"{key_name}, coalesce(count({match_name})==0, TRUE) AS IsOrphaned") - - if "IsOrphaned" not in source_rel.columns: - result: DuckDBPyRelation = source_rel.join( - joined_rel, condition=key_name, how="left" - ).select(StarExpression(exclude=[key_name])) - else: - result = source_rel.set_alias("source").join( - joined_rel.set_alias("joined"), - condition=f"source.{key_name} = joined.{key_name}", - how="left", + orphaned_rel: DuckDBPyRelation = ( + source_rel.join(target_rel, condition=config.join_condition, how="left") + .aggregate( + f"{config.entity_name}.{RECORD_INDEX_COLUMN_NAME}, {config.entity_name}.{config.id}, coalesce(count({match_name}), 0)==0 AS IsOrphaned" # pylint: disable=C0301 ) + .filter("IsOrphaned") + .select(RECORD_INDEX_COLUMN_NAME) + .set_alias("orphan") + ) - columns = {name: f"source.{name}" for name in source_rel.columns} - if "IsOrphaned" in source_rel.columns: - columns["IsOrphaned"] = ColumnExpression("source.IsOrphaned") | ColumnExpression("joined.IsOrphaned") # type: ignore # pylint: disable=line-too-long - columns.pop(key_name, None) + message_rel = ( + entities[config.entity_name] + .set_alias(config.entity_name) + .join( + orphaned_rel, + f"{config.entity_name}.{RECORD_INDEX_COLUMN_NAME} = orphan.{RECORD_INDEX_COLUMN_NAME}", # pylint: disable=C0301 + "semi", + ) + ) - result = result.select( - ",".join([f"{column} as {name}" for name, column in columns.items()]) + filtered_rel = ( + entities[config.entity_name] + .set_alias(config.entity_name) + .join( + orphaned_rel, + f"{config.entity_name}.{RECORD_INDEX_COLUMN_NAME} = orphan.{RECORD_INDEX_COLUMN_NAME}", # pylint: disable=C0301 + "anti", ) + ) - entities[config.new_entity_name or config.entity_name] = result - return [] + entities[config.entity_name] = filtered_rel + + return duckdb_rel_to_dictionaries(message_rel) + + def check_mandatory_group( + self, entities: DuckDBEntities, *, config: GroupIdentification + ) -> Iterator: + """ + Check that a mandatory key in an entity has at least one valid entry in the all the + child entities. + """ + source_rel: DuckDBPyRelation = entities[config.entity_name] + source_rel = source_rel.set_alias(config.entity_name) + target_rel: DuckDBPyRelation = entities[config.target_name] + target_rel = target_rel.set_alias(config.target_name) + + source_columns = [f"{config.entity_name}.{c.strip()}" for c in source_rel.columns] + _pk, fk = config.join_condition.split("=") + + joined_rel = source_rel.join(target_rel, config.join_condition, "left").select( + *source_columns, + ColumnExpression(fk.strip()).alias("fk"), + ) + + missing_children_rel = joined_rel.filter("fk IS NULL") + filtered_rel = joined_rel.filter("fk IS NOT NULL").select(StarExpression(exclude=["fk"])) + + entities[config.entity_name] = filtered_rel + + return duckdb_rel_to_dictionaries(missing_children_rel) def union(self, entities: DuckDBEntities, *, config: TableUnion) -> Messages: """Union two entities together, taking the columns from each by name. @@ -496,7 +542,12 @@ def notify(self, entities: DuckDBEntities, *, config: Notification) -> Messages: """ messages: Messages = [] entity = entities[config.entity_name] - + if config.error_if_expression_null: + if self.get_entity_count(entity.filter(f"({config.expression}) IS NULL")) > 0: + raise ValueError( + f"The filter evaluated for error code {config.reporting.code}" + + f" in entity {config.entity_name} produced some NULL results. Please investigate." # pylint: disable=C0301 + ) matched = entity.filter(config.expression) if config.excluded_columns: matched = matched.select(StarExpression(exclude=config.excluded_columns)) @@ -521,3 +572,21 @@ def notify(self, entities: DuckDBEntities, *, config: Notification) -> Messages: ) ) return messages + + def cache_entity(self, entity_name: EntityName, entities: DuckDBEntities): + """Store the materialised query in memory and update entity to query directly. + If the entity is already cached, the new cache should be created first, then the old one + removed as part of the function (in case the newer cache depends on the older one).""" + _tmp_name = f"{entity_name}_{uuid4().hex}" + + if entity := entities.get(entity_name): # pylint: disable=W0612 + self.connection.sql(f"CREATE OR REPLACE TEMP TABLE {_tmp_name} AS SELECT * FROM entity") + entities[entity_name] = self.connection.table(_tmp_name) + + self._remove_cached_artifact(entity_name) + + self.entity_cache_tracker[entity_name] = _tmp_name + + def _remove_cached_artifact(self, entity_name: EntityName): + if _tbl := self.entity_cache_tracker.pop(entity_name, None): + self.connection.sql(f"DROP TABLE IF EXISTS {_tbl}") diff --git a/src/dve/core_engine/backends/implementations/spark/contract.py b/src/dve/core_engine/backends/implementations/spark/contract.py index d2fd9ae1..432a7314 100644 --- a/src/dve/core_engine/backends/implementations/spark/contract.py +++ b/src/dve/core_engine/backends/implementations/spark/contract.py @@ -156,8 +156,9 @@ def apply_data_contract( fld, fld_info.annotation ).alias(fld) if fld in record_df.columns - else lit(None).cast( - get_type_from_annotation(fld_info.annotation)).alias(fld) + else lit(None) + .cast(get_type_from_annotation(fld_info.annotation)) + .alias(fld) ) for fld, fld_info in entity_fields.items() ], diff --git a/src/dve/core_engine/backends/implementations/spark/readers/csv.py b/src/dve/core_engine/backends/implementations/spark/readers/csv.py index 2df30c5c..6b79bb35 100644 --- a/src/dve/core_engine/backends/implementations/spark/readers/csv.py +++ b/src/dve/core_engine/backends/implementations/spark/readers/csv.py @@ -12,6 +12,7 @@ from dve.core_engine.backends.exceptions import EmptyFileError from dve.core_engine.backends.implementations.spark.spark_helpers import ( get_type_from_annotation, + spark_check_entity_empty, spark_record_index, spark_write_parquet, ) @@ -20,6 +21,7 @@ from dve.parser.file_handling import get_content_length +@spark_check_entity_empty @spark_record_index @spark_write_parquet class SparkCSVReader(CSVFileReader): @@ -38,8 +40,8 @@ def __init__( null_empty_strings: bool = False, spark_session: Optional[SparkSession] = None, field_check: bool = False, - field_check_error_code: str = "ExpectedVsActualFieldMismatch", - field_check_error_message: str = "The submitted header is missing fields", + ft_error_code: str = "ExpectedVsActualFieldMismatch", + ft_error_message: str = "The submitted header is missing fields", **_, ) -> None: @@ -54,8 +56,8 @@ def __init__( quote_char=quote_char, header=header, field_check=field_check, - field_check_error_code=field_check_error_code, - field_check_error_message=field_check_error_message, + ft_error_code=ft_error_code, + ft_error_message=ft_error_message, ) def read_to_py_iterator( diff --git a/src/dve/core_engine/backends/implementations/spark/readers/json.py b/src/dve/core_engine/backends/implementations/spark/readers/json.py index 61230091..6404231c 100644 --- a/src/dve/core_engine/backends/implementations/spark/readers/json.py +++ b/src/dve/core_engine/backends/implementations/spark/readers/json.py @@ -11,6 +11,7 @@ from dve.core_engine.backends.exceptions import EmptyFileError from dve.core_engine.backends.implementations.spark.spark_helpers import ( get_type_from_annotation, + spark_check_entity_empty, spark_record_index, spark_write_parquet, ) @@ -18,6 +19,7 @@ from dve.parser.file_handling import get_content_length +@spark_check_entity_empty @spark_record_index @spark_write_parquet class SparkJSONReader(BaseFileReader): diff --git a/src/dve/core_engine/backends/implementations/spark/readers/xml.py b/src/dve/core_engine/backends/implementations/spark/readers/xml.py index ba42d29f..769c480e 100644 --- a/src/dve/core_engine/backends/implementations/spark/readers/xml.py +++ b/src/dve/core_engine/backends/implementations/spark/readers/xml.py @@ -17,6 +17,7 @@ from dve.core_engine.backends.implementations.spark.spark_helpers import ( df_is_empty, get_type_from_annotation, + spark_check_entity_empty, spark_record_index, spark_write_parquet, ) @@ -29,6 +30,7 @@ """The mode to use when parsing XML files with Spark.""" +@spark_check_entity_empty @spark_record_index @spark_write_parquet class SparkXMLStreamReader(XMLStreamReader): @@ -42,6 +44,7 @@ def read_to_dataframe( resource: URI, entity_name: EntityName, schema: type[BaseModel], + all_model_fields: Optional[set[str]] = None, ) -> DataFrame: """Stream an XML file into a Spark data frame""" if not self.spark: @@ -49,12 +52,13 @@ def read_to_dataframe( spark_schema = get_type_from_annotation(schema) return self.add_record_index( self.spark.createDataFrame( # type: ignore - list(self.read_to_py_iterator(resource, entity_name, schema)), + list(self.read_to_py_iterator(resource, entity_name, schema, all_model_fields)), schema=spark_schema, ) ) +@spark_check_entity_empty @spark_record_index @spark_write_parquet class SparkXMLReader(BasicXMLFileReader): # pylint: disable=too-many-instance-attributes @@ -76,8 +80,8 @@ def __init__( namespace=None, trim_cells=True, xsd_location: Optional[URI] = None, - xsd_error_code: Optional[str] = None, - xsd_error_message: Optional[str] = None, + ft_error_code: Optional[str] = None, + ft_error_message: Optional[str] = None, rules_location: Optional[URI] = None, **_, ) -> None: @@ -89,8 +93,8 @@ def __init__( null_values=null_values, sanitise_multiline=sanitise_multiline, xsd_location=xsd_location, - xsd_error_code=xsd_error_code, - xsd_error_message=xsd_error_message, + ft_error_code=ft_error_code, + ft_error_message=ft_error_message, rules_location=rules_location, ) diff --git a/src/dve/core_engine/backends/implementations/spark/rules.py b/src/dve/core_engine/backends/implementations/spark/rules.py index 825ee15d..8cdc8a1b 100644 --- a/src/dve/core_engine/backends/implementations/spark/rules.py +++ b/src/dve/core_engine/backends/implementations/spark/rules.py @@ -1,6 +1,7 @@ """Step implementations in Spark.""" -from collections.abc import Callable +# pylint: disable=R0801 +from collections.abc import Callable, Iterable, Iterator from typing import Optional from uuid import uuid4 @@ -12,9 +13,11 @@ from dve.core_engine.backends.exceptions import ConstraintError from dve.core_engine.backends.implementations.spark.spark_helpers import ( create_udf, + df_is_empty, get_all_registered_udfs, object_to_spark_literal, spark_filter_contract_errors, + spark_get_entity_count, spark_read_parquet, spark_record_index, spark_write_parquet, @@ -34,6 +37,7 @@ ColumnAddition, ColumnRemoval, ConfirmJoinHasMatch, + GroupIdentification, HeaderJoin, ImmediateFilter, InnerJoin, @@ -45,12 +49,14 @@ SemiJoin, TableUnion, ) +from dve.core_engine.constants import RECORD_INDEX_COLUMN_NAME from dve.core_engine.functions import implementations as functions from dve.core_engine.message import FeedbackMessage from dve.core_engine.templating import template_object -from dve.core_engine.type_hints import Messages +from dve.core_engine.type_hints import EntityName, Messages +@spark_get_entity_count @spark_record_index @spark_write_parquet @spark_read_parquet @@ -337,41 +343,107 @@ def union(self, entities: SparkEntities, *, config: TableUnion) -> Messages: return [] def identify_orphans( - self, entities: SparkEntities, *, config: OrphanIdentification - ) -> Messages: + self, + entities: SparkEntities, + *, + config: OrphanIdentification, + ) -> Iterable: + """Identify records in an entity which don't have at least one corresponding + match in the target. A new boolean column will be added to `entity` ('IsOrphaned') + indicating whether the condition matched. + + If there is already an 'IsOrphaned' column in the entity, this will be set to the + logical OR of its current value and the value it would have been set to otherwise. + + """ source_df: DataFrame = entities[config.entity_name] source_df = source_df.alias(config.entity_name) - target_df: DataFrame = entities[config.target_name] - target_df = target_df.alias(config.target_name) - key_name = f"key_{uuid4().hex}" - source_df = source_df.withColumn(key_name, sf.expr("uuid()")).alias(config.entity_name) + if df_is_empty(source_df): + self.logger.info(f"{config.entity_name} is empty. Skipping orphan check.") + return + + target_df: DataFrame = entities[config.target_name] match_name = f"matched_{uuid4().hex}" - target_df = target_df.withColumn(match_name, lit(1)).alias(config.target_name) + target_df = target_df.select("*", sf.lit(1).alias(match_name)).alias(config.target_name) - joined_df = ( + orphaned_df: DataFrame = ( source_df.join(target_df, on=sf.expr(config.join_condition), how="left") - .groupBy(col(key_name)) - .agg(sf.coalesce(sf.sum(col(match_name)) == lit(0), lit(True)).alias("IsOrphaned")) + .groupBy(f"{config.entity_name}.{config.id}") + .agg( + sf.first(f"{config.entity_name}.{RECORD_INDEX_COLUMN_NAME}").alias( + RECORD_INDEX_COLUMN_NAME + ), # pylint: disable=C0301 + (sf.coalesce(sf.count(match_name), sf.lit(0)) == sf.lit(0)).alias("IsOrphaned"), + ) + .filter(sf.col("IsOrphaned")) + .select(RECORD_INDEX_COLUMN_NAME) + .alias("orphan") ) - if "IsOrphaned" not in source_df.columns: - result = source_df.join(joined_df, on=[key_name], how="left").drop(key_name) - else: - result = source_df.alias("source").join( - joined_df.alias("joined"), - on=col(f"source.{key_name}") == col(f"joined.{key_name}"), - how="left", + message_df = ( + entities[config.entity_name] + .alias(config.entity_name) + .join( + orphaned_df, + sf.expr( + f"{config.entity_name}.{RECORD_INDEX_COLUMN_NAME} = orphan.{RECORD_INDEX_COLUMN_NAME}", # pylint: disable=C0301 + ), + "semi", ) + ) + filtered_rel = ( + entities[config.entity_name] + .alias(config.entity_name) + .join( + orphaned_df, + sf.expr( + f"{config.entity_name}.{RECORD_INDEX_COLUMN_NAME} = orphan.{RECORD_INDEX_COLUMN_NAME}", # pylint: disable=C0301 + ), + "anti", + ) + ) - columns = {name: col(f"source.{name}") for name in source_df.columns} - columns["IsOrphaned"] = col("source.IsOrphaned") | col("joined.IsOrphaned") - columns.pop(key_name, None) + entities[config.entity_name] = filtered_rel - result = result.select(*[column.alias(name) for name, column in columns.items()]) + for r in message_df.toLocalIterator(): + yield r.asDict() - entities[config.new_entity_name or config.entity_name] = result - return [] + def check_mandatory_group( + self, entities: SparkEntities, *, config: GroupIdentification + ) -> Iterator: + """ + Check that a mandatory key in an entity has at least one valid entry in the all the + child entities. + """ + source_df: DataFrame = entities[config.entity_name] + source_df = source_df.alias(config.entity_name) + target_df: DataFrame = entities[config.target_name] + target_df = target_df.alias(config.target_name) + + source_columns = [f"{config.entity_name}.{c.strip()}" for c in source_df.columns] + _pk, fk = config.join_condition.split("=") + + joined_df = source_df.join(target_df, sf.expr(config.join_condition), "left").select( + *source_columns, + sf.col(fk.strip()).alias("fk"), + ) + + missing_children_df = joined_df.filter("fk IS NULL") + filtered_df = joined_df.filter("fk IS NOT NULL").select("*").drop(sf.col("fk")) + + if not df_is_empty(missing_children_df): + _no_valid_children = missing_children_df.count() + else: + _no_valid_children = 0 + self.logger.info( + f"Found {_no_valid_children} records with no valid children in {config.entity_name}." + ) + + entities[config.entity_name] = filtered_df + + for r in missing_children_df.toLocalIterator(): + yield r.asDict() def filter(self, entities: SparkEntities, *, config: ImmediateFilter) -> Messages: """Filter an entity immediately, and do not emit any messages. @@ -393,6 +465,13 @@ def notify(self, entities: SparkEntities, *, config: Notification) -> Messages: messages: Messages = [] entity = entities[config.entity_name] + if config.error_if_expression_null: + if self.get_entity_count(entity.filter(f"({config.expression}) IS NULL")) > 0: + raise ValueError( + f"The filter evaluated for error code {config.reporting.code}" + + f" in entity {config.entity_name} produced some NULL results. Please investigate." # pylint: disable=C0301 + ) + matched = entity.filter(config.expression) if config.excluded_columns: matched = matched.drop(*config.excluded_columns) @@ -419,3 +498,26 @@ def notify(self, entities: SparkEntities, *, config: Notification) -> Messages: ) ) return messages + + def cache_entity(self, entity_name: str, entities: SparkEntities): + """Store the materialised query in memory and update entity to query directly. + If the entity is already cached, the new cache should be created first, then the old one + removed as part of the function (in case the newer cache depends on the older one).""" + if entity_name not in entities: + return + + _tmp_name = f"{entity_name}_{uuid4().hex}" + + entity = entities[entity_name] + entity.createOrReplaceTempView(_tmp_name) + self.spark_session.sql(f"CACHE TABLE {_tmp_name}") + self.spark_session.sql(f"SELECT count(*) FROM {_tmp_name}") + entity = self.spark_session.table(_tmp_name) + self._remove_cached_artifact(entity_name) + self.entity_cache_tracker[entity_name] = _tmp_name + + entities[entity_name] = entity + + def _remove_cached_artifact(self, entity_name: EntityName): + if _tbl := self.entity_cache_tracker.pop(entity_name, None): + self.spark_session.sql(f"DROP TABLE IF EXISTS {_tbl}") diff --git a/src/dve/core_engine/backends/implementations/spark/spark_helpers.py b/src/dve/core_engine/backends/implementations/spark/spark_helpers.py index 8c14132b..1257306e 100644 --- a/src/dve/core_engine/backends/implementations/spark/spark_helpers.py +++ b/src/dve/core_engine/backends/implementations/spark/spark_helpers.py @@ -415,14 +415,14 @@ def _spark_filter_contract_errors( st.StructField("RecordIndex", st.IntegerType()), st.StructField("FailureType", st.StringType()), st.StructField("Status", st.StringType()), - st.StructField("Entity", st.StringType()), + st.StructField("OriginalEntity", st.StringType()), ] ), ) .filter( (sf.col("FailureType") == sf.lit("record")) & (sf.col("Status") != sf.lit("informational")) - & (sf.col("Entity") == sf.lit(entity_name)) + & (sf.col("OriginalEntity") == sf.lit(entity_name)) ) .distinct() .orderBy(sf.asc(sf.col("RecordIndex"))) @@ -456,6 +456,16 @@ def spark_get_entity_count(cls): return cls +def _spark_check_entity_empty(self, entity: DataFrame) -> bool: # pylint: disable=W0613 + return entity.count() == 0 + + +def spark_check_entity_empty(cls): + """Class decorator to check whether a supplied entity is empty""" + cls.check_entity_empty = _spark_check_entity_empty + return cls + + def get_all_registered_udfs(spark: SparkSession) -> set[str]: """Function to supply the names of a registered functions stored in the supplied spark session. diff --git a/src/dve/core_engine/backends/metadata/contract.py b/src/dve/core_engine/backends/metadata/contract.py index e3eb0c09..2e7565d9 100644 --- a/src/dve/core_engine/backends/metadata/contract.py +++ b/src/dve/core_engine/backends/metadata/contract.py @@ -35,7 +35,7 @@ class DataContractMetadata(BaseModel, frozen=True, arbitrary_types_allowed=True) reporting_fields: dict[EntityName, ReportingFields] """The per-entity reporting fields.""" cache_originals: bool = False - """Whether to cache the original entities after loading.""" + """WARNING - Depreciated functionality. Whether to cache the original entities after loading.""" _schemas: dict[EntityName, type[BaseModel]] = PrivateAttr(default_factory=dict) """The pydantic models of the schmas.""" diff --git a/src/dve/core_engine/backends/metadata/rules.py b/src/dve/core_engine/backends/metadata/rules.py index f3a6305a..a1db7cc8 100644 --- a/src/dve/core_engine/backends/metadata/rules.py +++ b/src/dve/core_engine/backends/metadata/rules.py @@ -33,6 +33,7 @@ "CopyEntity", "DeferredFilter", "EntityRemoval", + "GroupIdentification", "HeaderJoin", "ImmediateFilter", "InnerJoin", @@ -40,6 +41,7 @@ "OneToOneJoin", "OneToOneJoin", "OrphanIdentification", + "OrphanRemoval", "ParentMetadata", "RenameEntity", "Rule", @@ -280,6 +282,8 @@ class Notification(AbstractStep): """Columns to be excluded from the record in the report.""" reporting: ReportingConfig """The reporting information for the filter.""" + error_if_expression_null: bool = False + """Raise error if the results of evaluating the expression passed leads to some NULL results""" def get_required_entities(self) -> set[EntityName]: return {self.entity_name} @@ -558,6 +562,17 @@ class OrphanIdentification(AbstractConditionalJoin): """A step within a rule. This is either a rule config or the literal string 'sync'.""" +class OrphanRemoval(BaseStep): + """Remove an orphan record from the `entity`.""" + + reporting: ReportingConfig + """The reporting information for the row removal.""" + + +class GroupIdentification(AbstractConditionalJoin): + """Identify mandatory records which do not have any valid child records""" + + class Rule(BaseModel): """A rule, made up of multiple steps.""" diff --git a/src/dve/core_engine/backends/readers/csv.py b/src/dve/core_engine/backends/readers/csv.py index cfa2dcde..619a3d19 100644 --- a/src/dve/core_engine/backends/readers/csv.py +++ b/src/dve/core_engine/backends/readers/csv.py @@ -42,8 +42,8 @@ def __init__( null_values: Collection[str] = frozenset({"NULL", "null", ""}), encoding: str = "utf-8-sig", field_check: bool = False, - field_check_error_code: str = "CSVFieldMismatch", - field_check_error_message: str = "The submitted header is invalid", + ft_error_code: Optional[str] = "MalformedCSVFile", + ft_error_message: Optional[str] = None, **_, ): """Init function for the base CSV reader. @@ -89,10 +89,8 @@ def __init__( """Encoding of the CSV file.""" self.field_check = field_check """Whether to check the fields are correct in the supplied header or not""" - self.field_check_error_code = field_check_error_code - """Error code to raise when fields are missing or unexpected""" - self.field_check_error_message = field_check_error_message - """Error message to raise when fields are missing or unexpected""" + self.ft_error_code = ft_error_code + self.ft_error_message = ft_error_message def _get_reader_args(self) -> dict[str, Any]: reader_args: dict[str, Any] = { @@ -218,8 +216,8 @@ def perform_field_check( entity_name, expected_schema, all_model_fields, - self.field_check_error_code, - self.field_check_error_message, + self.ft_error_code or "CSVFieldMismatch", + self.ft_error_message or "The submitted header is invalid", self.delimiter, self.quote_char, ) diff --git a/src/dve/core_engine/backends/readers/utilities.py b/src/dve/core_engine/backends/readers/utilities.py index 3948d709..86347dce 100644 --- a/src/dve/core_engine/backends/readers/utilities.py +++ b/src/dve/core_engine/backends/readers/utilities.py @@ -5,7 +5,8 @@ from pydantic import BaseModel -from dve.core_engine.backends.exceptions import MessageBearingError +from dve.core_engine.backends.exceptions import CriticalMessageBearingError +from dve.core_engine.constants import PRE_VALIDATION_ENTITY from dve.core_engine.message import FeedbackMessage from dve.core_engine.type_hints import URI, EntityName from dve.parser.file_handling.service import open_stream @@ -55,19 +56,17 @@ def raise_message_bearing_error_on_header_differences( record_details_additional = ( f"additional fields: {', '.join(sorted(additional))};" if additional else "" ) # pylint: disable=C0301 - raise MessageBearingError( + raise CriticalMessageBearingError( "The CSV header doesn't match what is expected", - messages=[ - FeedbackMessage( - entity="Pre-validation", - record={entity_name: f"{record_details_missing}{record_details_additional}"}, - failure_type="submission", - error_location=entity_name, - reporting_field="csv_header", - error_code=field_check_error_code, - error_message=field_check_error_message, - ) - ], + message=FeedbackMessage( + entity=PRE_VALIDATION_ENTITY, + record={entity_name: f"{record_details_missing}{record_details_additional}"}, + failure_type="submission", + error_location=entity_name, + reporting_field="csv_header", + error_code=field_check_error_code, + error_message=field_check_error_message, + ), ) diff --git a/src/dve/core_engine/backends/readers/xml.py b/src/dve/core_engine/backends/readers/xml.py index badc03a0..05167e18 100644 --- a/src/dve/core_engine/backends/readers/xml.py +++ b/src/dve/core_engine/backends/readers/xml.py @@ -132,8 +132,8 @@ def __init__( encoding: str = "utf-8-sig", n_records_to_read: Optional[int] = None, xsd_location: Optional[URI] = None, - xsd_error_code: Optional[str] = None, - xsd_error_message: Optional[str] = None, + ft_error_code: Optional[str] = None, + ft_error_message: Optional[str] = None, rules_location: Optional[URI] = None, **_, ): @@ -174,10 +174,8 @@ def __init__( else: self.xsd_location = xsd_location # type: ignore """The URI of the xsd file if wishing to perform xsd validation.""" - self.xsd_error_code = xsd_error_code - """The error code to be reported if xsd validation fails (if xsd)""" - self.xsd_error_message = xsd_error_message - """The error message to be reported if xsd validation fails""" + self.ft_error_code = ft_error_code or "MalformedXMLFile" + self.ft_error_message = ft_error_message super().__init__() self._logger = get_logger(__name__) @@ -255,6 +253,8 @@ def _get_elements_from_stream(self, stream: IO[bytes]) -> Iterator[XMLElement]: remove_comments=True, dtd_validation=False, resolve_entities=False, + no_network=True, + load_dtd=False, ) tree: etree._ElementTree = etree.parse(stream, parser) @@ -294,15 +294,11 @@ def _run_xmllint(self, file_uri: URI) -> FeedbackMessage | None: onto the system to run succesfully.""" if self.xsd_location is None: raise AttributeError("Trying to run XML lint with no `xsd_location` provided.") - if self.xsd_error_code is None: - raise AttributeError("Trying to run XML with no `xsd_error_code` provided.") - if self.xsd_error_message is None: - raise AttributeError("Trying to run XML with no `xsd_error_message` provided.") return run_xmllint( file_uri=file_uri, schema_uri=self.xsd_location, - error_code=self.xsd_error_code, - error_message=self.xsd_error_message, + error_code=self.ft_error_code or "XMLFailedXSDCheck", + error_message=self.ft_error_message or "XML Submission has failed XSD check", ) def read_to_py_iterator( @@ -375,6 +371,8 @@ def _get_elements_from_stream(self, stream: IO[bytes]) -> Iterator[XMLElement]: remove_comments=True, dtd_validation=False, resolve_entities=False, + no_network=True, + load_dtd=False, ) container_contexts = 1 if not self.root_tag else 0 diff --git a/src/dve/core_engine/backends/readers/xml_linting.py b/src/dve/core_engine/backends/readers/xml_linting.py index 529d8ee8..910bb65c 100644 --- a/src/dve/core_engine/backends/readers/xml_linting.py +++ b/src/dve/core_engine/backends/readers/xml_linting.py @@ -10,6 +10,7 @@ from typing import Union from uuid import uuid4 +from dve.core_engine.constants import PRE_VALIDATION_ENTITY from dve.core_engine.message import FeedbackMessage from dve.parser.file_handling import copy_resource, get_file_name, get_resource_exists, open_stream from dve.parser.file_handling.implementations.file import file_uri_to_local_path @@ -131,12 +132,12 @@ def run_xmllint( return None return FeedbackMessage( - entity="xsd_validation", + entity=PRE_VALIDATION_ENTITY, record={}, failure_type="submission", is_informational=False, error_type="xsd check", - error_location="Whole File", + error_location="XSD Validation", error_message=error_message, error_code=error_code, ) diff --git a/src/dve/core_engine/configuration/v1/__init__.py b/src/dve/core_engine/configuration/v1/__init__.py index 959596fc..634cc112 100644 --- a/src/dve/core_engine/configuration/v1/__init__.py +++ b/src/dve/core_engine/configuration/v1/__init__.py @@ -1,9 +1,10 @@ """The loader for the first JSON-based dataset configuration.""" import json -from typing import Any, Optional, Union +from typing import Any, Optional, Type, Union -from pydantic import BaseModel, Field, PrivateAttr, validate_call +from pydantic import BaseModel, Field, PrivateAttr, field_validator, model_validator, validate_call +from pydantic_core.core_schema import FieldValidationInfo from typing_extensions import Literal from dve.core_engine.backends.base.reference_data import ReferenceConfig, ReferenceConfigUnion @@ -22,7 +23,14 @@ ) from dve.core_engine.configuration.v1.steps import StepConfigUnion from dve.core_engine.message import DataContractErrorDetail -from dve.core_engine.type_hints import EntityName, ErrorCategory, ErrorType, TemplateVariables +from dve.core_engine.type_hints import ( + EntityName, + ErrorCategory, + ErrorCode, + ErrorMessage, + ErrorType, + TemplateVariables, +) from dve.core_engine.validation import RowValidator from dve.parser.file_handling import joinuri, open_stream, resolve_location from dve.parser.type_hints import URI, Extension @@ -38,6 +46,8 @@ FieldName = str """The name of a field within a model/schema.""" +JoinFields = Optional[dict[str, str]] +"""The fields required ( parent > child ) to join a child entity back to the parent""" TypeOrDef = Union[ # pylint: disable=C0103 TypeName, "_CallableTypeDefinition", "_ModelTypeDefinition", "_TypeAliasDefinition" ] @@ -47,6 +57,8 @@ """The operation """ RuleType = type[AbstractStep] """The metadata step type implemented by the rule.""" +AllowedAdditionalReaderChecks = Literal["check_empty"] +"""Additional checks to be performed in the file_transformation stage""" class _BaseTypeDefintion(BaseModel): @@ -81,6 +93,67 @@ class _TypeAliasDefinition(_BaseTypeDefintion): """The name of the Python type.""" +class _LinkageConfig(BaseModel): + """Specify how to link entities back to parents if required""" + + parent_entity: Optional[EntityName] = None + """The name of the parent entity""" + join_fields: JoinFields = Field(default_factory=dict) + """The fields that can be used to link back to the parent entity""" + is_root_entity: bool = False + """Whether the entity is the highest level parent in a tree""" + mandatory: bool = False + """If the entity is a child, is it a mandatory field of the parent""" + no_valid_records_error_code: Optional[ErrorCode] = "NoValidRecords" + """The error code to emit if the entity has no valid records and is mandatory in the parent entity""" # pylint: disable=C0301 + no_valid_records_error_message: Optional[ErrorMessage] = ( + "parent record removed as no valid child records" + ) + """The error message to emit if the entity has no valid records and is mandatory in the parent entity""" # pylint: disable=C0301 + missing_parent_id_error_code: Optional[ErrorCode] = "MissingParentRecord" + """The error code to emit if the entity contains records that are orphaned by parent record rejections""" # pylint: disable=C0301 + missing_parent_id_error_message: Optional[ErrorMessage] = ( + "Records removed due to no valid parent record" + ) + """The error code to emit if the entity contains records that are orphaned by parent record rejections""" # pylint: disable=C0301 + empty_entity_error_code: ErrorCode = "EmptyEntity" + """The error code to emit if a mandatory entity has no valid remaining records""" + empty_entity_error_message: ErrorMessage = "no valid records remaining" + """The error message to emit if a mandatory entity has no valid remaining records""" + + @model_validator(mode="after") + def _check_root_no_parent_or_join_keys(self): + if self.is_root_entity and (self.parent_entity or self.join_fields): + raise ValueError( + "If entity is root, neither parent_entity nor join keys should be specified" + ) + return self + + @model_validator(mode="after") + def _check_non_root_entities_have_a_defined_parent(self): + """Check that non root entities have a parent defined.""" + if not self.is_root_entity and self.parent_entity is None: + raise ValueError( + "Non-root entity has no defined parent entity. If you intend this to be a root " + 'entity you must specify `"is_root_entity": true` for the entity. ' + 'Otherwise you must specify a `"parent_entity": ""` for this entity.' + ) + return self + + @model_validator(mode="after") + def _check_root_mandatory(self): + if self.is_root_entity and not self.mandatory: + raise ValueError("If entity is root, it must be labelled mandatory") + return self + + @model_validator(mode="after") + def _check_parent_entity_with_join_keys(self): + if self.parent_entity or self.join_fields: + if not (self.parent_entity and self.join_fields): + raise ValueError("Both parent_entity and join_fields must be supplied if one is") + return self + + class _SchemaConfig(BaseModel): """Configuration for a component schema within a dataset.""" @@ -90,6 +163,11 @@ class _SchemaConfig(BaseModel): """A list of the field names within the schema which _must_ be provided.""" +class _ReaderAdditionalChecksConfig(BaseModel): + error_code: str + error_message: str + + class _ReaderConfig(BaseModel): # type: ignore """Reader configuration options for a model.""" @@ -110,6 +188,10 @@ class _ModelConfig(_SchemaConfig): """A single key field to be used by the model.""" reader_config: dict[Extension, _ReaderConfig] """Reader configuration options for the model.""" + reader_additional_checks: dict[AllowedAdditionalReaderChecks, _ReaderAdditionalChecksConfig] = ( + Field(default_factory=dict) + ) + """Additional checks to be performed after the entity is read""" aliases: dict[FieldName, FieldName] = Field(default_factory=dict) """An alias field name mapping.""" @@ -136,7 +218,7 @@ class V1DataContractConfig(BaseModel): """Configuration for the data contract component of the dataset.""" cache_originals: bool = False - """Whether to cache the original entities after loading.""" + """WARNING - Depreciated functionality. Whether to cache the original entities after loading.""" error_details: Optional[URI] = None """Optional URI containing custom data contract error codes and messages""" types: dict[TypeName, TypeOrDef] = Field(default_factory=dict) @@ -177,6 +259,8 @@ class V1EngineConfig(BaseEngineConfig): default_factory=dict ) """Rule store rules from the loaded rule stores.""" + entity_relationships: dict[EntityName, _LinkageConfig] = Field(default_factory=dict) + """The parent-child relationships linking the defined entities""" @validate_call def _update_rule_store(self, rule_store: dict[RuleName, BusinessComponentSpecConfigUnion]): @@ -322,14 +406,13 @@ def get_contract_metadata(self) -> DataContractMetadata: } reporting_fields[entity_name] = dataset_config.reporting_fields validators[entity_name] = RowValidator( - contract_dict, entity_name, error_info=error_info + contract_dict, entity_name, error_info=error_info.get(entity_name) ) return DataContractMetadata( reader_metadata=reader_metadata, validators=validators, reporting_fields=reporting_fields, - cache_originals=self.contract.cache_originals, ) def load_error_message_info(self, uri): diff --git a/src/dve/core_engine/configuration/v1/hierarchy.py b/src/dve/core_engine/configuration/v1/hierarchy.py new file mode 100644 index 00000000..35997eb1 --- /dev/null +++ b/src/dve/core_engine/configuration/v1/hierarchy.py @@ -0,0 +1,210 @@ +"""Classes to help determine and store entity hierarchy information.""" + +import json +from typing import Any, Iterable, Optional, Union + +from pydantic import BaseModel, Field, model_validator + +from dve.core_engine.configuration.v1 import V1EngineConfig, _LinkageConfig +from dve.core_engine.type_hints import EntityName, ErrorCode, ErrorMessage +from dve.metadata_parser.exc import EntityNotFoundError +from dve.parser.file_handling.service import open_stream +from dve.parser.type_hints import URI + + +class HierarchyNode(BaseModel): + """Stores entity hierarchy information""" + + entity_name: str + parent_entity: Optional[str] = None + children: list["HierarchyNode"] = Field(default_factory=list) + mandatory: bool = False + join_fields: dict[str, str] = Field(default_factory=dict) + no_valid_records_error_code: ErrorCode = "NoValidRecords" + no_valid_records_error_message: ErrorMessage = "parent record removed as no valid child records" + missing_parent_id_error_code: Optional[ErrorCode] = "MissingParentRecord" + missing_parent_id_error_message: Optional[ErrorMessage] = ( + "Records removed due to no valid parent record" + ) + empty_entity_error_code: ErrorCode = "EmptyEntity" + empty_entity_error_message: ErrorMessage = "no valid records remaining" + + @model_validator(mode="after") + def validate_empty_error_details(self): + """ + Removes the default messaging for checking empty entities as not performed on + non mandatory nodes/entities + """ + if not self.mandatory: + self.empty_entity_error_code = None + self.empty_entity_error_message = None + return self + + def get_descendents(self) -> list["HierarchyNode"]: + """Recursively list all descendents of the node""" + descendents = [] + for node in self.children: # type: ignore + descendents.append(node) + descendents.extend(node.get_descendents()) + return descendents + + def get_descendent_names(self) -> list[str]: + """Recursively list all names of descendents of the node""" + return [node.entity_name for node in self.get_descendents()] + + def get_node(self, entity_name: str) -> Union["HierarchyNode", None]: + """Recursively search for node and return if found""" + node = None + if self.entity_name == entity_name: + return self + for child in self.children: # type: ignore + node = child.get_node(entity_name) + if node: + break + return node + + def add_child_node(self, parent_entity: str, child_info: "HierarchyNode") -> None: + """Add a child node if the parent exists in the hierarchy""" + try: + self.get_node(parent_entity).children.append(child_info) # type: ignore + except AttributeError as exc: + raise EntityNotFoundError( + f"Can't find parent node {parent_entity} in {self.entity_name}" + ) from exc + + def as_dict(self) -> dict[str, dict[str, Any]]: + """Get dictionary representation of entity hierarchy""" + child_dict: dict[str, dict[str, Any]] = {} + for node in self.children: # type: ignore + child_dict.update(node.as_dict()) + + ret_dict = self.model_dump(exclude={"entity_name", "children"}) + ret_dict.update({"children": child_dict}) + + return {self.entity_name: ret_dict} + + def _get_full_tree(self): + """Get all nodes in tree, including the root""" + desc = self.get_descendents() + desc.insert(0, self) + return desc + + def iterate_root_down(self): + """Iterate through nodes from root to lowest descendent""" + yield from self._get_full_tree() + + def iterate_lowest_descendent_up(self): + """Iterate through nodes from lowest descendent to root""" + yield from self._get_full_tree()[::-1] + + +class EntityHierarchy: + """Determines and stores entity hierarchy information from config""" + + def __init__(self, entity_trees: dict[EntityName, HierarchyNode]): + self.entity_trees = entity_trees + + @staticmethod + def determine_trees( + all_datasets: Iterable[str], entity_relationships: dict[str, _LinkageConfig] + ) -> dict[EntityName, HierarchyNode]: + """Determine the entity hierarchy trees and store as HierarchyNodes""" + root_entities: dict[str, _LinkageConfig] = dict( + filter(lambda x: x[1].is_root_entity, entity_relationships.items()) + ) + top_level_parents: dict[EntityName, HierarchyNode] = { + entity_name: HierarchyNode( + entity_name=entity_name, + parent_entity=None, + **config.model_dump( + exclude={ + "parent_entity", + "missing_parent_id_error_code", + "missing_parent_id_error_message", + } + ), + missing_parent_id_error_code=None, + missing_parent_id_error_message=None, + ) + for entity_name, config in root_entities.items() + } + + if default_roots := [ + entity_name for entity_name in all_datasets if entity_name not in entity_relationships + ]: + for entity_name in default_roots: + top_level_parents[entity_name] = HierarchyNode( + entity_name=entity_name, + parent_entity=None, + missing_parent_id_error_code=None, + missing_parent_id_error_message=None, + ) + + for name, linkage_detail in entity_relationships.items(): + for main_entity, parent_node in top_level_parents.items(): + if linkage_detail.is_root_entity: + break + + if ( + linkage_detail.parent_entity == main_entity + or linkage_detail.parent_entity in parent_node.get_descendent_names() + ): + parent_node.add_child_node( + linkage_detail.parent_entity, # type: ignore + HierarchyNode(entity_name=name, **linkage_detail.model_dump()), + ) + break + else: + raise EntityNotFoundError( + f"Can't find parent entity {linkage_detail.parent_entity} defined to " + + f"establish hierarchy for {name} - please ensure it is defined above " + + "any child entities in the dischema." + ) + return top_level_parents + + @classmethod + def from_dischema(cls, dischema_uri: URI): + """Create entity hierarchy direct from dischema""" + with open_stream(dischema_uri) as dischema: + config_dict = json.load(dischema) + all_datasets = config_dict.get("contract", {}).get("datasets", {}).keys() + entity_relationships = { + k: _LinkageConfig(**v) for k, v in config_dict.get("entity_relationships", {}).items() + } + return cls(entity_trees=cls.determine_trees(all_datasets, entity_relationships)) + + @classmethod + def from_engine_config(cls, engine_config: V1EngineConfig): + """Create entity hierarchy direct from engine config""" + return cls( + entity_trees=cls.determine_trees( + all_datasets=engine_config.contract.datasets.keys(), + entity_relationships=engine_config.entity_relationships, + ) + ) + + def get_all_mandatory_nodes( + self, + node: Optional[HierarchyNode] = None, + mandatory_nodes: Optional[list[HierarchyNode]] = None, + nodes_visited: Optional[set[EntityName]] = None, + ) -> list[HierarchyNode]: + """Find and return all mandatory nodes""" + if mandatory_nodes is None: + mandatory_nodes = [] + + if nodes_visited is None: + nodes_visited = set() + + if node is None: + for _node in self.entity_trees.values(): + self.get_all_mandatory_nodes(_node, mandatory_nodes, nodes_visited) + + if node: + if node.mandatory and node.entity_name not in nodes_visited: + nodes_visited.add(node.entity_name) + mandatory_nodes.append(node) + for child_node in node.children: + self.get_all_mandatory_nodes(child_node, mandatory_nodes, nodes_visited) + + return mandatory_nodes diff --git a/src/dve/core_engine/constants.py b/src/dve/core_engine/constants.py index a2a4a655..9b67e471 100644 --- a/src/dve/core_engine/constants.py +++ b/src/dve/core_engine/constants.py @@ -6,3 +6,8 @@ CONTRACT_ERROR_VALUE_FIELD_NAME: str = "__error_value" """The name of the field that can be used to extract the field value that caused a pydantic validation error""" + +PRE_VALIDATION_ENTITY: str = "Pre-validation" +""" +Consistent name for the entity/group where errors are raised during file transformation +""" diff --git a/src/dve/core_engine/message.py b/src/dve/core_engine/message.py index 78024e97..05dbc174 100644 --- a/src/dve/core_engine/message.py +++ b/src/dve/core_engine/message.py @@ -90,7 +90,7 @@ def extract_error_value(records, error_location): class UserMessage: """The structure of the message that is used to populate the error report.""" - Entity: Optional[str] + ReportingEntity: Optional[str] """The entity that the message pertains to (if applicable).""" Key: Optional[str] "The key field(s) in string format to allow users to identify the record" @@ -176,7 +176,8 @@ class FeedbackMessage: # pylint: disable=too-many-instance-attributes """The category of the error.""" HEADER: ClassVar[list[str]] = [ - "Entity", + "ReportingEntity", + "OriginalEntity", "Key", "FailureType", "Status", @@ -307,6 +308,7 @@ def to_row( return ( self.entity, + self.original_entity, key, self.failure_type, "informational" if self.is_informational else "error", diff --git a/src/dve/core_engine/models.py b/src/dve/core_engine/models.py index bba2986a..49dba230 100644 --- a/src/dve/core_engine/models.py +++ b/src/dve/core_engine/models.py @@ -82,7 +82,9 @@ def _ensure_just_file_stem( @property def file_name_with_ext(self): """Return file name with extension.""" - return f"{self.file_name}.{self.file_extension}" + if self.file_extension: + return f"{self.file_name}.{self.file_extension}" + return self.file_name @classmethod def from_metadata_file(cls, submission_id: str, metadata_uri: Location): diff --git a/src/dve/core_engine/type_hints.py b/src/dve/core_engine/type_hints.py index 154ada65..50781ec5 100644 --- a/src/dve/core_engine/type_hints.py +++ b/src/dve/core_engine/type_hints.py @@ -133,12 +133,21 @@ """A string indicating the field that the error pertains to.""" FieldValue = Optional[Any] """The value that caused the error.""" -ErrorCategory = Literal["Blank", "Wrong format", "Bad value", "Bad file"] +ErrorCategory = Literal[ + "Blank", + "Wrong format", + "Bad value", + "Bad file", + "Parent Missing", + "Children missing", + "Empty entity", +] """A string indicating the category of the error.""" RecordIndex = Optional[int] """The record index that the error relates to (if applicable)""" MessageTuple = tuple[ + Optional[EntityName], Optional[EntityName], Key, FailureType, diff --git a/src/dve/metadata_parser/models.py b/src/dve/metadata_parser/models.py index 49e2386f..0b7d5359 100644 --- a/src/dve/metadata_parser/models.py +++ b/src/dve/metadata_parser/models.py @@ -391,6 +391,7 @@ class DatasetSpecification(BaseModel): """Configuration options for a dataset.""" cache_originals: bool = False + """WARNING - Depreciated functionality.""" types: dict[TypeName, FieldSpecification] = Field(default_factory=dict) """Predefined types to be used within schema/dataset definitions.""" schemas: dict[EntityName, EntitySpecification] = Field(default_factory=dict) diff --git a/src/dve/parser/file_handling/service.py b/src/dve/parser/file_handling/service.py index 9ee9d9fd..fbdc8ab4 100644 --- a/src/dve/parser/file_handling/service.py +++ b/src/dve/parser/file_handling/service.py @@ -273,9 +273,12 @@ def copy_resource(source_uri: URI, target_uri: URI, overwrite: bool = False) -> _transfer_resource(source_uri, target_uri, overwrite, "copy") -def move_resource(source_uri: URI, target_uri: URI, overwrite: bool = False) -> None: - """Move a resource from one location to another.""" +def move_resource(source_uri: URI, target_uri: URI, overwrite: bool = False) -> URI: + """ + Move a resource from one location to another. Returns the target_uri. + """ _transfer_resource(source_uri, target_uri, overwrite, "move") + return target_uri def create_directory(target_uri: URI): diff --git a/src/dve/pipeline/pipeline.py b/src/dve/pipeline/pipeline.py index a9be3ff9..2cf00553 100644 --- a/src/dve/pipeline/pipeline.py +++ b/src/dve/pipeline/pipeline.py @@ -1,4 +1,4 @@ -# pylint: disable=protected-access,too-many-instance-attributes,too-many-arguments,line-too-long +# pylint: disable=protected-access,too-many-instance-attributes,too-many-arguments,line-too-long,too-many-lines """Generic Pipeline object to define how DVE should be interacted with.""" import json @@ -18,6 +18,7 @@ import dve.reporting.excel_report as er from dve.common.error_utils import ( + BackgroundMessageWriter, dump_feedback_errors, dump_processing_errors, get_feedback_errors_uri, @@ -29,11 +30,12 @@ from dve.core_engine.backends.base.core import EntityManager from dve.core_engine.backends.base.reference_data import BaseRefDataLoader, ReferenceConfig from dve.core_engine.backends.base.rules import BaseStepImplementations -from dve.core_engine.backends.exceptions import MessageBearingError +from dve.core_engine.backends.exceptions import CriticalMessageBearingError, MessageBearingError from dve.core_engine.backends.readers import BaseFileReader from dve.core_engine.backends.readers.utilities import get_all_model_fields from dve.core_engine.backends.types import EntityType from dve.core_engine.backends.utilities import stringify_model +from dve.core_engine.configuration.v1.hierarchy import EntityHierarchy from dve.core_engine.exceptions import CriticalProcessingError from dve.core_engine.loggers import get_logger from dve.core_engine.message import FeedbackMessage @@ -215,10 +217,10 @@ def write_file_to_parquet( for model_name, model in models.items(): self._logger.info(f"Transforming {model_name} to stringified parquet") - reader: BaseFileReader = load_reader( - dataset, model_name, ext, self.backend_reader_kwargs - ) try: + reader: BaseFileReader = load_reader( + dataset, model_name, ext, self.backend_reader_kwargs + ) if not entity_type: reader.write_parquet( reader.read_to_py_iterator( @@ -237,10 +239,14 @@ def write_file_to_parquet( model_name, stringify_model(model), # type: ignore get_all_model_fields(models.values()), # type: ignore + dataset[model_name].reader_additional_checks, ), f"{out}{model_name}", ) except MessageBearingError as exc: + self._logger.error( + f"While processing {model_name}, an issue was encountered", exc_info=exc + ) errors.extend(exc.messages) return list(dict.fromkeys(errors)) # remove any duplicate errors @@ -341,6 +347,9 @@ def file_transformation( except MessageBearingError as exc: self._logger.exception("Unexpected file transformation error:") errors.extend(exc.messages) + except CriticalMessageBearingError as exc: + self._logger.exception("Unexpected critical file transformation error:") + errors.append(exc.message) if errors: dump_feedback_errors( @@ -543,7 +552,45 @@ def data_contract_step( return processed_files, failed_processing - def apply_business_rules( # pylint: disable=R0914 + def check_mandatory_entities_have_records( + self, + working_directory: URI, + entities: EntityManager, + entity_hierarchy: EntityHierarchy, + key_fields: Optional[dict[str, list[str]]] = None, + ) -> None: + """ + Check that mandatory entities have at least one record post business rules. Otherwise, + raise a submission rejection error message. + """ + with BackgroundMessageWriter( + working_directory=working_directory, + dve_stage="business_rules", + key_fields=key_fields, + logger=self._logger, + ) as msg_writer: + _msgs = [] + for node in entity_hierarchy.get_all_mandatory_nodes(): + entity_name = node.entity_name + if node.mandatory and self.get_entity_count(entities[entity_name]) == 0: + self._logger.info( + f"Found 0 records in mandatory entity {entity_name} after applying all business rules" # pylint: disable=C0301 + ) + _msgs.append( + FeedbackMessage( + entity=entity_name, + record=None, + error_location=entity_name, + error_message=node.empty_entity_error_message, + failure_type="submission", + error_type="submission", + error_code=node.empty_entity_error_code, + category="Empty entity", + ) + ) + msg_writer.write_queue.put(_msgs) + + def apply_business_rules( # pylint: disable=R0914,R0915 self, submission_info: SubmissionInfo, submission_status: Optional[SubmissionStatus] = None ) -> tuple[SubmissionInfo, SubmissionStatus]: """Apply the business rules to a given submission, the submission may have failed at the @@ -583,7 +630,6 @@ def apply_business_rules( # pylint: disable=R0914 entities[file_name] = self.step_implementations.add_record_index( # type: ignore self.step_implementations.read_parquet(parquet_uri) # type: ignore ) - entities[f"Original{file_name}"] = self.step_implementations.read_parquet(parquet_uri) # type: ignore sub_info_entity = ( self._audit_tables._submission_info.conv_to_entity( # pylint: disable=protected-access @@ -596,8 +642,13 @@ def apply_business_rules( # pylint: disable=R0914 key_fields = {model: conf.reporting_fields for model, conf in model_config.items()} + entity_hierarchy = EntityHierarchy.from_engine_config(config) + _errors_uri, rules_success = self.step_implementations.apply_rules( # type: ignore - working_directory, entity_manager, rules, key_fields + working_directory, + entity_manager, + rules, + key_fields, ) rule_messages = load_feedback_messages( @@ -614,21 +665,17 @@ def apply_business_rules( # pylint: disable=R0914 for entity_name, entity in entity_manager.entities.items(): # Note BI filtering done within the apply_rules self._logger.info(f"applying data contract filter to {entity_name}.") - if not entity_name.startswith("Original"): - filtered_entity = self._step_implementations.filter_data_contract_record_rejections( - working_directory, - entity, - entity_name, - ) - else: - self._logger.info(f"Skipping {entity_name}. Marked original.") - filtered_entity = entity + filtered_entity = self._step_implementations.filter_data_contract_record_rejections( + working_directory, + entity, + entity_name, + ) projected = self._step_implementations.write_parquet( # type: ignore filtered_entity, fh.joinuri( self.processed_files_path, submission_info.submission_id, - "business_rules", + "temp_business_rules", entity_name, ), ) @@ -636,10 +683,104 @@ def apply_business_rules( # pylint: disable=R0914 projected ) + _, orph_issues_1 = self.step_implementations.identify_and_remove_orphans( # type: ignore + working_directory, + entity_manager.entities, + entity_hierarchy, + key_fields, + ) + + _, grp_issues_1 = self.step_implementations.identify_and_remove_missing_mandatory_groups( # type: ignore + working_directory, + entity_manager.entities, + entity_hierarchy, + key_fields, + ) + + # Perform a second time incase the mandatory groups result in new orphans + _, orph_issues_2 = self.step_implementations.identify_and_remove_orphans( # type: ignore + working_directory, + entity_manager.entities, + entity_hierarchy, + key_fields, + ) + + entity_issues: dict[EntityName, bool] = { + entity: any( + val + for val in ( + orph_issues_1.get(entity, False), + grp_issues_1.get(entity, False), + orph_issues_2.get(entity, False), + ) + ) + for entity in orph_issues_1.keys() + } + + unchanged_entities: list[EntityName] = [] + for entity_name, entity in entity_manager.entities.items(): + if entity_issues.get(entity_name, False): + self._logger.info(f"Writing {entity_name} out to disk.") + final_projection = self._step_implementations.write_parquet( # type: ignore + entity, + fh.joinuri( + self.processed_files_path, + submission_info.submission_id, + "business_rules", + entity_name, + ), + ) + + entity_manager.entities[entity_name] = self.step_implementations.read_parquet( # type: ignore + final_projection + ) + else: + unchanged_entities.append(entity_name) + + for entity_name in unchanged_entities: + self._logger.info(f"Moving {entity_name} from temp_business_rules to business_rules") + final_projection = fh.move_resource( + source_uri=fh.joinuri( + self.processed_files_path, + submission_info.submission_id, + "temp_business_rules", + entity_name, + ), + target_uri=fh.joinuri( + self.processed_files_path, + submission_info.submission_id, + "business_rules", + entity_name, + ), + overwrite=True, + ) + + entity_manager.entities[entity_name] = self.step_implementations.read_parquet( # type: ignore + final_projection + ) + + self.step_implementations.clear_entity_cache() # type: ignore + + fh.remove_prefix( + fh.joinuri( + self.processed_files_path, submission_info.submission_id, "temp_business_rules" + ), + recursive=True, + ) + + self.check_mandatory_entities_have_records( + working_directory, entity_manager, entity_hierarchy + ) + submission_status.number_of_records = self.get_entity_count( - entity=entity_manager.entities[f"""Original{rules.global_variables.get( - 'entity', - submission_info.dataset_id)}"""] + entity=self.step_implementations.read_parquet( # type: ignore + fh.joinuri( + self.processed_files_path, + submission_info.submission_id, + "data_contract", + rules.global_variables.get("entity", submission_info.dataset_id), + ) + ) ) submission_status.number_of_records_rejected = ( submission_status.number_of_records @@ -776,7 +917,7 @@ def _get_error_dataframes(self, submission_id: str): .alias("error_type") # type: ignore ) df = df.select( - pl.col("Entity").alias("Table"), # type: ignore + pl.col("ReportingEntity").alias("Table"), # type: ignore pl.col("error_type").alias("Type"), # type: ignore pl.col("ErrorCode").alias("Error_Code"), # type: ignore pl.col("ReportingField").alias("Data_Item"), # type: ignore diff --git a/src/dve/pipeline/utils.py b/src/dve/pipeline/utils.py index e6122c23..0b946b11 100644 --- a/src/dve/pipeline/utils.py +++ b/src/dve/pipeline/utils.py @@ -11,9 +11,11 @@ import dve.core_engine.backends.implementations.duckdb # pylint: disable=unused-import import dve.core_engine.backends.implementations.spark # pylint: disable=unused-import import dve.parser.file_handling as fh +from dve.core_engine.backends.exceptions import MessageBearingError from dve.core_engine.backends.readers import _READER_REGISTRY from dve.core_engine.configuration.v1 import SchemaName, V1EngineConfig, _ModelConfig from dve.core_engine.loggers import get_logger +from dve.core_engine.message import FeedbackMessage from dve.core_engine.type_hints import URI, SubmissionResult from dve.metadata_parser.model_generator import JSONtoPyd @@ -52,7 +54,31 @@ def load_reader( backend_reader_kwargs: Optional[dict[str, Any]] = None, ): """Loads the readers for the diven feed, model name and file extension""" - reader_config = dataset[model_name].reader_config[f".{file_extension.lower()}"] + try: + reader_config = dataset[model_name].reader_config[f".{file_extension.lower()}"] + except KeyError as exc: + if file_extension: + err_msg = ( + f"The supplied file extension `{file_extension}`" + + f" is not a supported file format for {model_name}." + ) + else: + err_msg = "No supplied file extension. Unable to parse file without a file extension." + + raise MessageBearingError( + f"The file extension provided ({file_extension}) is not supported for this collection.", + messages=[ + FeedbackMessage( + entity=model_name, + record=None, + failure_type="submission", + error_location="Whole File", + error_code="InvalidFileExtension", + error_message=err_msg, + ) + ], + ) from exc + reader = _READER_REGISTRY[reader_config.reader]( **reader_config.kwargs_, **backend_reader_kwargs if backend_reader_kwargs else {} ) @@ -68,7 +94,7 @@ def unpersist_all_rdds(spark: SparkSession): rdd.unpersist() -def deadletter_file(source_uri: URI) -> None: +def deadletter_file(source_uri: URI) -> URI | None: """Move files that can't be processed to a deadletter location""" try: source_parent: URI = source_uri.rsplit("/", 1)[0] diff --git a/src/dve/reporting/__init__.py b/src/dve/reporting/__init__.py index 9a93c67f..476e0d0d 100644 --- a/src/dve/reporting/__init__.py +++ b/src/dve/reporting/__init__.py @@ -1 +1,3 @@ """Error reports module.""" + +# pylint: disable=R0801 diff --git a/src/dve/reporting/error_report.py b/src/dve/reporting/error_report.py index 95137b51..bf1708ab 100644 --- a/src/dve/reporting/error_report.py +++ b/src/dve/reporting/error_report.py @@ -91,7 +91,7 @@ def create_error_dataframe(errors: deque[FeedbackMessage], key_fields): .alias("error_type") ) df = df.select( # type: ignore - col("Entity").alias("Table"), # type: ignore + col("ReportingEntity").alias("Table"), # type: ignore col("error_type").alias("Type"), # type: ignore col("ErrorCode").alias("Error_Code"), # type: ignore col("ReportingField").alias("Data_Item"), # type: ignore diff --git a/src/dve/reporting/excel_report.py b/src/dve/reporting/excel_report.py index 5876cdd4..e0c48917 100644 --- a/src/dve/reporting/excel_report.py +++ b/src/dve/reporting/excel_report.py @@ -153,17 +153,21 @@ def _add_submission_info(self, status: str, summary: Worksheet): ), # pylint: disable=C0301 ] ) - if status not in ( + if status in ( ErrorReportStatus.PROCESSING_FAILED, ErrorReportStatus.FILE_REJECTION, ): - summary.append( - [ - "", - "Total Number of Records Rejected", - self.submission_status.number_of_records_rejected, - ] - ) + _records_rejected = self.submission_status.number_of_records + else: + _records_rejected = self.submission_status.number_of_records_rejected + summary.append( + [ + "", + "Total Number of Records Rejected", + _records_rejected, + ] + ) + summary.append(["", ""]) diff --git a/tests/features/books.feature b/tests/features/books.feature index 8551a6fc..a36c9e9f 100644 --- a/tests/features/books.feature +++ b/tests/features/books.feature @@ -37,8 +37,9 @@ Feature: Pipeline tests using the books dataset And I add initial audit entries for the submission Then the latest audit record for the submission is marked with processing status file_transformation When I run the file transformation phase - Then the latest audit record for the submission is marked with processing status failed - # TODO - handle above within the stream xml reader - specific + Then the latest audit record for the submission is marked with processing status error_report + When I run the error report phase + Then An error report is produced Scenario: Handle a file that fails XSD validation (duckdb) Given I submit the books file books_xsd_fail.xml for processing diff --git a/tests/features/flights.feature b/tests/features/flights.feature new file mode 100644 index 00000000..3bf2ee06 --- /dev/null +++ b/tests/features/flights.feature @@ -0,0 +1,366 @@ +Feature: Pipeline tests using the flights dataset + Test hierarchical record rejection and ensuring that records are removed correctly including + any "orphan" records generated from record removal in parent entities. + + Scenario: A perfect flights file + Given I submit the flights file perfect_flights.xml for processing + And A duckdb pipeline is configured with schema file 'flights.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And the airport entity is stored as a parquet after the file_transformation phase + And the staff entity is stored as a parquet after the file_transformation phase + And the flights entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the latest audit record for the submission is marked with processing status data_contract + When I run the data contract phase + Then there are no file rejections from the data_contract phase + And there are no record rejections from the data_contract phase + When I run the business rules phase + Then there are no file rejections from the business_rules phase + And there are no record rejections from the business_rules phase + And the final entities have the following row counts + | entity_name | row_count | + | country | 1 | + | airport | 3 | + | staff | 15 | + | flights | 10 | + | passengers | 25 | + When I run the error report phase + Then An error report is produced + And The statistics entry for the submission shows the following information + | parameter | value | + | record_count | 1 | + | number_submission_rejections | 0 | + | number_record_rejections | 0 | + | number_warnings | 0 | + + Scenario: A flights submission where the root record is rejected + Given I submit the flights file missing_country_id.xml for processing + And A duckdb pipeline is configured with schema file 'flights.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And the airport entity is stored as a parquet after the file_transformation phase + And the staff entity is stored as a parquet after the file_transformation phase + And the flights entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the latest audit record for the submission is marked with processing status data_contract + When I run the data contract phase + Then there are no file rejections from the data_contract phase + And there is 1 record rejection from the data_contract phase + And there are errors with the following details and associated error_count from the data_contract phase + | FailureType | ErrorCode | error_count | + | record | CountryIdIsMissing | 1 | + When I run the business rules phase + Then there are errors with the following details and associated error_count from the business_rules phase + | ErrorType | ErrorCode | error_count | + | record | AirportHasNoCountry | 3 | + | record | StaffHasNoAirport | 15 | + | record | FlightHasNoAirport | 10 | + | record | PassengerHasNoFlight | 25 | + | submission | NoValidCountries | 1 | + | submission | NoValidAirports | 1 | + | submission | NoValidStaff | 1 | + And the final entities have the following row counts + | entity_name | row_count | + | country | 0 | + | airport | 0 | + | staff | 0 | + | flights | 0 | + | passengers | 0 | + When I run the error report phase + Then An error report is produced + And The statistics entry for the submission shows the following information + | parameter | value | + | record_count | 1 | + | number_submission_rejections | 3 | + | number_record_rejections | 54 | + | number_warnings | 0 | + + Scenario: A flights submission where a child primary key is rejected + Given I submit the flights file missing_flight_id.xml for processing + And A duckdb pipeline is configured with schema file 'flights.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And the airport entity is stored as a parquet after the file_transformation phase + And the staff entity is stored as a parquet after the file_transformation phase + And the flights entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the latest audit record for the submission is marked with processing status data_contract + When I run the data contract phase + Then there are no file rejections from the data_contract phase + And there are no record rejections from the data_contract phase + When I run the business rules phase + Then there are errors with the following details and associated error_count from the business_rules phase + | ErrorType | ErrorCode | error_count | + | record | FlightIDMissing | 1 | + | record | PassengerHasNoFlight | 3 | + And the final entities have the following row counts + | entity_name | row_count | + | country | 1 | + | airport | 3 | + | staff | 15 | + | flights | 9 | + | passengers | 22 | + When I run the error report phase + Then An error report is produced + And The statistics entry for the submission shows the following information + | parameter | value | + | record_count | 1 | + | number_submission_rejections | 0 | + | number_record_rejections | 4 | + | number_warnings | 0 | + + Scenario: A flights submission with only country id and name submitted + Given I submit the flights file only_country_id.xml for processing + And A duckdb pipeline is configured with schema file 'flights.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And the airport entity is stored as a parquet after the file_transformation phase + And the flights entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the latest audit record for the submission is marked with processing status data_contract + When I run the data contract phase + Then there are no file rejections from the data_contract phase + And there are no record rejections from the data_contract phase + When I run the business rules phase + Then there are errors with the following details and associated error_count from the business_rules phase + | ErrorType | ErrorCode | error_count | + | record | CountryHasNoAirport | 1 | + | submission | NoValidCountries | 1 | + | submission | NoValidAirports | 1 | + | submission | NoValidStaff | 1 | + And the final entities have the following row counts + | entity_name | row_count | + | country | 0 | + | airport | 0 | + | staff | 0 | + | flights | 0 | + | passengers | 0 | + When I run the error report phase + Then An error report is produced + And The statistics entry for the submission shows the following information + | parameter | value | + | record_count | 1 | + | number_submission_rejections | 3 | + | number_record_rejections | 1 | + | number_warnings | 0 | + + Scenario: A flights submission with a rejection on a node with one mandatory node + Given I submit the flights file singular_node_rejections.xml for processing + And A duckdb pipeline is configured with schema file 'flights.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And the airport entity is stored as a parquet after the file_transformation phase + And the flights entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the latest audit record for the submission is marked with processing status data_contract + When I run the data contract phase + Then there are no file rejections from the data_contract phase + And there are no record rejections from the data_contract phase + When I run the business rules phase + Then there are errors with the following details and associated error_count from the business_rules phase + | ErrorType | Status | ErrorCode | error_count | + | record | error | InvalidFlightDestination | 2 | + | record | error | PassengerHasNoFlight | 4 | + | record | error | AirportHasNoStaff | 1 | + | record | error | CountryHasNoAirport | 1 | + | submission | error | NoValidCountries | 1 | + | submission | error | NoValidAirports | 1 | + | submission | error | NoValidStaff | 1 | + And the final entities have the following row counts + | entity_name | row_count | + | country | 0 | + | airport | 0 | + | staff | 0 | + | flights | 0 | + | passengers | 0 | + When I run the error report phase + Then An error report is produced + And The statistics entry for the submission shows the following information + | parameter | value | + | record_count | 1 | + | number_submission_rejections | 3 | + | number_record_rejections | 8 | + | number_warnings | 0 | + + Scenario: A flights submission with a rejection on a node with two mandatory nodes + Given I submit the flights file multi_node_file_rejection.xml for processing + And A duckdb pipeline is configured with schema file 'flights.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And the airport entity is stored as a parquet after the file_transformation phase + And the flights entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the latest audit record for the submission is marked with processing status data_contract + When I run the data contract phase + Then there are no file rejections from the data_contract phase + And there are no record rejections from the data_contract phase + When I run the business rules phase + Then there are errors with the following details and associated error_count from the business_rules phase + | ErrorType | Status | ErrorCode | error_count | + | record | error | StaffIDMissing | 7 | + | record | error | AirportHasNoStaff | 1 | + | record | error | FlightHasNoAirport | 1 | + | record | error | PassengerHasNoFlight | 1 | + And the final entities have the following row counts + | entity_name | row_count | + | country | 1 | + | airport | 1 | + | staff | 1 | + | flights | 1 | + | passengers | 1 | + When I run the error report phase + Then An error report is produced + And The statistics entry for the submission shows the following information + | parameter | value | + | record_count | 1 | + | number_submission_rejections | 0 | + | number_record_rejections | 10 | + | number_warnings | 0 | + + Scenario: A flights submission with many types of rejections in a single submission + Given I submit the flights file flights_full_regression.xml for processing + And A duckdb pipeline is configured with schema file 'flights.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And the airport entity is stored as a parquet after the file_transformation phase + And the flights entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the latest audit record for the submission is marked with processing status data_contract + When I run the data contract phase + Then there are errors with the following details and associated error_count from the data_contract phase + | FailureType | ErrorCode | error_count | + | record | AirportIdIsMissing | 1 | + When I run the business rules phase + Then there are errors with the following details and associated error_count from the business_rules phase + | ErrorType | Status | ErrorCode | error_count | + | record | error | InvalidFlightDestination | 1 | + | record | error | PassengerNameMissing | 1 | + | record | error | StaffIDMissing | 4 | + | record | error | PassengerHasNoFlight | 3 | + | record | error | StaffHasNoAirport | 1 | + | record | error | FlightHasNoAirport | 2 | + | record | error | AirportHasNoStaff | 1 | + And the final entities have the following row counts + | entity_name | row_count | + | country | 1 | + | airport | 3 | + | staff | 3 | + | flights | 3 | + | passengers | 2 | + When I run the error report phase + Then An error report is produced + And The statistics entry for the submission shows the following information + | parameter | value | + | record_count | 1 | + | number_submission_rejections | 0 | + | number_record_rejections | 14 | + | number_warnings | 0 | + + Scenario: A flights submission where mandatory entity has no records submitted + Given I submit the flights file only_country_id.xml for processing + And A duckdb pipeline is configured with schema file 'flights_add_reader_checks.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And there are errors with the following details and associated error_count from the file_transformation phase + | FailureType | ErrorCode | error_count | + | submission | AIRPORTEMPTY | 1 | + And the latest audit record for the submission is marked with processing status error_report + When I run the error report phase + Then An error report is produced + + Scenario: A flights submission with a rejection on a node with two mandatory nodes (spark) + Given I submit the flights file multi_node_file_rejection.xml for processing + And A spark pipeline is configured with schema file 'flights_spark.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And the airport entity is stored as a parquet after the file_transformation phase + And the flights entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the latest audit record for the submission is marked with processing status data_contract + When I run the data contract phase + Then there are no file rejections from the data_contract phase + And there are no record rejections from the data_contract phase + When I run the business rules phase + Then there are errors with the following details and associated error_count from the business_rules phase + | ErrorType | Status | ErrorCode | error_count | + | record | error | StaffIDMissing | 7 | + | record | error | AirportHasNoStaff | 1 | + | record | error | FlightHasNoAirport | 1 | + | record | error | PassengerHasNoFlight | 1 | + And the final entities have the following row counts + | entity_name | row_count | + | country | 1 | + | airport | 1 | + | staff | 1 | + | flights | 1 | + | passengers | 1 | + When I run the error report phase + Then An error report is produced + And The statistics entry for the submission shows the following information + | parameter | value | + | record_count | 1 | + | number_submission_rejections | 0 | + | number_record_rejections | 10 | + | number_warnings | 0 | + + Scenario: A flights submission with many types of rejections in a single submission (spark) + Given I submit the flights file flights_full_regression.xml for processing + And A spark pipeline is configured with schema file 'flights_spark.dischema.json' + And I add initial audit entries for the submission + Then the latest audit record for the submission is marked with processing status file_transformation + When I run the file transformation phase + Then the country entity is stored as a parquet after the file_transformation phase + And the airport entity is stored as a parquet after the file_transformation phase + And the flights entity is stored as a parquet after the file_transformation phase + And the passengers entity is stored as a parquet after the file_transformation phase + And the latest audit record for the submission is marked with processing status data_contract + When I run the data contract phase + Then there are errors with the following details and associated error_count from the data_contract phase + | FailureType | ErrorCode | error_count | + | record | AirportIdIsMissing | 1 | + When I run the business rules phase + Then there are errors with the following details and associated error_count from the business_rules phase + | ErrorType | Status | ErrorCode | error_count | + | record | error | InvalidFlightDestination | 1 | + | record | error | PassengerNameMissing | 1 | + | record | error | StaffIDMissing | 4 | + | record | error | PassengerHasNoFlight | 3 | + | record | error | StaffHasNoAirport | 1 | + | record | error | FlightHasNoAirport | 2 | + | record | error | AirportHasNoStaff | 1 | + And the final entities have the following row counts + | entity_name | row_count | + | country | 1 | + | airport | 3 | + | staff | 3 | + | flights | 3 | + | passengers | 2 | + When I run the error report phase + Then An error report is produced + And The statistics entry for the submission shows the following information + | parameter | value | + | record_count | 1 | + | number_submission_rejections | 0 | + | number_record_rejections | 14 | + | number_warnings | 0 | diff --git a/tests/features/movies.feature b/tests/features/movies.feature index 750975e2..c6d5ad61 100644 --- a/tests/features/movies.feature +++ b/tests/features/movies.feature @@ -22,7 +22,7 @@ Feature: Pipeline tests using the movies dataset Then there is 1 submission rejection from the data_contract phase And there are 3 record rejections from the data_contract phase And there are errors with the following details and associated error_count from the data_contract phase - | Entity | ErrorCode | ErrorMessage | RecordIndex | error_count | + | ReportingEntity | ErrorCode | ErrorMessage | RecordIndex | error_count | | movies | BLANKYEAR | year not provided | 2 | 1 | | movies_rename_test | DODGYYEAR | year value (NOT_A_NUMBER) is invalid | 1 | 1 | | movies | DODGYDATE | date_joined value is not valid: daft_date | 1 | 1 | @@ -61,10 +61,10 @@ Feature: Pipeline tests using the movies dataset Then there is 1 submission rejection from the data_contract phase And there are 3 record rejections from the data_contract phase And there are errors with the following details and associated error_count from the data_contract phase - | Entity | ErrorCode | ErrorMessage | RecordIndex | error_count | - | movies | BLANKYEAR | year not provided | 2 | 1 | - | movies_rename_test | DODGYYEAR | year value (NOT_A_NUMBER) is invalid | 1 | 1 | - | movies | DODGYDATE | date_joined value is not valid: daft_date | 1 | 1 | + | ReportingEntity | ErrorCode | ErrorMessage | RecordIndex | error_count | + | movies | BLANKYEAR | year not provided | 2 | 1 | + | movies_rename_test | DODGYYEAR | year value (NOT_A_NUMBER) is invalid | 1 | 1 | + | movies | DODGYDATE | date_joined value is not valid: daft_date | 1 | 1 | | movies | BLANKTITLE | title should not be blank | 4 | 1 | And the movies entity is stored as a parquet after the data_contract phase And the latest audit record for the submission is marked with processing status business_rules diff --git a/tests/features/planets.feature b/tests/features/planets.feature index b37b60b9..0c4b21d6 100644 --- a/tests/features/planets.feature +++ b/tests/features/planets.feature @@ -43,7 +43,9 @@ Feature: Pipeline tests using the planets dataset And I add initial audit entries for the submission Then the latest audit record for the submission is marked with processing status file_transformation When I run the file transformation phase - Then the latest audit record for the submission is marked with processing status failed + Then the latest audit record for the submission is marked with processing status error_report + When I run the error report phase + Then An error report is produced Scenario: Handle a file with duplicated extension provided (spark) Given I submit the planets file planets.csv.csv for processing diff --git a/tests/features/steps/steps_pipeline.py b/tests/features/steps/steps_pipeline.py index c72c8737..2bfcf1a9 100644 --- a/tests/features/steps/steps_pipeline.py +++ b/tests/features/steps/steps_pipeline.py @@ -33,8 +33,8 @@ from utilities import ( load_errors_from_service, - get_test_file_path, SERVICE_TO_STORAGE_PATH_MAPPING, + get_test_file_path, get_all_errors_df, ) @@ -183,8 +183,10 @@ def check_error_record_details_from_service(context: Context, service:str): message_df = load_errors_from_service(processing_path, service) for err_details in error_details: filter_expr, error_count = err_details - assert message_df.filter(filter_expr).shape[0] == error_count - + assert message_df.filter(filter_expr).shape[0] == error_count, message_df.select( + *[pl.col(c) for c in table.headings if c not in ["error_count"]] + ) + @given("A {implementation} pipeline is configured") @given("A {implementation} pipeline is configured with schema file '{schema_file_name}'") @@ -282,7 +284,7 @@ def check_rows_removed_with_error_code(context: Context, entity_name: str, error err_df = get_all_errors_df(context) recs_with_err_code = err_df.filter( - (pl.col("Entity").eq(entity_name)) & (pl.col("ErrorCode").eq(error_code)) + (pl.col("ReportingEntity").eq(entity_name)) & (pl.col("ErrorCode").eq(error_code)) ).shape[0] assert recs_with_err_code >= 1 @@ -317,8 +319,3 @@ def create_refdata_tables(context: Context, database: str): pipeline._connection.sql(f"ATTACH '{ref_db_file}' AS {database}") for tbl, source in refdata_tables.items(): pipeline._connection.read_parquet(source).to_table(f"{database}.{tbl}") - - - - - diff --git a/tests/features/steps/steps_post_pipeline.py b/tests/features/steps/steps_post_pipeline.py index 906445e4..b284b3cc 100644 --- a/tests/features/steps/steps_post_pipeline.py +++ b/tests/features/steps/steps_post_pipeline.py @@ -110,10 +110,24 @@ def check_stats_record(context): stats = ( get_pipeline(context)._audit_tables.get_submission_statistics(sub_info.submission_id).model_dump() ) - assert all([val == stats.get(fld) for fld, val in expected.items()]) + assert all([val == stats.get(fld) for fld, val in expected.items()]), stats @then("the error aggregates are persisted") def check_error_aggregates_persisted(context): processing_location = get_processing_location(context) agg_file = Path(processing_location, "audit", "error_aggregates.parquet") assert agg_file.exists() and agg_file.is_file() + +@then("the final entities have the following row counts") +def check_entity_row_counts(context: Context): + processing_loc = get_processing_location(context) + submission_info = get_submission_info(context) + table: Table = context.table + if table is None: + raise ValueError("No table supplied in step") + for row in table: + record = row.as_dict() + entity_name = record["entity_name"] + expected_count = int(record["row_count"]) + output_df = read_output_parquet(processing_loc, entity_name, "business_rules") + assert expected_count == output_df.shape[0], output_df diff --git a/tests/features/steps/utilities.py b/tests/features/steps/utilities.py index 58edc677..95b73ad7 100644 --- a/tests/features/steps/utilities.py +++ b/tests/features/steps/utilities.py @@ -15,7 +15,7 @@ from dve.parser.type_hints import URI ERROR_DF_FIELDS: List[str] = [ - "Entity", + "ReportingEntity", "Key", "ErrorCode", "FailureType", @@ -49,7 +49,7 @@ def load_errors_from_service(processing_folder: Path, service: str) -> pl.DataFr err_location = Path( processing_folder, "errors", - f"{SERVICE_TO_STORAGE_PATH_MAPPING.get(service, service)}_errors.jsonl", + f"{service}_errors.jsonl", ) msgs = [] try: diff --git a/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_data_contract.py b/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_data_contract.py index 2019a665..2e6bf87f 100644 --- a/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_data_contract.py +++ b/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_data_contract.py @@ -381,4 +381,4 @@ def test_duckdb_data_contract_custom_error_details(nested_all_string_parquet_w_e assert messages[0].ErrorMessage == "subfield id is invalid: subfield.id - WRONG" assert messages[1].ErrorCode == "TESTIDBAD" assert messages[1].ErrorMessage == "id is invalid: id - WRONG" - assert messages[1].Entity == "test_rename" \ No newline at end of file + assert messages[1].ReportingEntity == "test_rename" \ No newline at end of file diff --git a/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_duckdb_helpers.py b/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_duckdb_helpers.py index 4eefe05f..0e81cf46 100644 --- a/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_duckdb_helpers.py +++ b/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_duckdb_helpers.py @@ -76,7 +76,8 @@ def example_data_contract_error_codes(temp_ddb_conn): test_entity = con.sql("SELECT * FROM test_df") error_contract_messages = [ { - "Entity": "test_entity", + "ReportingEntity": "test_entity", + "OriginalEntity": "test_entity", "Key": "", "FailureType": "record", "Status": "error", @@ -90,7 +91,8 @@ def example_data_contract_error_codes(temp_ddb_conn): "Category": "Bad value" }, { - "Entity": "test_entity", + "ReportingEntity": "test_entity", + "OriginalEntity": "test_entity", "Key": "", "FailureType": "record", "Status": "error", diff --git a/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_rules.py b/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_rules.py index 35007a9c..aae56967 100644 --- a/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_rules.py +++ b/tests/test_core_engine/test_backends/test_implementations/test_duckdb/test_rules.py @@ -1,6 +1,10 @@ """Test DuckDB backend steps.""" # pylint: disable=redefined-outer-name,unused-import,line-too-long +import datetime +from io import StringIO +import json +import tempfile from pathlib import Path from typing import Iterator, List, Optional, Set, Tuple, Type @@ -39,6 +43,9 @@ SemiJoin, TableUnion, ) +from dve.core_engine.configuration.v1.hierarchy import ( + EntityHierarchy, HierarchyNode +) from dve.core_engine.type_hints import MultipleExpressions from tests.test_core_engine.test_backends.fixtures import ( duckdb_connection, @@ -581,91 +588,106 @@ def test_header_multi_rows_raises( DUCKDB_STEP_BACKEND.join_header(entities, config=header_join) -def test_orphans_planets_satellites( - planets_rel: DuckDBPyRelation, largest_satellites_rel: DuckDBPyRelation -): - """Test a basic orphan idenfitication from satellites to planets.""" - # Each satellite _must_ have a planet. - join = OrphanIdentification( - entity_name="satellites", - target_name="planets", - join_condition="satellites.planet == planets.planet", - ) - entities = EntityManager( - { - "planets": planets_rel.filter(ColumnExpression("Planet") != ConstantExpression("Mars")), - "satellites": largest_satellites_rel, - } - ) - - DUCKDB_STEP_BACKEND.evaluate(entities, config=join) - actual_rel = ( - entities["satellites"] - .filter(ColumnExpression("IsOrphaned")) - .select(ColumnExpression("name")) - ) - actual_rows = sorted(actual_rel.df().to_dict(orient="records"), key=lambda row: row["name"]) - - expected_rel = largest_satellites_rel.filter( - ColumnExpression("Planet") == ConstantExpression("Mars") - ).select(ColumnExpression("name")) - expected_rows = sorted(expected_rel.df().to_dict(orient="records"), key=lambda row: row["name"]) - - assert actual_rows == expected_rows - - -def test_chained_orphans_planets_satellites( - planets_rel: DuckDBPyRelation, largest_satellites_rel: DuckDBPyRelation -): - """Test a basic chained orphan idenfitication from satellites to planets.""" - join = OrphanIdentification( - entity_name="satellites", - target_name="planets", - join_condition="satellites.planet == planets.planet", - ) - entities = EntityManager( - { - "planets": planets_rel.filter(ColumnExpression("planet") != ConstantExpression("Mars")), - "satellites": largest_satellites_rel, - } - ) - DUCKDB_STEP_BACKEND.evaluate(entities, config=join) - entities["planets"] = planets_rel.filter( - ColumnExpression("planet") != ConstantExpression("Earth") - ) - DUCKDB_STEP_BACKEND.evaluate(entities, config=join) - - actual_rel = ( - entities["satellites"] - .filter(ColumnExpression("IsOrphaned")) - .select(ColumnExpression("name")) - ) - actual_rows = sorted(actual_rel.df().to_dict(orient="records"), key=lambda row: row["name"]) - - expected_rel = largest_satellites_rel.filter( - ColumnExpression("Planet").isin(ConstantExpression("Mars"), ConstantExpression("Earth")) - ).select(ColumnExpression("name")) - expected_rows = sorted(expected_rel.df().to_dict(orient="records"), key=lambda row: row["name"]) - - assert actual_rows == expected_rows - - -def test_orphans_missing_entities_raises( - planets_rel: DuckDBPyRelation, satellites_rel: DuckDBPyRelation -): - """Test that trying to join orphans from missing entities raises correctly.""" - join = OrphanIdentification( - entity_name="planets", - target_name="satellites", - join_condition="planets.planet == satellites.planet", - ) +class TestOrphanRecords: + """ + Check that Orphan records identification and removal is working as expected. - entities = EntityManager({"planets": planets_rel}) - with pytest.raises(MissingEntity): - DUCKDB_STEP_BACKEND.identify_orphans(entities, config=join) - entities = EntityManager({"satellites": satellites_rel}) - with pytest.raises(MissingEntity): - DUCKDB_STEP_BACKEND.identify_orphans(entities, config=join) + Current scenarios are: + Flight ID 1 = Perfect Record - no orphans + Flight ID 2 = Record rejected at the flights entity, therefore, two expected orphans in passengers and food entities. + """ + mod_flights_df = pl.DataFrame([ + {'flight_id': 1, '__record_index__': 1}, + ]) + mod_passengers_df = pl.DataFrame([ + {'flight_id': 1, 'passenger_id': 1, '__record_index__': 1}, + {'flight_id': 1, 'passenger_id': 2, '__record_index__': 2}, + {'flight_id': 2, 'passenger_id': 3, '__record_index__': 3}, + ]) + mod_food_df = pl.DataFrame([ + {'passenger_id': 1, 'food_id': 1, '__record_index__': 1}, + {'passenger_id': 1, 'food_id': 2, '__record_index__': 2}, + {'passenger_id': 3, 'food_id': 3, '__record_index__': 3}, + ]) + + def test_identify_orphan_record_single_entity(self): + """Ensure that a single one-to-one check works to identify orphan records.""" + with duckdb.connect() as cnn: + cnn.register("mod_flights", self.mod_flights_df) + cnn.register("mod_passengers", self.mod_passengers_df) + + mod_entities = EntityManager( + entities={ + "flights": cnn.sql("SELECT * FROM mod_flights"), + "passengers": cnn.sql("SELECT * FROM mod_passengers"), + } + ) + + rules = DuckDBStepImplementations(connection=cnn) + msgs = rules.identify_orphans( + mod_entities.entities, + config=OrphanIdentification( + id="flight_id", + entity_name="passengers", + target_name="flights", + join_condition="passengers.flight_id = flights.flight_id" + ) + ) + assert len(list(msgs)) == 1 + + def test_identify_and_remove_orphans(self): + with duckdb.connect() as cnn: + cnn.register("mod_flights", self.mod_flights_df) + cnn.register("mod_passengers", self.mod_passengers_df) + cnn.register("mod_food", self.mod_food_df) + + mod_entities = EntityManager( + entities={ + "flights": cnn.sql("SELECT * FROM mod_flights"), + "passengers": cnn.sql("SELECT * FROM mod_passengers"), + "food": cnn.sql("SELECT * FROM mod_food"), + } + ) + + rules = DuckDBStepImplementations(connection=cnn) + hierarchy = EntityHierarchy({ + "flights": HierarchyNode( + entity_name="flights", + children=[ + HierarchyNode( + entity_name="passengers", + parent_entity="flights", + children=[ + HierarchyNode( + entity_name="food", + parent_entity="passengers", + children=[], + join_fields={"passenger_id": "passenger_id"}, + mandatory=False + ) + ], + join_fields={"flight_id": "flight_id"}, + mandatory=True + ) + ] + ) + }) + + with tempfile.TemporaryDirectory() as wd: + rules.identify_and_remove_orphans( + wd, + mod_entities.entities, + hierarchy + ) + + flights_rel = mod_entities["flights"] + assert flights_rel.select("__record_index__").unique("*").count("*").fetchone()[0] == 1 # type: ignore + + passenger_rel = mod_entities["passengers"] + assert passenger_rel.select("__record_index__").unique("*").count("*").fetchone()[0] == 2 # type: ignore + + food_rel = mod_entities["food"] + assert food_rel.select("__record_index__").unique("*").count("*").fetchone()[0] == 2 # type: ignore def test_has_match_planets_satellites( @@ -798,6 +820,25 @@ def test_planets_notify(planets_rel: DuckDBPyRelation): assert len(messages[0]) == 4 +def test_notify_null_errors(planets_rel: DuckDBPyRelation): + + config = Notification( + entity_name="planets", + expression="CASE WHEN planet=='Mercury' THEN NULL ELSE False END", + excluded_columns=["mass", "diameter"], + reporting=ReportingConfig( + code="TESTNULLERROR", message="this is a test", location="planet, has_ring_system" + ), + error_if_expression_null=True + ) + entities = EntityManager({"planets": planets_rel}) + messages, success = DUCKDB_STEP_BACKEND.evaluate(entities, config=config) + + assert not success + assert len(messages) == 1 + assert messages[0].is_critical + + def test_read_and_write_simple_parquet(simple_typecast_parquet): parquet_uri, data = simple_typecast_parquet entity: DuckDBPyRelation = DUCKDB_STEP_BACKEND.read_parquet(path=parquet_uri) @@ -846,3 +887,112 @@ def test_read_and_write_nested_parquet(nested_typecast_parquet): "datetimefield": "TIMESTAMP", "subfield": "STRUCT(id BIGINT, substrfield VARCHAR, subarrayfield DATE[])[]", } + +def test_cache_management(): + conn = DUCKDB_STEP_BACKEND.connection + with tempfile.NamedTemporaryFile(mode="w") as tf1, tempfile.NamedTemporaryFile(mode="w") as tf2: + td1 = [ + {"greeting": "hi", "num_one": 2, "num_two": 4, "test_date": datetime.date(2020,5,1), "active": True}, + {"greeting": "bonjour", "num_one": 3, "num_two": 9, "test_date": datetime.date(2025,7,4), "active": False}, + ] + tf1.write(json.dumps(td1, default=str)) + + tf1.seek(0) + + td1_schema = {"greeting": "STRING", "num_one": "BIGINT", "num_two" :"BIGINT", "test_date": "DATE", "active": "BOOLEAN"} + + td2 = [ + {"farewell": "aurevoir", "lots_of_nums": [1,2,3], "nested_field": {"nested_str": "test1", "nested_timestamp": datetime.datetime(2024,3,5,1,2,3)}}, + {"farewell": "bye", "lots_of_nums": [4,5,6,7,8], "nested_field": {"nested_str": "test2", "nested_timestamp": datetime.datetime(2022,8,12,10,12,14)}}, + ] + + tf2.write(json.dumps(td2, default=str)) + + tf2.seek(0) + + td2_schema = {"farewell": "STRING", "lots_of_nums": "BIGINT[]", "nested_field": "STRUCT(nested_str STRING, nested_timestamp TIMESTAMP)"} + + em = EntityManager({}) + em.entities["test_one"] = conn.read_json(tf1.name, columns=td1_schema) + em.entities["test_two"] = conn.read_json(tf2.name, columns=td2_schema) + + DUCKDB_STEP_BACKEND.cache_entity("test_one", em.entities) + + cached_tables = [rw["table_name"] for rw in conn.sql("SELECT table_name from duckdb_tables() WHERE database_name = 'temp'").pl().to_dicts()] + + test_one_temp_name = list(filter(lambda x: x.startswith("test_one_"), cached_tables))[0] + + assert "test_one" in DUCKDB_STEP_BACKEND.entity_cache_tracker + assert DUCKDB_STEP_BACKEND.entity_cache_tracker["test_one"] == test_one_temp_name + assert sorted(em.entities["test_one"].pl().to_dicts(), key=lambda x: x.get("test_date")) == td1 + assert sorted(conn.table(test_one_temp_name).pl().to_dicts(), key=lambda x: x.get("test_date")) == td1 + assert "test_two" not in DUCKDB_STEP_BACKEND.entity_cache_tracker + + DUCKDB_STEP_BACKEND.cache_entity("test_two", em.entities) + DUCKDB_STEP_BACKEND._remove_cached_artifact("test_one") + + cached_tables = [rw["table_name"] for rw in conn.sql("SELECT table_name from duckdb_tables() WHERE database_name = 'temp'").pl().to_dicts()] + test_two_temp_name = list(filter(lambda x: x.startswith("test_two_"), cached_tables))[0] + + assert "test_two" in DUCKDB_STEP_BACKEND.entity_cache_tracker + assert DUCKDB_STEP_BACKEND.entity_cache_tracker["test_two"] in cached_tables + assert sorted(em.entities["test_two"].pl().to_dicts(), key=lambda x: x.get("farewell")) == td2 + assert sorted(conn.table(test_two_temp_name).pl().to_dicts(), key=lambda x: x.get("farewell")) == td2 + assert "test_one" not in DUCKDB_STEP_BACKEND.entity_cache_tracker + + DUCKDB_STEP_BACKEND.clear_entity_cache() + + cached_tables = [rw["table_name"] for rw in conn.sql("SELECT table_name from duckdb_tables() WHERE database_name = 'temp'").pl().to_dicts()] + + assert not DUCKDB_STEP_BACKEND.entity_cache_tracker + assert not any(tbl.startswith("test_one_") for tbl in cached_tables) + assert not any(tbl.startswith("test_two_") for tbl in cached_tables) + +def test_cache_management_with_update(): + conn = DUCKDB_STEP_BACKEND.connection + with tempfile.NamedTemporaryFile(mode="w") as tf, tempfile.NamedTemporaryFile(mode="w") as ef: + td = [ + {"idx": 1, "lots_of_nums": [1,2,3], "nested_field": {"nested_str": "test1", "nested_timestamp": datetime.datetime(2024,3,5,1,2,3)}}, + {"idx": 2, "lots_of_nums": [4,5,6,7,8], "nested_field": {"nested_str": "test2", "nested_timestamp": datetime.datetime(2022,8,12,10,12,14)}}, + ] + + tf.write(json.dumps(td, default=str)) + tf.seek(0) + td_schema = {"idx": "BIGINT", + "lots_of_nums": "BIGINT[]", + "nested_field": "STRUCT(nested_str STRING, nested_timestamp TIMESTAMP)"} + + + em = EntityManager({}) + em.entities["test_df"] = conn.read_json(tf.name, columns=td_schema) + + DUCKDB_STEP_BACKEND.cache_entity("test_df", em.entities) + + cached_tables = [rw["table_name"] for rw in conn.sql("SELECT table_name from duckdb_tables() WHERE database_name = 'temp'").pl().to_dicts()] + + first_temp_name = list(filter(lambda x: x.startswith("test_df_"), cached_tables))[0] + + assert "test_df" in DUCKDB_STEP_BACKEND.entity_cache_tracker + assert DUCKDB_STEP_BACKEND.entity_cache_tracker["test_df"] == first_temp_name + assert sorted(em.entities["test_df"].pl().to_dicts(), key=lambda x: x.get("idx")) == td + + extra_data = [{"idx": 3, "lots_of_nums": [9], "nested_field": {"nested_str": "test3", "nested_timestamp": datetime.datetime(2024,1,9,3,2,1)}}] + ef.write(json.dumps(extra_data, default=str)) + ef.seek(0) + em.entities["test_df"] = em.entities["test_df"].union(conn.read_json(ef.name, columns=td_schema)) + DUCKDB_STEP_BACKEND.cache_entity("test_df", em.entities) + + cached_tables = [rw["table_name"] for rw in conn.sql("SELECT table_name from duckdb_tables() WHERE database_name = 'temp'").pl().to_dicts()] + tables_of_interest = list(filter(lambda x: x.startswith("test_df_"), cached_tables)) + assert len(tables_of_interest) == 1 + assert first_temp_name not in tables_of_interest + second_temp_name = tables_of_interest[0] + assert sorted(em.entities["test_df"].pl().to_dicts(), key=lambda x: x.get("idx")) == td + extra_data + assert sorted(conn.table(second_temp_name).pl().to_dicts(), key=lambda x: x.get("idx")) == td + extra_data + + DUCKDB_STEP_BACKEND.clear_entity_cache() + + cached_tables = [rw["table_name"] for rw in conn.sql("SELECT table_name from duckdb_tables() WHERE database_name = 'temp'").pl().to_dicts()] + + assert not DUCKDB_STEP_BACKEND.entity_cache_tracker + assert not any(tbl.startswith("test_df_") for tbl in cached_tables) diff --git a/tests/test_core_engine/test_backends/test_implementations/test_spark/test_data_contract.py b/tests/test_core_engine/test_backends/test_implementations/test_spark/test_data_contract.py index 70c6b9c4..ea5fc461 100644 --- a/tests/test_core_engine/test_backends/test_implementations/test_spark/test_data_contract.py +++ b/tests/test_core_engine/test_backends/test_implementations/test_spark/test_data_contract.py @@ -242,6 +242,6 @@ def test_spark_data_contract_custom_error_details(nested_all_string_parquet_w_er assert messages[0].ErrorMessage == "subfield id is invalid: subfield.id - WRONG" assert messages[1].ErrorCode == "TESTIDBAD" assert messages[1].ErrorMessage == "id is invalid: id - WRONG" - assert messages[1].Entity == "test_rename" + assert messages[1].ReportingEntity == "test_rename" \ No newline at end of file diff --git a/tests/test_core_engine/test_backends/test_implementations/test_spark/test_rules.py b/tests/test_core_engine/test_backends/test_implementations/test_spark/test_rules.py index 673e6119..2a79d81b 100644 --- a/tests/test_core_engine/test_backends/test_implementations/test_spark/test_rules.py +++ b/tests/test_core_engine/test_backends/test_implementations/test_spark/test_rules.py @@ -1,6 +1,7 @@ """Test Spark backend steps.""" # pylint: disable=redefined-outer-name,unused-import,line-too-long +import datetime from pathlib import Path from typing import List, Optional, Set, Tuple, Type @@ -9,6 +10,7 @@ from pyspark.sql.functions import col, lit from pyspark.sql.types import ( ArrayType, + BooleanType, DateType, LongType, Row, @@ -21,6 +23,7 @@ from dve.core_engine.backends.base.core import EntityManager from dve.core_engine.backends.exceptions import MissingEntity from dve.core_engine.backends.implementations.spark.rules import SparkStepImplementations +from dve.core_engine.backends.metadata.reporting import ReportingConfig from dve.core_engine.backends.metadata.rules import ( Aggregation, AntiJoin, @@ -32,6 +35,7 @@ HeaderJoin, InnerJoin, LeftJoin, + Notification, OneToOneJoin, OrphanIdentification, RenameEntity, @@ -447,6 +451,24 @@ def test_join_can_take_all_cols( expected_rows = sorted(expected_df.collect(), key=lambda row: row.planet) assert actual_rows == expected_rows + +def test_notify_null_errors(planets_df: DataFrame): + + config = Notification( + entity_name="planets", + expression="CASE WHEN planet=='Mercury' THEN NULL ELSE False END", + excluded_columns=["mass", "diameter"], + reporting=ReportingConfig( + code="TESTNULLERROR", message="this is a test", location="planet, has_ring_system" + ), + error_if_expression_null=True + ) + entities = EntityManager({"planets": planets_df}) + messages, success = SPARK_STEP_BACKEND.evaluate(entities, config=config) + + assert not success + assert len(messages) == 1 + assert messages[0].is_critical def test_one_to_one_join_multi_matches_raises(planets_df: DataFrame, satellites_df: DataFrame): @@ -568,6 +590,7 @@ def test_header_multi_rows_raises(planets_df: DataFrame, value_literal_1_header: SPARK_STEP_BACKEND.join_header(entities, config=header_join) +@pytest.mark.skip(reason="Logic is no longer valid") def test_orphans_planets_satellites(planets_df: DataFrame, largest_satellites_df: DataFrame): """Test a basic orphan idenfitication from satellites to planets.""" # Each satellite _must_ have a planet. @@ -593,6 +616,7 @@ def test_orphans_planets_satellites(planets_df: DataFrame, largest_satellites_df assert actual_rows == expected_rows +@pytest.mark.skip(reason="Logic is no longer valid") def test_chained_orphans_planets_satellites( planets_df: DataFrame, largest_satellites_df: DataFrame ): @@ -623,6 +647,7 @@ def test_chained_orphans_planets_satellites( assert actual_rows == expected_rows +@pytest.mark.skip(reason="Logic is no longer valid") def test_orphans_missing_entities_raises(planets_df: DataFrame, satellites_df: DataFrame): """Test that trying to join orphans from missing entities raises correctly.""" join = OrphanIdentification( @@ -829,3 +854,107 @@ def test_read_and_write_nested_parquet(nested_typecast_parquet): ), ] ) + +def test_cache_management(): + spark = SPARK_STEP_BACKEND._spark_session + td1 = [ + {"greeting": "hi", "num_one": 2, "num_two": 4, "test_date": datetime.date(2020,5,1), "active": True}, + {"greeting": "bonjour", "num_one": 3, "num_two": 9, "test_date": datetime.date(2025,7,4), "active": False}, + ] + td1_schema = StructType([ + StructField("greeting", StringType()), + StructField("num_one", LongType()), + StructField("num_two", LongType()), + StructField("test_date", DateType()), + StructField("active", BooleanType())]) + td2 = [ + {"farewell": "aurevoir", "lots_of_nums": [1,2,3], "nested_field": {"nested_str": "test1", "nested_timestamp": datetime.datetime(2024,3,5,1,2,3)}}, + {"farewell": "bye", "lots_of_nums": [4,5,6,7,8], "nested_field": {"nested_str": "test2", "nested_timestamp": datetime.datetime(2022,8,12,10,12,14)}}, + ] + td2_schema = StructType( + [ + StructField("farewell", StringType()), + StructField("lots_of_nums", ArrayType(LongType())), + StructField("nested_field", StructType([StructField("nested_str", StringType()), StructField("nested_timestamp", TimestampType())])) + ] + ) + em = EntityManager({}) + em.entities["test_one"] = spark.createDataFrame(td1, schema=td1_schema) + em.entities["test_two"] = spark.createDataFrame(td2, schema=td2_schema) + + SPARK_STEP_BACKEND.cache_entity("test_one", em.entities) + + cached_tables = (rw.tableName for rw in spark.sql("SHOW TABLES").collect()) + + test_one_temp_name = list(filter(lambda x: x.startswith("test_one_"), cached_tables))[0] + + assert "test_one" in SPARK_STEP_BACKEND.entity_cache_tracker + assert SPARK_STEP_BACKEND.entity_cache_tracker["test_one"] == test_one_temp_name + assert sorted([rw.asDict(True) for rw in em.entities["test_one"].collect()], key=lambda x: x.get("test_date")) == td1 + assert sorted([rw.asDict(True) for rw in spark.table(test_one_temp_name).collect()], key=lambda x: x.get("test_date")) == td1 + assert "test_two" not in SPARK_STEP_BACKEND.entity_cache_tracker + + SPARK_STEP_BACKEND.cache_entity("test_two", em.entities) + SPARK_STEP_BACKEND._remove_cached_artifact("test_one") + + cached_tables = list(rw.tableName for rw in spark.sql("SHOW TABLES").collect()) + test_two_temp_name = list(filter(lambda x: x.startswith("test_two_"), cached_tables))[0] + + assert "test_two" in SPARK_STEP_BACKEND.entity_cache_tracker + assert SPARK_STEP_BACKEND.entity_cache_tracker["test_two"] in cached_tables + assert sorted([rw.asDict(True) for rw in em.entities["test_two"].collect()], key=lambda x: x.get("farewell")) == td2 + assert sorted([rw.asDict(True) for rw in spark.table(test_two_temp_name).collect()], key=lambda x: x.get("farewell")) == td2 + assert "test_one" not in SPARK_STEP_BACKEND.entity_cache_tracker + + SPARK_STEP_BACKEND.clear_entity_cache() + + cached_tables = list(rw.tableName for rw in spark.sql("SHOW TABLES").collect()) + + assert not SPARK_STEP_BACKEND.entity_cache_tracker + assert not any(tbl.startswith("test_one_") for tbl in cached_tables) + assert not any(tbl.startswith("test_two_") for tbl in cached_tables) + +def test_cache_management_with_update(): + spark = SPARK_STEP_BACKEND._spark_session + td = [ + {"idx": 1, "lots_of_nums": [1,2,3], "nested_field": {"nested_str": "test1", "nested_timestamp": datetime.datetime(2024,3,5,1,2,3)}}, + {"idx": 2, "lots_of_nums": [4,5,6,7,8], "nested_field": {"nested_str": "test2", "nested_timestamp": datetime.datetime(2022,8,12,10,12,14)}}, + ] + td_schema = StructType( + [ + StructField("idx", LongType()), + StructField("lots_of_nums", ArrayType(LongType())), + StructField("nested_field", StructType([StructField("nested_str", StringType()), StructField("nested_timestamp", TimestampType())])) + ] + ) + em = EntityManager({}) + em.entities["test_df"] = spark.createDataFrame(td, schema=td_schema) + + SPARK_STEP_BACKEND.cache_entity("test_df", em.entities) + + cached_tables = (rw.tableName for rw in spark.sql("SHOW TABLES").collect()) + + first_temp_name = list(filter(lambda x: x.startswith("test_df_"), cached_tables))[0] + + assert "test_df" in SPARK_STEP_BACKEND.entity_cache_tracker + assert SPARK_STEP_BACKEND.entity_cache_tracker["test_df"] == first_temp_name + assert sorted([rw.asDict(True) for rw in em.entities["test_df"].collect()], key=lambda x: x.get("idx")) == td + + extra_data = [{"idx": 3, "lots_of_nums": [9], "nested_field": {"nested_str": "test3", "nested_timestamp": datetime.datetime(2024,1,9,3,2,1)}}] + em.entities["test_df"] = em.entities["test_df"].union(spark.createDataFrame(extra_data, schema=td_schema)) + SPARK_STEP_BACKEND.cache_entity("test_df", em.entities) + + cached_tables = (rw.tableName for rw in spark.sql("SHOW TABLES").collect()) + tables_of_interest = list(filter(lambda x: x.startswith("test_df_"), cached_tables)) + assert len(tables_of_interest) == 1 + assert first_temp_name not in tables_of_interest + second_temp_name = tables_of_interest[0] + assert sorted([rw.asDict(True) for rw in em.entities["test_df"].collect()], key=lambda x: x.get("idx")) == td + extra_data + assert sorted([rw.asDict(True) for rw in spark.table(second_temp_name).collect()], key=lambda x: x.get("idx")) == td + extra_data + + SPARK_STEP_BACKEND.clear_entity_cache() + + cached_tables = list(rw.tableName for rw in spark.sql("SHOW TABLES").collect()) + + assert not SPARK_STEP_BACKEND.entity_cache_tracker + assert not any(tbl.startswith("test_df_") for tbl in cached_tables) diff --git a/tests/test_core_engine/test_backends/test_implementations/test_spark/test_spark_helpers.py b/tests/test_core_engine/test_backends/test_implementations/test_spark/test_spark_helpers.py index e7a37eb1..41c9e93e 100644 --- a/tests/test_core_engine/test_backends/test_implementations/test_spark/test_spark_helpers.py +++ b/tests/test_core_engine/test_backends/test_implementations/test_spark/test_spark_helpers.py @@ -63,7 +63,7 @@ def example_data_contract_error_codes(spark: SparkSession): ]) error_contract_messages = [ { - "Entity": "test_entity", + "OriginalEntity": "test_entity", "Key": "", "FailureType": "record", "Status": "error", @@ -77,7 +77,7 @@ def example_data_contract_error_codes(spark: SparkSession): "Category": "Bad value" }, { - "Entity": "test_entity", + "OriginalEntity": "test_entity", "Key": "", "FailureType": "record", "Status": "error", diff --git a/tests/test_core_engine/test_backends/test_readers/test_csv.py b/tests/test_core_engine/test_backends/test_readers/test_csv.py index f2e2c8df..413b6145 100644 --- a/tests/test_core_engine/test_backends/test_readers/test_csv.py +++ b/tests/test_core_engine/test_backends/test_readers/test_csv.py @@ -12,6 +12,7 @@ from pydantic import BaseModel from dve.core_engine.backends.exceptions import ( + CriticalMessageBearingError, EmptyFileError, FieldCountMismatch, MessageBearingError, @@ -287,7 +288,7 @@ def test_base_csv_reader_with_additional_fields( ): """Test that message bearing error raised when additional fields provided""" reader = CSVFileReader(field_check=True) - with pytest.raises(MessageBearingError) as exc_info: + with pytest.raises(CriticalMessageBearingError) as exc_info: list(reader.read_to_py_iterator( planet_additional_field_location, "test", @@ -295,7 +296,7 @@ def test_base_csv_reader_with_additional_fields( get_all_model_fields([Planets]) )) - error_msg = exc_info.value.messages[0] + error_msg = exc_info.value.message assert error_msg.record["test"] == "additional fields: add_field1, add_field2;" assert "missing_fields" not in error_msg.record["test"] @@ -306,7 +307,7 @@ def test_base_csv_reader_with_missing_fields( """Test that message bearing error raised when fields are missing from the expected schema""" reader = CSVFileReader(field_check=True) - with pytest.raises(MessageBearingError) as exc_info: + with pytest.raises(CriticalMessageBearingError) as exc_info: list(reader.read_to_py_iterator( planet_location, "test", @@ -314,6 +315,6 @@ def test_base_csv_reader_with_missing_fields( get_all_model_fields([PlanetsWithExtra]) )) - error_msg = exc_info.value.messages[0] + error_msg = exc_info.value.message assert "additional_fields" not in error_msg.record["test"] assert error_msg.record["test"] == "missing fields: random_null;" diff --git a/tests/test_core_engine/test_hierarchy.py b/tests/test_core_engine/test_hierarchy.py new file mode 100644 index 00000000..bbce74f0 --- /dev/null +++ b/tests/test_core_engine/test_hierarchy.py @@ -0,0 +1,396 @@ +import json +import pytest +from tempfile import NamedTemporaryFile +from dve.core_engine.configuration.v1 import V1EngineConfig +from dve.core_engine.configuration.v1.hierarchy import EntityHierarchy + +CONFIG_WITHOUT_LINKAGE = """{ + "contract": { + "schemas": {}, + "datasets": { + "animals": { + "fields": { + "name": "str", + "height": "float", + "weight": "float", + "region": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "animal", + "root_tag": "animals" + } + } + }, + "mandatory_fields": [ + "name" + ] + } + } + }, + "transformations": { + "filters": [ + { + "entity": "animals", + "name": "check_valid_region", + "expression": "lower(region) in ('africa', 'asia')", + "error_code": "ANE01", + "failure_message": "Record rejected - `{{ region }}` is not in a valid region." + }, + { + "entity": "animals", + "name": "check_for_pets", + "expression": "lower(name) != 'human'", + "error_code": "ANE02", + "failure_message": "Submission Rejected - 'Human' is not a valid animal.", + "failure_type": "submission" + }, + { + "entity": "animals", + "name": "check_valid_weight", + "expression": "weight > 0", + "error_code": "ANE03", + "failure_message": "Warning - `{{ weight }}` is below zero.", + "is_informational": true + } + ] + } +}""" + +CONFIG_WITH_LINKAGE = """{ + "contract": { + "schemas": {}, + "datasets": { + "ds_001": { + "fields": { + "ds_001_id": "str", + "patient_id": "str", + "address": "str", + "name": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "001", + "root_tag": "header" + } + } + }, + "mandatory_fields": [ + "ds_001_id", + "patient_id", + "address", + "name" + ] + }, + "ds_002": { + "fields": { + "ds_002_id": "str", + "gp_name": "str", + "gp_address": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "002", + "root_tag": "header" + } + } + }, + "mandatory_fields": [ + "ds_002_id", + "gp_name", + "gp_address" + ] + }, + "ds_003": { + "fields": { + "ds_003_id": "str", + "ds_001_id": "str", + "total_income": "int" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "003", + "root_tag": "header" + } + } + }, + "mandatory_fields": [ + "ds_003_id", + "ds_001_id" + ] + }, + "ds_101": { + "fields": { + "ds_001_id": "str", + "referral_id": "int", + "consultant_name": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "101", + "root_tag": "header" + } + } + }, + "mandatory_fields": [ + "referral_id", + "ds_001_id" + ] + }, + "ds_201": { + "fields": { + "ds_201_id": "str", + "ds_101_id": "str", + "contact_date": "date" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "201", + "root_tag": "header" + } + } + }, + "mandatory_fields": [ + "ds_101_id", + "ds_201_id", + "contact_date" + ] + }, + "ds_202": { + "fields": { + "ds_202_id": "str", + "ds_201_id": "str", + "contact_name": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "202", + "root_tag": "header" + } + } + }, + "mandatory_fields": [ + "ds_202_id", + "ds_201_id" + ] + } + } + }, + "transformations": { + "filters": [ + { + "entity": "001", + "name": "check_name", + "expression": "len(name) > 2", + "error_code": "CHECK1", + "failure_message": "Record rejected - `{{ name }}` is not valid." + } + ] + }, + "entity_relationships": { + "ds_003": { + "parent_entity": "ds_001", + "join_fields": {"ds_001_id": "ds_001_id"}, + "mandatory": false, + "missing_parent_id_error_code": "DS003NoParent", + "missing_parent_id_error_message": "record removed as no parent" + }, + "ds_101": { + "parent_entity": "ds_001", + "join_fields": {"ds_001_id": "ds_001_id"}, + "mandatory_entity": true, + "no_valid_records_error_code": "DS101NOVALIDRECS", + "no_valid_records_error_message": "{{ ds_001_id }} removed as no valid ds_101 records", + "missing_parent_id_error_code": "DS101NoParent", + "missing_parent_id_error_message": "record removed as no parent" + }, + "ds_201": { + "parent_entity": "ds_101", + "join_fields": {"referral_id": "ds_101_id"}, + "mandatory": false, + "missing_parent_id_error_code": "DS201NoParent", + "missing_parent_id_error_message": "record removed as no parent" + }, + "ds_202": { + "parent_entity": "ds_201", + "join_fields": {"ds_201_id": "ds_201_id"}, + "mandatory": true + } + } +}""" + +def test_no_linkage_config_load(): + config = V1EngineConfig(location="", + **json.loads(CONFIG_WITHOUT_LINKAGE)) + assert len(config.contract.datasets) == 1 + hierarchy = EntityHierarchy.from_engine_config(config) + assert len(hierarchy.entity_trees) == 1 + assert not hierarchy.entity_trees.get("animals").children + + +def test_linkage_config_load(): + config = V1EngineConfig(location="", + **json.loads(CONFIG_WITH_LINKAGE)) + assert len(config.contract.datasets) == 6 + with NamedTemporaryFile("w") as tmp: + tmp.write(CONFIG_WITH_LINKAGE) + tmp.flush() + hierarchy = EntityHierarchy.from_dischema(tmp.name) + assert len(hierarchy.entity_trees) == 2 + assert not hierarchy.entity_trees.get("ds_002").children + assert len(hierarchy.entity_trees.get("ds_001").get_descendents()) == 4 + children_001 = sorted(hierarchy.entity_trees.get("ds_001").children, key=lambda x: x.entity_name) + dict_rep_001 = hierarchy.entity_trees.get("ds_001").as_dict() + assert len(children_001) == 2 + assert children_001[0].entity_name == "ds_003" + assert not children_001[0].children + assert children_001[1].entity_name == "ds_101" + assert dict_rep_001 == json.loads(""" +{ + "ds_001": { + "parent_entity": null, + "join_fields": {}, + "mandatory": false, + "no_valid_records_error_code": "NoValidRecords", + "no_valid_records_error_message": "parent record removed as no valid child records", + "missing_parent_id_error_code": null, + "missing_parent_id_error_message": null, + "empty_entity_error_code": null, + "empty_entity_error_message": null, + "children": { + "ds_003": { + "parent_entity": "ds_001", + "join_fields": { + "ds_001_id": "ds_001_id" + }, + "mandatory": false, + "no_valid_records_error_code": "NoValidRecords", + "no_valid_records_error_message": "parent record removed as no valid child records", + "missing_parent_id_error_code": "DS003NoParent", + "missing_parent_id_error_message": "record removed as no parent", + "empty_entity_error_code": null, + "empty_entity_error_message": null, + "children": {} + }, + "ds_101": { + "parent_entity": "ds_001", + "join_fields": { + "ds_001_id": "ds_001_id" + }, + "mandatory": false, + "no_valid_records_error_code": "DS101NOVALIDRECS", + "no_valid_records_error_message": "{{ ds_001_id }} removed as no valid ds_101 records", + "missing_parent_id_error_code": "DS101NoParent", + "missing_parent_id_error_message": "record removed as no parent", + "empty_entity_error_code": null, + "empty_entity_error_message": null, + "children": { + "ds_201": { + "parent_entity": "ds_101", + "join_fields": { + "referral_id": "ds_101_id" + }, + "mandatory": false, + "no_valid_records_error_code": "NoValidRecords", + "no_valid_records_error_message": "parent record removed as no valid child records", + "missing_parent_id_error_code": "DS201NoParent", + "missing_parent_id_error_message": "record removed as no parent", + "empty_entity_error_code": null, + "empty_entity_error_message": null, + "children": { + "ds_202": { + "parent_entity": "ds_201", + "join_fields": { + "ds_201_id": "ds_201_id" + }, + "mandatory": true, + "no_valid_records_error_code": "NoValidRecords", + "no_valid_records_error_message": "parent record removed as no valid child records", + "missing_parent_id_error_code": "MissingParentRecord", + "missing_parent_id_error_message": "Records removed due to no valid parent record", + "empty_entity_error_code": "EmptyEntity", + "empty_entity_error_message": "no valid records remaining", + "children": {} + } + } + } + } + } + } + } + }""" + ) + + dict_rep_101 = dict_rep_001["ds_001"]["children"]["ds_101"] + children_101 = children_001[1].children + assert len(children_101) == 1 + assert children_101[0].entity_name == "ds_201" + assert children_101[0].children[0].entity_name == "ds_202" + assert not children_101[0].children[0].children + assert dict_rep_101 == json.loads(""" + { "parent_entity": "ds_001", + "join_fields": { + "ds_001_id": "ds_001_id" + }, + "mandatory": false, + "no_valid_records_error_code": "DS101NOVALIDRECS", + "no_valid_records_error_message": "{{ ds_001_id }} removed as no valid ds_101 records", + "missing_parent_id_error_code": "DS101NoParent", + "missing_parent_id_error_message": "record removed as no parent", + "empty_entity_error_code": null, + "empty_entity_error_message": null, + "children": { + "ds_201": { + "parent_entity": "ds_101", + "join_fields": { + "referral_id": "ds_101_id" + }, + "mandatory": false, + "no_valid_records_error_code": "NoValidRecords", + "no_valid_records_error_message": "parent record removed as no valid child records", + "missing_parent_id_error_code": "DS201NoParent", + "missing_parent_id_error_message": "record removed as no parent", + "empty_entity_error_code": null, + "empty_entity_error_message": null, + "children": { + "ds_202": { + "parent_entity": "ds_201", + "join_fields": { + "ds_201_id": "ds_201_id" + }, + "mandatory": true, + "no_valid_records_error_code": "NoValidRecords", + "no_valid_records_error_message": "parent record removed as no valid child records", + "missing_parent_id_error_code": "MissingParentRecord", + "missing_parent_id_error_message": "Records removed due to no valid parent record", + "empty_entity_error_code": "EmptyEntity", + "empty_entity_error_message": "no valid records remaining", + "children": {} + } + } + } + } + }""") + + +def test_get_all_mandatory_nodes(): + with NamedTemporaryFile("w") as tmp: + tmp.write(CONFIG_WITH_LINKAGE) + tmp.flush() + hierarchy = EntityHierarchy.from_dischema(tmp.name) + + assert len(hierarchy.get_all_mandatory_nodes()) == 1 diff --git a/tests/test_pipeline/pipeline_helpers.py b/tests/test_pipeline/pipeline_helpers.py index b13bef38..efed6e20 100644 --- a/tests/test_pipeline/pipeline_helpers.py +++ b/tests/test_pipeline/pipeline_helpers.py @@ -372,7 +372,7 @@ def error_data_after_business_rules() -> Iterator[Tuple[SubmissionInfo, str]]: error_data = json.loads( """[ { - "Entity": "planets", + "ReportingEntity": "planets", "Key": "", "FailureType": "record", "Status": "error", @@ -386,7 +386,7 @@ def error_data_after_business_rules() -> Iterator[Tuple[SubmissionInfo, str]]: "RecordIndex": "1" }, { - "Entity": "planets", + "ReportingEntity": "planets", "Key": "", "FailureType": "record", "Status": "error", diff --git a/tests/test_pipeline/test_foundry_ddb_pipeline.py b/tests/test_pipeline/test_foundry_ddb_pipeline.py index 9b7b60d6..6edc06bc 100644 --- a/tests/test_pipeline/test_foundry_ddb_pipeline.py +++ b/tests/test_pipeline/test_foundry_ddb_pipeline.py @@ -102,7 +102,7 @@ def test_foundry_runner_validation_success(movies_test_files, temp_ddb_conn): ) output_loc, report_uri, audit_files = dve_pipeline.run_pipeline(sub_info) assert fh.get_resource_exists(report_uri) - assert len(list(fh.iter_prefix(output_loc))) == 2 + assert len(list(fh.iter_prefix(output_loc))) == 1 assert len(list(fh.iter_prefix(audit_files))) == 3 def test_foundry_runner_error(planet_test_files, temp_ddb_conn): @@ -158,7 +158,7 @@ def test_foundry_runner_error(planet_test_files, temp_ddb_conn): .select(pl.col("step_name"), pl.col("error_location"), pl.col("error_message")) ) actual_error_df = ( - pl.read_json(perror_path, schema=perror_schema) + pl.read_ndjson(perror_path, schema=perror_schema) .select(pl.col("step_name"), pl.col("error_location"), pl.col("error_message")) ) assert actual_error_df.equals(expected_error_df) @@ -197,7 +197,7 @@ def test_foundry_runner_with_submitted_files_path(movies_test_files, temp_ddb_co assert Path(processing_folder, sub_id, sub_info.file_name_with_ext).exists() assert fh.get_resource_exists(report_uri) - assert len(list(fh.iter_prefix(output_loc))) == 2 + assert len(list(fh.iter_prefix(output_loc))) == 1 assert len(list(fh.iter_prefix(audit_files))) == 3 diff --git a/tests/test_pipeline/test_pipeline_utils.py b/tests/test_pipeline/test_pipeline_utils.py new file mode 100644 index 00000000..fc283068 --- /dev/null +++ b/tests/test_pipeline/test_pipeline_utils.py @@ -0,0 +1,36 @@ +from dve.core_engine.backends.exceptions import MessageBearingError +from dve.core_engine.configuration.v1 import _ModelConfig, _ReaderConfig +from dve.pipeline.utils import load_reader + +import pytest + + +class TestLoadReader: + test_model_config = _ModelConfig( + fields={"test": "str"}, + reporting_fields=["test"], + key_field="test", + reader_config={ + ".csv": _ReaderConfig(reader="TestCsvReader"), + } + ) + + def test_invalid_load_reader_with_file_ext(self): + with pytest.raises(MessageBearingError) as exc_info: + load_reader( + {"test": self.test_model_config}, + "test_model", + "jpeg" + ) + + assert exc_info.value.messages[0].error_message == "The supplied file extension `jpeg` is not a supported file format for test_model." + + def test_invalid_load_reader_missing_file_ext(self): + with pytest.raises(MessageBearingError) as exc_info: + load_reader( + {"test": self.test_model_config}, + "test_model", + "" + ) + + assert exc_info.value.messages[0].error_message == "No supplied file extension. Unable to parse file without a file extension." diff --git a/tests/test_pipeline/test_spark_pipeline.py b/tests/test_pipeline/test_spark_pipeline.py index dd28e266..642a7ead 100644 --- a/tests/test_pipeline/test_spark_pipeline.py +++ b/tests/test_pipeline/test_spark_pipeline.py @@ -162,7 +162,7 @@ def test_apply_data_contract_failed( # pylint: disable=redefined-outer-name expected_errors = [ { - "Entity": "planets", + "ReportingEntity": "planets", "Key": "", "FailureType": "record", "Status": "error", @@ -176,7 +176,7 @@ def test_apply_data_contract_failed( # pylint: disable=redefined-outer-name "Category": "Bad value", }, { - "Entity": "planets", + "ReportingEntity": "planets", "Key": "", "FailureType": "record", "Status": "error", @@ -190,7 +190,7 @@ def test_apply_data_contract_failed( # pylint: disable=redefined-outer-name "Category": "Bad value", }, { - "Entity": "planets", + "ReportingEntity": "planets", "Key": "", "FailureType": "record", "Status": "error", @@ -274,12 +274,6 @@ def test_apply_business_rules_success( assert largest_satellites_entity_path.exists() assert spark.read.parquet(str(largest_satellites_entity_path)).count() == 1 - og_planets_entity_path = Path( - Path(processed_file_path), sub_info.submission_id, "business_rules", "Originalplanets" - ) - assert og_planets_entity_path.exists() - assert spark.read.parquet(str(og_planets_entity_path)).count() == 1 - def test_apply_business_rules_with_data_errors( # pylint: disable=redefined-outer-name spark: SparkSession, @@ -317,16 +311,12 @@ def test_apply_business_rules_with_data_errors( # pylint: disable=redefined-out assert largest_satellites_entity_path.exists() assert spark.read.parquet(str(largest_satellites_entity_path)).count() == 1 - og_planets_entity_path = br_path / "Originalplanets" - assert og_planets_entity_path.exists() - assert spark.read.parquet(str(og_planets_entity_path)).count() == 1 - errors_path = Path(br_path.parent, "errors", "business_rules_errors.jsonl") assert errors_path.exists() expected_errors = [ { - "Entity": "planets", + "ReportingEntity": "planets", "Key": "", "FailureType": "record", "Status": "error", @@ -340,7 +330,7 @@ def test_apply_business_rules_with_data_errors( # pylint: disable=redefined-out "RecordIndex": "1" }, { - "Entity": "planets", + "ReportingEntity": "planets", "Key": "", "FailureType": "record", "Status": "error", diff --git a/tests/testdata/books/nested_books.dischema.json b/tests/testdata/books/nested_books.dischema.json index 52add54f..9b2e8afb 100644 --- a/tests/testdata/books/nested_books.dischema.json +++ b/tests/testdata/books/nested_books.dischema.json @@ -34,8 +34,8 @@ "record_tag": "bookstore", "n_records_to_read": 1, "xsd_location": "nested_books.xsd", - "xsd_error_code": "TESTXSDERROR", - "xsd_error_message": "the xml is poorly structured" + "ft_error_code": "TESTXSDERROR", + "ft_error_message": "the xml is poorly structured" } } } diff --git a/tests/testdata/books/nested_books_ddb.dischema.json b/tests/testdata/books/nested_books_ddb.dischema.json index d53c4165..f697ffa8 100644 --- a/tests/testdata/books/nested_books_ddb.dischema.json +++ b/tests/testdata/books/nested_books_ddb.dischema.json @@ -34,8 +34,8 @@ "record_tag": "bookstore", "n_records_to_read": 1, "xsd_location": "nested_books.xsd", - "xsd_error_code": "TESTXSDERROR", - "xsd_error_message": "the xml is poorly structured" + "ft_error_code": "TESTXSDERROR", + "ft_error_message": "the xml is poorly structured" } } } diff --git a/tests/testdata/flights/flights.dischema.json b/tests/testdata/flights/flights.dischema.json new file mode 100644 index 00000000..a0342cc0 --- /dev/null +++ b/tests/testdata/flights/flights.dischema.json @@ -0,0 +1,212 @@ +{ + "contract": { + "schemas": { + "passengers": { + "fields": { + "flight_id": "int", + "passenger_id": "int", + "passenger_name": "str" + } + } + }, + "error_details": "flights_data_contract_error_details.json", + "datasets": { + "country": { + "fields": { + "country_id": "int", + "country_name": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "country", + "root_tag": "country" + } + } + }, + "key_field": "country_id", + "mandatory_fields": [ + "country_id", + "country_name" + ] + }, + "airport": { + "fields": { + "country_id": "int", + "airport_id": "int", + "airport_name": "str", + "postcode": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "airport", + "root_tag": "country" + } + } + }, + "key_field": "airport_id", + "mandatory_fields": [ + "airport_id" + ] + }, + "staff": { + "fields": { + "airport_id": "int", + "staff_id": "int", + "staff_name": "str", + "role": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "staff_member", + "root_tag": "country" + } + } + }, + "key_field": "staff_id" + }, + "flights": { + "fields": { + "airport_id": "int", + "flight_id": "int", + "destination": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "flight", + "root_tag": "country" + } + } + }, + "key_field": "flight_id" + }, + "passengers": { + "fields": { + "flight_id": "int", + "passenger_id": "int", + "passenger_name": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "passenger", + "root_tag": "country" + } + } + }, + "key_field": "passenger_id" + } + } + }, + "transformations": { + "parameters": { + "entity": "country" + }, + "filters": [ + { + "entity": "flights", + "name": "flight_missing_id", + "expression": "flight_id IS NOT NULL", + "failure_type": "record", + "failure_message": "Record Rejected - Flight is missing an id", + "reporting_field": "flight_id", + "reporting_entity": "flights", + "category": "Blank", + "error_code": "FlightIDMissing" + }, + { + "entity": "flights", + "name": "invalid_destination", + "expression": "lower(destination) IN ('paris', 'madrid', 'new york', 'amsterdam', 'rome', 'dubai', 'dublin', 'lisbon', 'toronto')", + "failure_type": "record", + "failure_message": "Record Rejected - {{ destination }} is not a valid destination", + "reporting_field": "flight_id", + "reporting_entity": "flights", + "category": "Bad value", + "error_code": "InvalidFlightDestination" + }, + { + "entity": "passengers", + "name": "passenger_name_is_null", + "expression": "passenger_name IS NOT NULL", + "failure_type": "record", + "failure_message": "Record Rejected - Passenger Name is missing", + "reporting_field": "passenger_name", + "reporting_entity": "passengers", + "category": "Blank", + "error_code": "PassengerNameMissing" + }, + { + "entity": "staff", + "name": "staff_id_is_null", + "expression": "staff_id IS NOT NULL", + "failure_type": "record", + "failure_message": "Record Rejected - staff_id is missing", + "reporting_field": "passenger_name", + "reporting_entity": "passengers", + "category": "Blank", + "error_code": "StaffIDMissing" + } + ] + }, + "entity_relationships": { + "country": { + "is_root_entity": true, + "mandatory": true, + "empty_entity_error_code": "NoValidCountries", + "empty_entity_error_message": "File Rejected - There are no valid country records" + }, + "airport": { + "parent_entity": "country", + "join_fields": { + "country_id": "country_id" + }, + "mandatory": true, + "missing_parent_id_error_code": "AirportHasNoCountry", + "missing_parent_id_error_message": "Record rejected - No valid country id found for airport", + "no_valid_records_error_code": "CountryHasNoAirport", + "no_valid_records_error_message": "Group rejected - Unable to find any valid airports", + "empty_entity_error_code": "NoValidAirports", + "empty_entity_error_message": "File Rejected - There are no valid airport records" + }, + "staff": { + "parent_entity": "airport", + "join_fields": { + "airport_id": "airport_id" + }, + "mandatory": true, + "missing_parent_id_error_code": "StaffHasNoAirport", + "missing_parent_id_error_message": "Record rejected - No valid airport id found for staff. Airport ID = {{ airport_id }}, Staff ID = {{ staff_id }}", + "no_valid_records_error_code": "AirportHasNoStaff", + "no_valid_records_error_message": "Group rejected - Airport has no valid staff. Airport ID = {{ airport_id }}", + "empty_entity_error_code": "NoValidStaff", + "empty_entity_error_message": "File Rejected - There are no valid staff records" + }, + "flights": { + "parent_entity": "airport", + "join_fields": { + "airport_id": "airport_id" + }, + "mandatory": false, + "missing_parent_id_error_code": "FlightHasNoAirport", + "missing_parent_id_error_message": "Record Rejected - No valid airport found for flight" + }, + "passengers": { + "parent_entity": "flights", + "join_fields": { + "flight_id": "flight_id" + }, + "mandatory": false, + "missing_parent_id_error_code": "PassengerHasNoFlight", + "missing_parent_id_error_message": "Record rejected - No valid flight found for passenger" + } + } +} \ No newline at end of file diff --git a/tests/testdata/flights/flights_add_reader_checks.dischema.json b/tests/testdata/flights/flights_add_reader_checks.dischema.json new file mode 100644 index 00000000..5f8e63b5 --- /dev/null +++ b/tests/testdata/flights/flights_add_reader_checks.dischema.json @@ -0,0 +1,208 @@ +{ + "contract": { + "schemas": { + "passengers": { + "fields": { + "flight_id": "int", + "passenger_id": "int", + "passenger_name": "str" + } + } + }, + "error_details": "flights_data_contract_error_details.json", + "datasets": { + "country": { + "fields": { + "country_id": "int", + "country_name": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "country", + "root_tag": "country" + } + } + }, + "key_field": "country_id", + "mandatory_fields": [ + "country_id", + "country_name" + ] + }, + "airport": { + "fields": { + "country_id": "int", + "airport_id": "int", + "airport_name": "str", + "postcode": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "airport", + "root_tag": "country" + } + } + }, + "reader_additional_checks": { + "check_empty": { + "error_code": "AIRPORTEMPTY", + "error_message": "No airport records included in submission" + } + }, + "key_field": "airport_id", + "mandatory_fields": [ + "airport_id" + ] + }, + "staff": { + "fields": { + "airport_id": "int", + "staff_id": "int", + "staff_name": "str", + "role": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "staff_member", + "root_tag": "country" + } + } + }, + "key_field": "staff_id" + }, + "flights": { + "fields": { + "airport_id": "int", + "flight_id": "int", + "destination": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "flight", + "root_tag": "country" + } + } + }, + "key_field": "flight_id" + }, + "passengers": { + "fields": { + "flight_id": "int", + "passenger_id": "int", + "passenger_name": "str" + }, + "reader_config": { + ".xml": { + "reader": "DuckDBXMLStreamReader", + "kwargs": { + "record_tag": "passenger", + "root_tag": "country" + } + } + }, + "key_field": "passenger_id" + } + } + }, + "transformations": { + "parameters": { + "entity": "country" + }, + "filters": [ + { + "entity": "flights", + "name": "flight_missing_id", + "expression": "flight_id IS NOT NULL", + "failure_type": "record", + "failure_message": "Record Rejected - Flight is missing an id", + "reporting_field": "flight_id", + "reporting_entity": "flights", + "category": "Blank", + "error_code": "FlightIDMissing" + }, + { + "entity": "flights", + "name": "invalid_destination", + "expression": "lower(destination) IN ('paris', 'madrid', 'new york', 'amsterdam', 'rome', 'dubai', 'dublin', 'lisbon', 'toronto')", + "failure_type": "record", + "failure_message": "Record Rejected - {{ destination }} is not a valid destination", + "reporting_field": "flight_id", + "reporting_entity": "flights", + "category": "Bad value", + "error_code": "InvalidFlightDestination" + }, + { + "entity": "passengers", + "name": "passenger_name_is_null", + "expression": "passenger_name IS NOT NULL", + "failure_type": "record", + "failure_message": "Record Rejected - Passenger Name is missing", + "reporting_field": "passenger_name", + "reporting_entity": "passengers", + "category": "Blank", + "error_code": "PassengerNameMissing" + }, + { + "entity": "staff", + "name": "staff_id_is_null", + "expression": "staff_id IS NOT NULL", + "failure_type": "record", + "failure_message": "Record Rejected - staff_id is missing", + "reporting_field": "passenger_name", + "reporting_entity": "passengers", + "category": "Blank", + "error_code": "StaffIDMissing" + } + ] + }, + "entity_relationships": { + "airport": { + "parent_entity": "country", + "join_fields": { + "country_id": "country_id" + }, + "mandatory": true, + "missing_parent_id_error_code": "AirportHasNoCountry", + "missing_parent_id_error_message": "Record rejected - No valid country id found for airport", + "no_valid_records_error_code": "CountryHasNoAirport", + "no_valid_records_error_message": "Group rejected - Unable to find any valid airports" + }, + "staff": { + "parent_entity": "airport", + "join_fields": { + "airport_id": "airport_id" + }, + "mandatory": true, + "missing_parent_id_error_code": "StaffHasNoAirport", + "missing_parent_id_error_message": "Record rejected - No valid airport id found for staff. Airport ID = {{ airport_id }}, Staff ID = {{ staff_id }}", + "no_valid_records_error_code": "AirportHasNoStaff", + "no_valid_records_error_message": "Group rejected - Airport has no valid staff. Airport ID = {{ airport_id }}" + }, + "flights": { + "parent_entity": "airport", + "join_fields": { + "airport_id": "airport_id" + }, + "mandatory": false, + "missing_parent_id_error_code": "FlightHasNoAirport", + "missing_parent_id_error_message": "Record Rejected - No valid airport found for flight" + }, + "passengers": { + "parent_entity": "flights", + "join_fields": { + "flight_id": "flight_id" + }, + "mandatory": false, + "missing_parent_id_error_code": "PassengerHasNoFlight", + "missing_parent_id_error_message": "Record rejected - No valid flight found for passenger" + } + } +} \ No newline at end of file diff --git a/tests/testdata/flights/flights_data_contract_error_details.json b/tests/testdata/flights/flights_data_contract_error_details.json new file mode 100644 index 00000000..3cf4327d --- /dev/null +++ b/tests/testdata/flights/flights_data_contract_error_details.json @@ -0,0 +1,18 @@ +{ + "country": { + "country_id": { + "Blank": { + "error_code": "CountryIdIsMissing", + "error_message": "Record Rejected - Country is missing an id" + } + } + }, + "airport": { + "airport_id": { + "Blank": { + "error_code": "AirportIdIsMissing", + "error_message": "Record Rejected - Airport is missing an id" + } + } + } +} \ No newline at end of file diff --git a/tests/testdata/flights/flights_full_regression.xml b/tests/testdata/flights/flights_full_regression.xml new file mode 100644 index 00000000..1ed73cf3 --- /dev/null +++ b/tests/testdata/flights/flights_full_regression.xml @@ -0,0 +1,171 @@ + + + 1 + England + + + 1 + 1 + Heathrow + TW6 1EW + + + 1 + 1 + Paris + + + 1 + 1 + John + + + + + + + 1 + 1 + Marge + Manager + + + + + 1 + 2 + Manchester + M90 1QX + + + 2 + 2 + Venus + + + 2 + 2 + Jane + + + + + 2 + 3 + Rome + + + 3 + 3 + + + + + + + 2 + 2 + Thomas + Pilot + + + + + 1 + Birmingham + B26 3QJ + + + 3 + 4 + Amsterdam + + + 4 + 4 + Billy + + + + + + + 3 + 3 + Joanne + Security + + + + + 1 + 4 + Leeds & Bradford + LS19 7TU + + + 4 + 5 + Dubai + + + 5 + 5 + Terry + + + + + + + 4 + Rebecca + Ground Crew + + + 4 + Tim + Ground Crew + + + 4 + Julie + Pilot + + + + + 1 + 5 + Newcastle + NE13 8BZ + + + 5 + 6 + Toronto + + + 6 + 6 + Bob + + + + + + + 5 + 8 + Jasmine + Pilot + + + 5 + Rodger + Security + + + + + \ No newline at end of file diff --git a/tests/testdata/flights/flights_spark.dischema.json b/tests/testdata/flights/flights_spark.dischema.json new file mode 100644 index 00000000..b318f823 --- /dev/null +++ b/tests/testdata/flights/flights_spark.dischema.json @@ -0,0 +1,212 @@ +{ + "contract": { + "schemas": { + "passengers": { + "fields": { + "flight_id": "int", + "passenger_id": "int", + "passenger_name": "str" + } + } + }, + "error_details": "flights_data_contract_error_details.json", + "datasets": { + "country": { + "fields": { + "country_id": "int", + "country_name": "str" + }, + "reader_config": { + ".xml": { + "reader": "SparkXMLStreamReader", + "kwargs": { + "record_tag": "country", + "root_tag": "country" + } + } + }, + "key_field": "country_id", + "mandatory_fields": [ + "country_id", + "country_name" + ] + }, + "airport": { + "fields": { + "country_id": "int", + "airport_id": "int", + "airport_name": "str", + "postcode": "str" + }, + "reader_config": { + ".xml": { + "reader": "SparkXMLStreamReader", + "kwargs": { + "record_tag": "airport", + "root_tag": "country" + } + } + }, + "key_field": "airport_id", + "mandatory_fields": [ + "airport_id" + ] + }, + "staff": { + "fields": { + "airport_id": "int", + "staff_id": "int", + "staff_name": "str", + "role": "str" + }, + "reader_config": { + ".xml": { + "reader": "SparkXMLStreamReader", + "kwargs": { + "record_tag": "staff_member", + "root_tag": "country" + } + } + }, + "key_field": "staff_id" + }, + "flights": { + "fields": { + "airport_id": "int", + "flight_id": "int", + "destination": "str" + }, + "reader_config": { + ".xml": { + "reader": "SparkXMLStreamReader", + "kwargs": { + "record_tag": "flight", + "root_tag": "country" + } + } + }, + "key_field": "flight_id" + }, + "passengers": { + "fields": { + "flight_id": "int", + "passenger_id": "int", + "passenger_name": "str" + }, + "reader_config": { + ".xml": { + "reader": "SparkXMLStreamReader", + "kwargs": { + "record_tag": "passenger", + "root_tag": "country" + } + } + }, + "key_field": "passenger_id" + } + } + }, + "transformations": { + "parameters": { + "entity": "country" + }, + "filters": [ + { + "entity": "flights", + "name": "flight_missing_id", + "expression": "flight_id IS NOT NULL", + "failure_type": "record", + "failure_message": "Record Rejected - Flight is missing an id", + "reporting_field": "flight_id", + "reporting_entity": "flights", + "category": "Blank", + "error_code": "FlightIDMissing" + }, + { + "entity": "flights", + "name": "invalid_destination", + "expression": "lower(destination) IN ('paris', 'madrid', 'new york', 'amsterdam', 'rome', 'dubai', 'dublin', 'lisbon', 'toronto')", + "failure_type": "record", + "failure_message": "Record Rejected - {{ destination }} is not a valid destination", + "reporting_field": "flight_id", + "reporting_entity": "flights", + "category": "Bad value", + "error_code": "InvalidFlightDestination" + }, + { + "entity": "passengers", + "name": "passenger_name_is_null", + "expression": "passenger_name IS NOT NULL", + "failure_type": "record", + "failure_message": "Record Rejected - Passenger Name is missing", + "reporting_field": "passenger_name", + "reporting_entity": "passengers", + "category": "Blank", + "error_code": "PassengerNameMissing" + }, + { + "entity": "staff", + "name": "staff_id_is_null", + "expression": "staff_id IS NOT NULL", + "failure_type": "record", + "failure_message": "Record Rejected - staff_id is missing", + "reporting_field": "passenger_name", + "reporting_entity": "passengers", + "category": "Blank", + "error_code": "StaffIDMissing" + } + ] + }, + "entity_relationships": { + "country": { + "is_root_entity": true, + "mandatory": true, + "empty_entity_error_code": "NoValidCountries", + "empty_entity_error_message": "File Rejected - There are no valid country records" + }, + "airport": { + "parent_entity": "country", + "join_fields": { + "country_id": "country_id" + }, + "mandatory": true, + "missing_parent_id_error_code": "AirportHasNoCountry", + "missing_parent_id_error_message": "Record rejected - No valid country id found for airport", + "no_valid_records_error_code": "CountryHasNoAirport", + "no_valid_records_error_message": "Group rejected - Unable to find any valid airports", + "empty_entity_error_code": "NoValidAirports", + "empty_entity_error_message": "File Rejected - There are no valid airport records" + }, + "staff": { + "parent_entity": "airport", + "join_fields": { + "airport_id": "airport_id" + }, + "mandatory": true, + "missing_parent_id_error_code": "StaffHasNoAirport", + "missing_parent_id_error_message": "Record rejected - No valid airport id found for staff. Airport ID = {{ airport_id }}, Staff ID = {{ staff_id }}", + "no_valid_records_error_code": "AirportHasNoStaff", + "no_valid_records_error_message": "Group rejected - Airport has no valid staff. Airport ID = {{ airport_id }}", + "empty_entity_error_code": "NoValidStaff", + "empty_entity_error_message": "File Rejected - There are no valid staff records" + }, + "flights": { + "parent_entity": "airport", + "join_fields": { + "airport_id": "airport_id" + }, + "mandatory": false, + "missing_parent_id_error_code": "FlightHasNoAirport", + "missing_parent_id_error_message": "Record Rejected - No valid airport found for flight" + }, + "passengers": { + "parent_entity": "flights", + "join_fields": { + "flight_id": "flight_id" + }, + "mandatory": false, + "missing_parent_id_error_code": "PassengerHasNoFlight", + "missing_parent_id_error_message": "Record rejected - No valid flight found for passenger" + } + } +} \ No newline at end of file diff --git a/tests/testdata/flights/missing_country_id.xml b/tests/testdata/flights/missing_country_id.xml new file mode 100644 index 00000000..14ccec27 --- /dev/null +++ b/tests/testdata/flights/missing_country_id.xml @@ -0,0 +1,322 @@ + + + England + + + 1 + 1 + Heathrow + TW6 1EW + + + 1 + 1 + Paris + + + 1 + 1 + John + + + 1 + 2 + Jane + + + 1 + 3 + Peter + + + + + 1 + 2 + Madrid + + + 2 + 4 + Homer + + + 2 + 5 + Marge + + + + + 1 + 3 + New York + + + 3 + 6 + Lisa + + + 3 + 7 + Bart + + + 3 + 8 + Maggie + + + + + + + 1 + 1 + Alice + Manager + + + 1 + 2 + Bob + Pilot + + + 1 + 3 + Charlie + Ground Crew + + + 1 + 4 + Diana + Security + + + + + 1 + 2 + Gatwick + RH6 0NP + + + 2 + 4 + Amsterdam + + + 4 + 9 + Oliver + + + 4 + 10 + Emily + + + + + 2 + 5 + Rome + + + 5 + 11 + George + + + 5 + 12 + Charlotte + + + 5 + 13 + Harry + + + + + + 2 + 6 + Dubai + + + 6 + 14 + William + + + 6 + 15 + Amelia + + + + + + + 2 + 5 + Edward + Manager + + + 2 + 6 + Fiona + Air Traffic Controller + + + 2 + 7 + Graham + Ground Crew + + + 2 + 8 + Hannah + Security + + + 2 + 9 + Ian + Engineer + + + + + 1 + 3 + Manchester + M90 1QX + + + 3 + 7 + Dublin + + + 7 + 16 + Jack + + + 7 + 17 + Isla + + + + + 3 + 8 + Lisbon + + + 8 + 18 + Thomas + + + 8 + 19 + Grace + + + 8 + 20 + Jacob + + + + + 3 + 9 + Toronto + + + 9 + 21 + Leo + + + 9 + 22 + Sophie + + + + + 3 + 10 + New York + + + 10 + 23 + Daniel + + + 10 + 24 + Ella + + + 10 + 25 + Oscar + + + + + + + 3 + 10 + Kevin + Manager + + + 3 + 11 + Laura + Pilot + + + 3 + 12 + Michael + Air Traffic Controller + + + 3 + 13 + Natalie + Ground Crew + + + 3 + 14 + Oliver + Security + + + 3 + 15 + Paula + Engineer + + + + + diff --git a/tests/testdata/flights/missing_flight_id.xml b/tests/testdata/flights/missing_flight_id.xml new file mode 100644 index 00000000..fdd7cf7e --- /dev/null +++ b/tests/testdata/flights/missing_flight_id.xml @@ -0,0 +1,322 @@ + + + 1 + England + + + 1 + 1 + Heathrow + TW6 1EW + + + 1 + Paris + + + 1 + 1 + John + + + 1 + 2 + Jane + + + 1 + 3 + Peter + + + + + 1 + 2 + Madrid + + + 2 + 4 + Homer + + + 2 + 5 + Marge + + + + + 1 + 3 + New York + + + 3 + 6 + Lisa + + + 3 + 7 + Bart + + + 3 + 8 + Maggie + + + + + + + 1 + 1 + Alice + Manager + + + 1 + 2 + Bob + Pilot + + + 1 + 3 + Charlie + Ground Crew + + + 1 + 4 + Diana + Security + + + + + 1 + 2 + Gatwick + RH6 0NP + + + 2 + 4 + Amsterdam + + + 4 + 9 + Oliver + + + 4 + 10 + Emily + + + + + 2 + 5 + Rome + + + 5 + 11 + George + + + 5 + 12 + Charlotte + + + 5 + 13 + Harry + + + + + + 2 + 6 + Dubai + + + 6 + 14 + William + + + 6 + 15 + Amelia + + + + + + + 2 + 5 + Edward + Manager + + + 2 + 6 + Fiona + Air Traffic Controller + + + 2 + 7 + Graham + Ground Crew + + + 2 + 8 + Hannah + Security + + + 2 + 9 + Ian + Engineer + + + + + 1 + 3 + Manchester + M90 1QX + + + 3 + 7 + Dublin + + + 7 + 16 + Jack + + + 7 + 17 + Isla + + + + + 3 + 8 + Lisbon + + + 8 + 18 + Thomas + + + 8 + 19 + Grace + + + 8 + 20 + Jacob + + + + + 3 + 9 + Toronto + + + 9 + 21 + Leo + + + 9 + 22 + Sophie + + + + + 3 + 10 + New York + + + 10 + 23 + Daniel + + + 10 + 24 + Ella + + + 10 + 25 + Oscar + + + + + + + 3 + 10 + Kevin + Manager + + + 3 + 11 + Laura + Pilot + + + 3 + 12 + Michael + Air Traffic Controller + + + 3 + 13 + Natalie + Ground Crew + + + 3 + 14 + Oliver + Security + + + 3 + 15 + Paula + Engineer + + + + + diff --git a/tests/testdata/flights/multi_node_file_rejection.xml b/tests/testdata/flights/multi_node_file_rejection.xml new file mode 100644 index 00000000..466a35b9 --- /dev/null +++ b/tests/testdata/flights/multi_node_file_rejection.xml @@ -0,0 +1,92 @@ + + + 1 + England + + + 1 + 1 + Heathrow + TW6 1EW + + + 1 + 1 + Paris + + + 1 + 1 + John + + + + + + + 1 + 1 + Alice + Manager + + + 1 + Alice + Manager + + + + + 1 + 2 + Manchester + M90 1QX + + + 2 + 2 + Dublin + + + 2 + 2 + Jack + + + + + + + 3 + Kevin + Manager + + + 3 + Laura + Pilot + + + 3 + Michael + Air Traffic Controller + + + 3 + Natalie + Ground Crew + + + 3 + Oliver + Security + + + 3 + Paula + Engineer + + + + + diff --git a/tests/testdata/flights/only_country_id.xml b/tests/testdata/flights/only_country_id.xml new file mode 100644 index 00000000..b0ebd076 --- /dev/null +++ b/tests/testdata/flights/only_country_id.xml @@ -0,0 +1,5 @@ + + + 1 + England + \ No newline at end of file diff --git a/tests/testdata/flights/perfect_flights.xml b/tests/testdata/flights/perfect_flights.xml new file mode 100644 index 00000000..88346dcc --- /dev/null +++ b/tests/testdata/flights/perfect_flights.xml @@ -0,0 +1,323 @@ + + + 1 + England + + + 1 + 1 + Heathrow + TW6 1EW + + + 1 + 1 + Paris + + + 1 + 1 + John + + + 1 + 2 + Jane + + + 1 + 3 + Peter + + + + + 1 + 2 + Madrid + + + 2 + 4 + Homer + + + 2 + 5 + Marge + + + + + 1 + 3 + New York + + + 3 + 6 + Lisa + + + 3 + 7 + Bart + + + 3 + 8 + Maggie + + + + + + + 1 + 1 + Alice + Manager + + + 1 + 2 + Bob + Pilot + + + 1 + 3 + Charlie + Ground Crew + + + 1 + 4 + Diana + Security + + + + + 1 + 2 + Gatwick + RH6 0NP + + + 2 + 4 + Amsterdam + + + 4 + 9 + Oliver + + + 4 + 10 + Emily + + + + + 2 + 5 + Rome + + + 5 + 11 + George + + + 5 + 12 + Charlotte + + + 5 + 13 + Harry + + + + + + 2 + 6 + Dubai + + + 6 + 14 + William + + + 6 + 15 + Amelia + + + + + + + 2 + 5 + Edward + Manager + + + 2 + 6 + Fiona + Air Traffic Controller + + + 2 + 7 + Graham + Ground Crew + + + 2 + 8 + Hannah + Security + + + 2 + 9 + Ian + Engineer + + + + + 1 + 3 + Manchester + M90 1QX + + + 3 + 7 + Dublin + + + 7 + 16 + Jack + + + 7 + 17 + Isla + + + + + 3 + 8 + Lisbon + + + 8 + 18 + Thomas + + + 8 + 19 + Grace + + + 8 + 20 + Jacob + + + + + 3 + 9 + Toronto + + + 9 + 21 + Leo + + + 9 + 22 + Sophie + + + + + 3 + 10 + New York + + + 10 + 23 + Daniel + + + 10 + 24 + Ella + + + 10 + 25 + Oscar + + + + + + + 3 + 10 + Kevin + Manager + + + 3 + 11 + Laura + Pilot + + + 3 + 12 + Michael + Air Traffic Controller + + + 3 + 13 + Natalie + Ground Crew + + + 3 + 14 + Oliver + Security + + + 3 + 15 + Paula + Engineer + + + + + diff --git a/tests/testdata/flights/singular_node_rejections.xml b/tests/testdata/flights/singular_node_rejections.xml new file mode 100644 index 00000000..134ee89b --- /dev/null +++ b/tests/testdata/flights/singular_node_rejections.xml @@ -0,0 +1,49 @@ + + + 1 + England + + + 1 + 1 + Heathrow + TW6 1EW + + + 1 + 1 + Mars + + + 1 + 1 + John + + + 1 + 2 + Jane + + + + + 1 + 2 + Venus + + + 2 + 3 + Homer + + + 2 + 4 + Marge + + + + + + + \ No newline at end of file diff --git a/tests/testdata/movies/movies_contract_error_details.json b/tests/testdata/movies/movies_contract_error_details.json index 260ee96f..e457555e 100644 --- a/tests/testdata/movies/movies_contract_error_details.json +++ b/tests/testdata/movies/movies_contract_error_details.json @@ -1,27 +1,29 @@ { - "title": { - "Blank": { - "error_code": "BLANKTITLE", - "error_message": "title should not be blank", - "error_level": "submission" - } - }, - "year": { - "Blank": { - "error_code": "BLANKYEAR", - "error_message": "year not provided", - "is_informational": true + "movies": { + "title": { + "Blank": { + "error_code": "BLANKTITLE", + "error_message": "title should not be blank", + "error_level": "submission" + } }, - "Bad value": { - "error_code": "DODGYYEAR", - "error_message": "year value ({{year}}) is invalid", - "reporting_entity": "movies_rename_test" - } - }, - "cast.date_joined": { - "Bad value": { - "error_code": "DODGYDATE", - "error_message": "date_joined value is not valid: {{__error_value}}" + "year": { + "Blank": { + "error_code": "BLANKYEAR", + "error_message": "year not provided", + "is_informational": true + }, + "Bad value": { + "error_code": "DODGYYEAR", + "error_message": "year value ({{year}}) is invalid", + "reporting_entity": "movies_rename_test" + } + }, + "cast.date_joined": { + "Bad value": { + "error_code": "DODGYDATE", + "error_message": "date_joined value is not valid: {{__error_value}}" + } } } } \ No newline at end of file diff --git a/zensical.toml b/zensical.toml index 064cb26a..0141b6df 100644 --- a/zensical.toml +++ b/zensical.toml @@ -25,6 +25,7 @@ nav = [ {"File Transformation" = "user_guidance/file_transformation.md"}, {"Data Contract" = "user_guidance/data_contract.md"}, {"Business Rules" = "user_guidance/business_rules.md"}, + {"Entity Relationships" = "user_guidance/entity_relationships.md"} ]}, {"Backend Implementations" = [ {"DuckDB" = "user_guidance/implementations/duckdb.md"}, @@ -56,6 +57,9 @@ nav = [ {"Refdata" = [ {"Refdata Types" = "advanced_guidance/package_documentation/refence_data_types.md"}, {"Refdata Loaders" = "advanced_guidance/package_documentation/refdata_loaders.md"}, + ]}, + {"Entity Relationships" = [ + {"Entity Hierarchy" = "advanced_guidance/package_documentation/entity_hierarchy.md"} ]} ]}, {"Feedback" = [ @@ -193,6 +197,9 @@ options.custom_icons = ["overrides/.icons"] auto_append = ["includes/jargon_and_acronyms.md"] [project.markdown_extensions.pymdownx.superfences] +custom_fences = [ + { name = "mermaid", class = "mermaid", format = "pymdownx.superfences.fence_code_format" }, +] [project.markdown_extensions.pymdownx.tabbed] alternate_style = true