Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- `datacontract lint --all-errors`: unknown fields are warnings instead of errors, unless a custom `--schema` rejects them

### Added
- `datacontract test`: a schema with the custom property `additionalProperties: false` fails on columns it does not declare
- `datacontract edit`: enable the editor's AI assistant via `DATACONTRACT_EDITOR_AI_*` environment variables (endpoint, API key, model, provider, auth header)
- `datacontract test`: check constraints and quality rules on nested properties on Athena and Trino servers

Expand Down
7 changes: 5 additions & 2 deletions datacontract/engines/checks/check_spec.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,7 @@ class MetricType(str, Enum):
FIELD_TYPE = "field_type"
FIELD_PHYSICAL_TYPE = "field_physical_type"
FIELD_NESTED_TYPE = "field_nested_type"
FIELD_NAMES = "field_names"
FRESHNESS = "freshness"
RETENTION = "retention"
CUSTOM_SQL = "custom_sql"
Expand All @@ -46,6 +47,7 @@ class MetricType(str, Enum):
MetricType.FIELD_TYPE,
MetricType.FIELD_PHYSICAL_TYPE,
MetricType.FIELD_NESTED_TYPE,
MetricType.FIELD_NAMES,
}


Expand Down Expand Up @@ -159,7 +161,8 @@ class CheckSpec:
expected_schema_property: Optional["SchemaProperty"] = None # FIELD_TYPE: structural comparison
expected_physical_type: Optional[str] = None # FIELD_PHYSICAL_TYPE: contract physicalType

columns: Optional[List[str]] = None # DUPLICATE_COUNT across multiple columns; MISSING_REFERENCE_COUNT keys
# DUPLICATE_COUNT across multiple columns; MISSING_REFERENCE_COUNT keys; FIELD_NAMES the declared fields
columns: Optional[List[str]] = None

referenced_model: Optional[str] = None # MISSING_REFERENCE_COUNT
referenced_columns: Optional[List[str]] = None # MISSING_REFERENCE_COUNT, pairwise with `columns`
Expand All @@ -170,7 +173,7 @@ class CheckSpec:

seconds: Optional[int] = None # FRESHNESS / RETENTION threshold in seconds

uses_raw_view: bool = False # FIELD_PRESENT against the duckdb {model}__raw__ view
uses_raw_view: bool = False # FIELD_PRESENT / FIELD_NAMES against the duckdb {model}__raw__ view

# Preset result/reason for checks that are not executed (UNSUPPORTED).
preset_result: Optional[str] = None
Expand Down
23 changes: 23 additions & 0 deletions datacontract/engines/checks/create_checks.py
Original file line number Diff line number Diff line change
Expand Up @@ -63,6 +63,15 @@ def _get_logical_type_option(prop: SchemaProperty, key: str):
return prop.logicalTypeOptions.get(key)


def _no_additional_properties(schema_object: SchemaObject) -> bool:
"""Whether the schema declares, as in JSON Schema, that the data has no fields beyond its properties.

ODCS has no field for this, so it is the schema's custom property `additionalProperties: false`.
"""
value = next((c.value for c in schema_object.customProperties or [] if c.property == "additionalProperties"), None)
return value is False or str(value).strip().lower() == "false"


def is_check_types(server: Optional[Server]) -> bool:
"""Type checks only make sense where the data source carries real types."""
if server is None:
Expand Down Expand Up @@ -625,6 +634,20 @@ def _to_schema_checks(
check.preset_result = "warning"
check.preset_reason = reason

if _no_additional_properties(schema_object):
checks.append(
CheckSpec(
key=f"{model}__no_additional_fields",
category="schema",
type="model_no_additional_fields",
name=f"Check that {model} has no fields the contract does not declare",
model=model,
metric=MetricType.FIELD_NAMES,
columns=[prop.physicalName or prop.name for prop in properties],
uses_raw_view=uses_raw_view,
)
)

if primary_key_is_composite:
primary_key_fields = [prop.physicalName or prop.name for prop in primary_key_props]
checks.append(
Expand Down
1 change: 1 addition & 0 deletions datacontract/engines/checks/dimensions.py
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,7 @@
# (the SAP HANA engine reports the missing table as a check of its own)
"model_exists": "conformity",
"field_is_present": "conformity",
"model_no_additional_fields": "conformity",
"field_type": "conformity",
"field_physical_type": "conformity",
"field_nested_type": "conformity",
Expand Down
24 changes: 24 additions & 0 deletions datacontract/engines/ibis/ibis_check_execute.py
Original file line number Diff line number Diff line change
Expand Up @@ -94,6 +94,8 @@ def _describe(spec: CheckSpec) -> str:
return f"nested_types({spec.field}) match the contract"
if spec.metric == MetricType.FIELD_PRESENT:
return f"present({spec.field})"
if spec.metric == MetricType.FIELD_NAMES:
return f"fields({spec.model}) in ({', '.join(spec.columns or [])})"
if spec.threshold is not None:
target = spec.field or spec.model
return f"{spec.metric.value}({target}) {spec.threshold.describe()}"
Expand Down Expand Up @@ -336,6 +338,8 @@ def _invalid(c, _t=t, _dtype=dtype, _spec=spec, _flag=unconstrained):
_run_missing_reference(run, con, server, t, columns, spec, schema_name)
elif spec.metric == MetricType.FIELD_PRESENT:
_run_present(run, con, model, columns, schema, spec)
elif spec.metric == MetricType.FIELD_NAMES:
_run_field_names(run, con, model, t, spec)
elif spec.metric == MetricType.FIELD_TYPE:
_run_type(run, schema, columns, spec, structured_types, native_types)
elif spec.metric == MetricType.FIELD_PHYSICAL_TYPE:
Expand Down Expand Up @@ -891,6 +895,26 @@ def _run_present(run: Run, con, model: str, columns, schema, spec: CheckSpec):
)


def _run_field_names(run: Run, con, model: str, t, spec: CheckSpec):
"""The fields of the data that the contract does not declare; names compare case-insensitively."""
table = t
if spec.uses_raw_view:
try:
table = _resolve_table(con, f"{model}__raw__")
except Exception:
pass
_set_impl(run, spec.key, _describe(spec), "introspection")
declared = {name.lower() for name in spec.columns or []}
additional = [name for name in table.columns if name.lower() not in declared]
_set_diagnostics(run, spec.key, _diag(metric="field_names", additional_fields=additional))
set_result(
run,
spec.key,
ResultEnum.failed if additional else ResultEnum.passed,
f"Fields not in the contract: {', '.join(additional)}" if additional else None,
)


def _run_type(
run: Run,
schema,
Expand Down
16 changes: 15 additions & 1 deletion docs/docs/schema.md
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,7 @@ datacontract test --checks properties datacontract.yaml
| `logicalTypeOptions.minItems` / `maxItems` | array property | Number of elements within bounds |
| `logicalTypeOptions.uniqueItems` | array property | Elements of the array are distinct |
| `quality` | schema, property | See [Define your Quality Rules](./quality-rules/index.md) |
| `customProperties` `additionalProperties: false` | schema | The data has no columns the schema does not declare |

A contract that uses all of them:

Expand Down Expand Up @@ -91,6 +92,19 @@ schema:
physicalName: order_total # the real column
```

Columns the contract does not declare are allowed, as many contracts describe only part of a table. To fail on them, declare `additionalProperties: false` (the JSON Schema term) as a custom property of the schema, since ODCS has no field for it. The check names the undeclared columns; names compare case-insensitively.

```yaml
schema:
- name: orders
customProperties:
- property: additionalProperties
value: false
properties:
- name: order_id
- name: total
```

## Types

A property can declare a portable `logicalType`, a native `physicalType`, or both. Which one is checked depends on the backend:
Expand Down Expand Up @@ -218,7 +232,7 @@ These are common sources of confusion. They are valid ODCS and appear in exports

- **`isNullable`** — the CLI reads `required`, not `isNullable`. Write `required: true` to assert that a column has no nulls.
- **`logicalTypeOptions.format`** — never enforced, on any property. On a `string` property (`email`, `uuid`, `uri`, …) use `pattern` for an enforceable equivalent. On a `date`, `timestamp` or `time` property `format` holds a date pattern such as `yyyy-MM-dd`, and `pattern` is *not* an equivalent there: by the time a check runs, the column has already been parsed as a date, so there is no string left to match. Such a column is validated as a date, in whatever format the server stores it.
- **Descriptive attributes** — `description`, `businessName`, `examples`, `tags`, `classification`, `criticalDataElement`, `transformSourceObjects`, and `customProperties`. `authoritativeDefinitions` generates no check either, but it *is* resolved and inlined before the checks are built — see [Link your Semantics](./semantics.md).
- **Descriptive attributes** — `description`, `businessName`, `examples`, `tags`, `classification`, `criticalDataElement`, `transformSourceObjects`, and `customProperties` (except `additionalProperties` on a schema). `authoritativeDefinitions` generates no check either, but it *is* resolved and inlined before the checks are built — see [Link your Semantics](./semantics.md).
- **Schema-level attributes** other than `name`, `physicalName`, `properties`, and `quality`.
- **Constraints and quality rules on nested properties, on some servers** — `required`, `unique`, `pattern`, `enum`, the other `logicalTypeOptions` and `quality` rules below the top level are checked on the `dataframe`, `databricks`, `trino` and `athena` servers and on every server read through DuckDB (`local`, `s3`, `gcs`, `azure`, `duckdb`, `iceberg`, `kafka`). On every other server, nested properties are only type-checked, and each of their constraints and quality rules is reported as a warning. On SAP HANA, only their quality rules are reported as warnings; their constraints are not checked.

Expand Down
104 changes: 104 additions & 0 deletions tests/test_test_additional_fields.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,104 @@
"""A schema with the custom property `additionalProperties: false` fails on data with fields it does not declare."""

import pytest

from datacontract.data_contract import DataContract
from datacontract.model.run import ResultEnum

PROPERTIES = """ - name: order_id
logicalType: string
- name: quantity
logicalType: integer"""


def _contract(
path, file_format, switch="\n customProperties:\n - property: additionalProperties\n value: false"
):
return f"""apiVersion: v3.2.0
kind: DataContract
id: orders
name: orders
version: 1.0.0
status: active
servers:
- server: local
type: local
format: {file_format}
path: {path}
schema:
- name: orders{switch}
properties:
{PROPERTIES}
"""


def _csv(tmp_path, header):
data = tmp_path / "orders.csv"
data.write_text(f"{header}\nA1|3\n" if header.count("|") == 1 else f"{header}\nA1|3|x\n")
return data


def _check(run):
return next(c for c in run.checks if c.type == "model_no_additional_fields")


def test_a_field_the_contract_does_not_declare_fails(tmp_path):
run = DataContract(data_contract_str=_contract(_csv(tmp_path, "order_id|quantity|note"), "csv")).test()
print(run.pretty())
check = _check(run)
assert check.result == ResultEnum.failed
assert check.reason == "Fields not in the contract: note"
assert check.diagnostics["additional_fields"] == ["note"]
assert [c.type for c in run.checks if c.result != ResultEnum.passed] == ["model_no_additional_fields"]


def test_exactly_the_declared_fields_pass(tmp_path):
run = DataContract(data_contract_str=_contract(_csv(tmp_path, "order_id|quantity"), "csv")).test()
assert _check(run).result == ResultEnum.passed
assert run.result == ResultEnum.passed


def test_field_names_compare_case_insensitively(tmp_path):
run = DataContract(data_contract_str=_contract(_csv(tmp_path, "ORDER_ID|Quantity"), "csv")).test()
assert _check(run).result == ResultEnum.passed


def test_without_the_custom_property_additional_fields_are_allowed(tmp_path):
run = DataContract(data_contract_str=_contract(_csv(tmp_path, "order_id|quantity|note"), "csv", switch="")).test()
assert not [c for c in run.checks if c.type == "model_no_additional_fields"]
assert run.result == ResultEnum.passed


@pytest.mark.parametrize("value", ["'false'", "False"])
def test_the_value_may_be_text(tmp_path, value):
switch = f"\n customProperties:\n - property: additionalProperties\n value: {value}"
run = DataContract(data_contract_str=_contract(_csv(tmp_path, "order_id|quantity|note"), "csv", switch)).test()
assert _check(run).result == ResultEnum.failed


def test_a_json_record_with_an_undeclared_key_fails(tmp_path):
data = tmp_path / "orders.json"
data.write_text('[{"order_id": "A1", "quantity": 3, "note": "x"}]')
run = DataContract(data_contract_str=_contract(data, "json")).test()
assert _check(run).reason == "Fields not in the contract: note"


def test_the_check_reads_no_rows(tmp_path):
contract = _contract(_csv(tmp_path, "order_id|quantity|note"), "csv")
run = DataContract(data_contract_str=contract, metadata_only=True).test()
assert _check(run).result == ResultEnum.failed


def test_a_database_table_with_an_undeclared_column_fails(tmp_path):
import duckdb

database = tmp_path / "orders.duckdb"
with duckdb.connect(str(database)) as con:
con.sql("CREATE TABLE orders (order_id VARCHAR, quantity INTEGER, note VARCHAR)")
con.sql("INSERT INTO orders VALUES ('A1', 3, 'x')")
contract = _contract(tmp_path, "csv").replace(
f" type: local\n format: csv\n path: {tmp_path}", f" type: duckdb\n database: {database}"
)
run = DataContract(data_contract_str=contract).test()
print(run.pretty())
assert _check(run).reason == "Fields not in the contract: note"