{
  "$schema": "https://json-schema.org/draft/2020-12/schema",
  "$id": "https://datacoolie.github.io/datacoolie/schema/0.2.0/metadata.schema.json",
  "title": "DataCoolie Metadata",
  "description": "Metadata definition for the DataCoolie ETL framework. Defines connections, dataflows, sources, destinations, and transforms.",
  "type": "object",
  "properties": {
    "$schema": {
      "type": "string",
      "description": "Reference to the DataCoolie metadata JSON Schema for IDE validation and autocomplete."
    },
    "connections": {
      "type": "array",
      "description": "Endpoint configurations for data sources and destinations.",
      "items": { "$ref": "#/$defs/Connection" }
    },
    "dataflows": {
      "type": "array",
      "description": "ETL pipeline definitions composing source, destination, and transform.",
      "items": { "$ref": "#/$defs/DataFlow" }
    },
    "schema_hints": {
      "type": "array",
      "description": "Shared schema hints that apply across dataflows. Matched by connection_name + table_name.",
      "items": { "$ref": "#/$defs/SharedSchemaHint" }
    },
    "extensions": {
      "type": "object",
      "description": "Project-owned metadata extensions. Framework semantics do not interpret these values.",
      "additionalProperties": true
    }
  },
  "anyOf": [
    { "required": ["connections"] },
    { "required": ["dataflows"] }
  ],
  "additionalProperties": false,

  "$defs": {
    "Connection": {
      "type": "object",
      "description": "An endpoint: file root, lakehouse path, RDBMS, REST API, or Python function.",
      "required": ["name"],
      "properties": {
        "name": {
          "type": "string",
          "minLength": 1,
          "description": "Unique connection identifier. Used as connection_id via name_to_uuid."
        },
        "connection_id": {
          "type": ["string", "null"],
          "description": "Optional stable connection identifier. When omitted, the framework derives it from name."
        },
        "workspace_id": {
          "type": ["string", "null"],
          "description": "Optional provider/workspace scope identifier retained by database/API metadata."
        },
        "connection_type": {
          "type": "string",
          "enum": ["file", "lakehouse", "database", "api", "function", "streaming"],
          "description": "Connection endpoint category. Auto-derived from format if omitted."
        },
        "format": {
          "type": "string",
          "enum": ["delta", "iceberg", "parquet", "csv", "json", "jsonl", "avro", "excel", "sql", "api", "function"],
          "description": "Data format for this connection."
        },
        "catalog": {
          "type": ["string", "null"],
          "description": "Catalog name for Unity Catalog or Iceberg catalogs."
        },
        "database": {
          "type": ["string", "null"],
          "description": "Database or schema namespace."
        },
        "configure": {
          "type": "object",
          "description": "Type-specific settings. Structure depends on connection_type. Common fields below; type-specific fields added conditionally.",
          "properties": {
            "read_options": { "type": "object", "additionalProperties": true, "description": "Engine-specific read options (e.g. header, inferSchema, delimiter)." },
            "write_options": { "type": "object", "additionalProperties": true, "description": "Engine-specific write options (e.g. compression, mode)." },
            "merge_options": { "type": "object", "additionalProperties": true, "description": "Engine-specific merge/upsert options. Legacy merge aliases in write_options remain readable during migration." },
            "use_schema_hint": { "type": "boolean", "default": true, "description": "Apply schema hints from transform. Set false to skip type casting." },
            "schema_hint_type_system": {
              "type": ["string", "null"],
              "enum": ["spark_sql", "spark sql", "spark", "postgresql", "postgres", "mysql", "mssql", "sql server", "sqlserver", "oracle", "sqlite", null],
              "description": "Optional source convention used to interpret transform.schema_hints. If omitted, the known database dialect is used; otherwise Spark SQL conventions apply."
            },
            "backward_years": { "type": "integer", "description": "Look-back years for watermark value." },
            "backward_months": { "type": "integer", "description": "Look-back months for watermark value." },
            "backward_days": { "type": "integer", "description": "Look-back days for watermark value." },
            "backward_hours": { "type": "integer", "description": "Look-back hours for watermark value." },
            "backward_closing_day": { "type": "integer", "description": "Look-back closing day for watermark value." },
            "backward": { "type": "object", "description": "Nested backward config: {days, months, hours, years, closing_day}.", "additionalProperties": true }
          },
          "additionalProperties": true
        },
        "secrets_ref": {
          "description": "Map of secret source to list of configure fields to resolve. Outer key is the vault identifier (ARN, Key Vault URL, or empty string for env vars).",
          "oneOf": [
            { "type": "null" },
            {
              "type": "object",
              "additionalProperties": {
                "type": "array",
                "items": { "type": "string" }
              }
            },
            { "type": "string" }
          ]
        },
        "is_active": {
          "type": "boolean",
          "default": true,
          "description": "Toggle connection on/off."
        }
      },
      "allOf": [
        { "$ref": "#/$defs/ConnectionConfigureFile" },
        { "$ref": "#/$defs/ConnectionConfigureLakehouse" },
        { "$ref": "#/$defs/ConnectionConfigureDatabase" },
        { "$ref": "#/$defs/ConnectionConfigureApi" }
      ],
      "additionalProperties": false
    },

    "ConnectionConfigureFile": {
      "if": {
        "properties": { "connection_type": { "const": "file" } },
        "required": ["connection_type"]
      },
      "then": {
        "properties": {
          "configure": {
            "type": "object",
            "properties": {
              "base_path": { "type": "string", "description": "Root file storage path." },
              "use_hive_partitioning": { "type": "boolean", "default": false, "description": "Enable Hive-style partition discovery." },
              "date_folder_partitions": { "type": "string", "description": "Date folder pattern, e.g. {year}/{month}/{day}." },
              "backward_years": { "type": "integer", "description": "Look-back years for watermark value." },
              "backward_months": { "type": "integer", "description": "Look-back months for watermark value." },
              "backward_days": { "type": "integer", "description": "Look-back days for watermark value." },
              "backward_hours": { "type": "integer", "description": "Look-back hours for watermark value." },
              "backward_closing_day": { "type": "integer", "description": "Look-back closing day for watermark value." },
              "backward": {
                "type": "object",
                "description": "Nested backward look-back config. Keys: days, months, hours, years, closing_day. Overrides connection-level setting.",
                "properties": {
                  "years": { "type": "integer" },
                  "months": { "type": "integer" },
                  "days": { "type": "integer" },
                  "hours": { "type": "integer" },
                  "closing_day": { "type": "integer" }
                },
                "additionalProperties": false
              }
            },
            "additionalProperties": true
          }
        }
      }
    },

    "ConnectionConfigureLakehouse": {
      "if": {
        "properties": { "connection_type": { "const": "lakehouse" } },
        "required": ["connection_type"]
      },
      "then": {
        "properties": {
          "configure": {
            "type": "object",
            "properties": {
              "base_path": { "type": "string", "description": "Root lakehouse storage path." },
              "catalog": { "type": "string", "description": "Catalog name (overrides top-level catalog)." },
              "database": { "type": "string", "description": "Database name (overrides top-level database)." },
              "athena_output_location": { "type": "string", "description": "S3 path for Athena DDL results. Triggers Glue catalog registration." },
              "generate_manifest": { "type": "boolean", "default": false, "description": "Write _symlink_format_manifest/ after writes." },
              "register_symlink_table": { "type": "boolean", "default": false, "description": "Register SymlinkTextInputFormat table in Glue. Implies generate_manifest." },
              "symlink_database_prefix": { "type": "string", "default": "symlink_", "description": "Prefix for symlink Glue database name." }
            },
            "additionalProperties": true
          }
        }
      }
    },

    "ConnectionConfigureDatabase": {
      "if": {
        "properties": { "connection_type": { "const": "database" } },
        "required": ["connection_type"]
      },
      "then": {
        "properties": {
          "configure": {
            "type": "object",
            "properties": {
              "database_type": {
                "type": "string",
                "enum": ["postgresql", "mysql", "mssql", "oracle", "sqlite"],
                "description": "Database dialect."
              },
              "auth_type": {
                "type": "string",
                "enum": ["password", "service_principal", "managed_identity", "access_token"],
                "default": "password",
                "description": "Authentication method. For service_principal: username=client_id, password=client_secret."
              },
              "host": { "type": "string", "description": "Database hostname." },
              "port": { "type": "integer", "description": "Database port." },
              "database": { "type": "string", "description": "Database name." },
              "username": { "type": "string", "description": "Database username or SPN client_id (env var placeholder for secrets_ref)." },
              "password": { "type": "string", "description": "Database password or SPN client_secret (env var placeholder for secrets_ref)." },
              "tenant_id": { "type": "string", "description": "Azure AD tenant ID (required for service_principal auth)." },
              "token": { "type": "string", "description": "Pre-fetched access token (required for access_token auth)." },
              "driver": { "type": "string", "description": "JDBC driver class name." },
              "url": { "type": "string", "description": "Explicit connection URL/string." },
              "encrypt": { "type": "string", "description": "Encryption option (MSSQL)." }
            },
            "additionalProperties": true
          }
        }
      }
    },

    "ConnectionConfigureApi": {
      "if": {
        "properties": { "connection_type": { "const": "api" } },
        "required": ["connection_type"]
      },
      "then": {
        "properties": {
          "configure": {
            "type": "object",
            "properties": {
              "base_url": { "type": "string", "description": "Base URL for API requests." },
              "timeout": { "type": "number", "description": "Request timeout in seconds (int or float)." },
              "auth_type": {
                "type": "string",
                "enum": ["bearer", "api_key", "basic", "oauth2_client_credentials", "aws_sigv4"],
                "description": "Authentication method."
              },
              "auth_token": { "type": "string", "description": "Bearer token (or env var placeholder)." },
              "api_key_header": { "type": "string", "description": "Header name for API key auth." },
              "api_key_value": { "type": "string", "description": "API key value (or env var placeholder)." },
              "username": { "type": "string", "description": "Username for basic auth." },
              "password": { "type": "string", "description": "Password for basic auth." },
              "token_url": { "type": "string", "description": "OAuth2 token endpoint." },
              "client_id": { "type": "string", "description": "OAuth2 client ID." },
              "client_secret": { "type": "string", "description": "OAuth2 client secret." },
              "token_auth_method": { "type": "string", "description": "OAuth2 token auth method (e.g. client_secret_basic)." },
              "scope": { "type": "string", "description": "OAuth2 scope." },
              "watermark_to_param_timezone": { "type": "string", "description": "Timezone for watermark→query param conversion (e.g. +07:00)." },
              "default_headers": { "type": "object", "additionalProperties": { "type": "string" }, "description": "Static headers added to every request." },
              "token_request_body_format": { "type": "string", "enum": ["form", "json"], "default": "form", "description": "OAuth2 token request body encoding format." },
              "token_request_extras": { "type": "object", "additionalProperties": true, "description": "Extra fields sent in the OAuth2 token request body." },
              "aws_region": { "type": "string", "description": "AWS region for SigV4 authentication (e.g. us-east-1)." },
              "aws_service": { "type": "string", "description": "AWS service name for SigV4 auth (e.g. execute-api)." },
              "aws_access_key_id": { "type": "string", "description": "AWS access key ID for SigV4 (or env var placeholder)." },
              "aws_secret_access_key": { "type": "string", "description": "AWS secret access key for SigV4 (or env var placeholder)." },
              "aws_session_token": { "type": "string", "description": "AWS session token for SigV4 temporary credentials (or env var placeholder)." }
            },
            "additionalProperties": true
          }
        }
      }
    },

    "DataFlow": {
      "type": "object",
      "description": "Complete ETL pipeline configuration composing source, destination, and transform.",
      "required": ["source", "destination"],
      "anyOf": [
        { "required": ["name"] },
        { "required": ["dataflow_id"] }
      ],
      "properties": {
        "name": {
          "type": ["string", "null"],
          "minLength": 1,
          "description": "Unique dataflow name. Used to derive dataflow_id."
        },
        "dataflow_id": {
          "type": ["string", "null"],
          "description": "Optional stable dataflow identifier. When omitted, the framework derives it from name."
        },
        "workspace_id": {
          "type": ["string", "null"],
          "description": "Optional provider/workspace scope identifier retained by database/API metadata."
        },
        "description": {
          "type": ["string", "null"],
          "description": "Human-readable description of this dataflow."
        },
        "stage": {
          "type": ["string", "null"],
          "description": "Project-defined runtime selection label used by driver.run(stage=...). Stage names are dynamic and imply no execution order."
        },
        "group_number": {
          "type": ["integer", "null"],
          "description": "Co-location group for job assignment and ordered normal ETL execution. Different groups run independently. Null means an independent dataflow."
        },
        "execution_order": {
          "type": ["integer", "null"],
          "description": "Normal ETL order bucket within a non-null group: lower buckets finish first, equal values may run in parallel, null is 0. Without a group it provides no dependency ordering."
        },
        "processing_mode": {
          "type": "string",
          "enum": ["batch", "microbatch", "streaming"],
          "default": "batch",
          "description": "Accepted processing-mode metadata value. The built-in driver currently executes normal ETL through the batch path; microbatch and streaming are reserved for specialized or future runtimes."
        },
        "is_active": {
          "type": "boolean",
          "default": true,
          "description": "Toggle dataflow on/off."
        },
        "source": { "$ref": "#/$defs/Source" },
        "destination": { "$ref": "#/$defs/Destination" },
        "transform": { "$ref": "#/$defs/Transform" },
        "configure": {
          "type": "object",
          "description": "Dataflow-level configuration options.",
          "additionalProperties": true
        }
      },
      "additionalProperties": false
    },

    "Source": {
      "type": "object",
      "description": "Read-side pipeline configuration. References a connection by name.",
      "oneOf": [
        {
          "required": ["connection_name"],
          "not": { "required": ["connection"] }
        },
        {
          "required": ["connection"],
          "not": { "required": ["connection_name"] }
        }
      ],
      "properties": {
        "connection_name": {
          "type": "string",
          "minLength": 1,
          "description": "Name of the connection to read from."
        },
        "connection": {
          "oneOf": [
            { "type": "string", "minLength": 1 },
            { "$ref": "#/$defs/Connection" }
          ],
          "description": "Named connection reference or an inline connection definition."
        },
        "schema_name": {
          "type": ["string", "null"],
          "description": "Schema namespace within the connection."
        },
        "table": {
          "type": ["string", "null"],
          "description": "Table or object name to read."
        },
        "query": {
          "type": ["string", "null"],
          "description": "SQL query (alternative to table for database sources)."
        },
        "python_function": {
          "type": ["string", "null"],
          "description": "Dotted path to a Python function for function sources (e.g. functions.sources.my_fn)."
        },
        "watermark_columns": {
          "type": "array",
          "items": { "type": "string" },
          "default": [],
          "description": "Columns used for incremental reads (watermark filtering)."
        },
        "filter_expression": {
          "type": ["string", "null"],
          "description": "SQL WHERE predicate applied after watermark filter on raw source columns."
        },
        "configure": {
          "type": "object",
          "description": "Source-level configuration. API fields documented from api_reader.py.",
          "properties": {
            "read_options": { "type": "object", "additionalProperties": true, "description": "Source-level read option overrides (file/lakehouse)." },

            "backward_days": { "type": "integer", "description": "Subtract N days from watermark for partition look-back. Overrides connection-level setting." },
            "backward_months": { "type": "integer", "description": "Subtract N months from watermark for partition look-back. Overrides connection-level setting." },
            "backward_hours": { "type": "integer", "description": "Subtract N hours from watermark for partition look-back. Overrides connection-level setting." },
            "backward_years": { "type": "integer", "description": "Subtract N years from watermark for partition look-back. Overrides connection-level setting." },
            "backward_closing_day": { "type": "integer", "description": "Day-of-month for closing-day look-back strategy. Overrides connection-level setting." },
            "backward": {
              "type": "object",
              "description": "Nested backward look-back config. Keys: days, months, hours, years, closing_day. Overrides connection-level setting.",
              "properties": {
                "days": { "type": "integer" },
                "months": { "type": "integer" },
                "hours": { "type": "integer" },
                "years": { "type": "integer" },
                "closing_day": { "type": "integer" }
              },
              "additionalProperties": false
            },

            "endpoint": { "type": "string", "description": "API endpoint path appended to base_url." },
            "method": { "type": "string", "default": "GET", "description": "HTTP method (GET, POST, PUT, etc.)." },
            "params": { "type": "object", "additionalProperties": true, "description": "Static query parameters added to every request." },
            "body": { "type": "object", "additionalProperties": true, "description": "Request body for POST/PUT requests." },
            "data_path": { "type": "string", "description": "Dot-separated path to the records array in the response JSON (e.g. 'data.items'). Defaults to root-level list." },

            "pagination_type": {
              "type": "string",
              "enum": ["offset", "cursor", "next_link"],
              "description": "Pagination strategy. Omit for single-page fetch."
            },
            "page_size": { "type": "integer", "default": 100, "description": "Records per page." },
            "max_pages": { "type": "integer", "default": 1000, "description": "Safety limit on total pages fetched." },
            "next_link_path": { "type": "string", "default": "next", "description": "Dot-separated path to the next-page URL in the response (next_link pagination)." },
            "next_link_bound_mode": {
              "type": "string",
              "enum": ["opaque", "repeat_query_bounds"],
              "default": "opaque",
              "description": "API next_link continuation policy. Opaque follows the returned URL after origin validation without adding query parameters. Opt in to repeat_query_bounds only when the endpoint requires active query bounds on subsequent pages; matching bounds are retained, missing bounds are added, and duplicate or conflicting bounds are rejected. This option reconciles query bounds only; existing method/body behavior and unrelated parameters remain governed by the API configuration."
            },
            "cursor_path": { "type": "string", "default": "next_cursor", "description": "Dot-separated path to the cursor token in the response (cursor pagination)." },
            "cursor_param": { "type": "string", "default": "cursor", "description": "Query parameter name for cursor value." },
            "offset_param": { "type": "string", "default": "offset", "description": "Query parameter name for page offset." },
            "limit_param": { "type": "string", "default": "limit", "description": "Query parameter name for page size." },
            "total_path": { "type": "string", "description": "Dot-separated path to total record count in first-page response (e.g. 'meta.total'). Enables concurrent offset fetching when set." },
            "offset_max_workers": { "type": "integer", "default": 4, "description": "Max parallel workers for concurrent offset pagination (requires total_path)." },
            "rate_limit_delay": { "type": "number", "default": 0, "description": "Seconds to wait between sequential page requests." },
            "max_retries": { "type": "integer", "default": 10, "description": "Max retries on HTTP 429 responses. Uses Retry-After header or exponential backoff." },

            "watermark_param_mapping": {
              "type": "object",
              "additionalProperties": { "type": "string" },
              "description": "Maps watermark_columns entries to API parameter names (e.g. {'updated_at': 'updated_since'}). Pushes watermark server-side instead of filtering in memory."
            },
            "range_param_mapping": {
              "type": "object",
              "minProperties": 1,
              "additionalProperties": {
                "type": "object",
                "properties": {
                  "lower": {
                    "oneOf": [
                      { "type": "string", "minLength": 1 },
                      {
                        "type": "object",
                        "required": ["name"],
                        "properties": {
                          "location": { "type": "string", "enum": ["params", "body"], "default": "params" },
                          "name": { "type": "string", "minLength": 1 },
                          "operator": { "type": "string", "enum": [">", ">=", "<", "<="], "default": ">=" }
                        },
                        "additionalProperties": false
                      }
                    ]
                  },
                  "upper": {
                    "oneOf": [
                      { "type": "string", "minLength": 1 },
                      {
                        "type": "object",
                        "required": ["name"],
                        "properties": {
                          "location": { "type": "string", "enum": ["params", "body"], "default": "params" },
                          "name": { "type": "string", "minLength": 1 },
                          "operator": { "type": "string", "enum": [">", ">=", "<", "<="], "default": "<" }
                        },
                        "additionalProperties": false
                      }
                    ]
                  },
                  "format": {
                    "type": "string",
                    "enum": ["iso", "date", "timestamp", "timestamp_ms", "datetime", "datetime_ms", "integer", "int", "native_integer", "number"],
                    "default": "iso"
                  },
                  "response_column": { "type": ["string", "null"] },
                  "watermark_value": { "type": "string", "enum": ["observed_max", "request_end"], "default": "observed_max" }
                },
                "additionalProperties": false,
                "anyOf": [{ "required": ["lower"] }, { "required": ["upper"] }]
              },
              "description": "Canonical API per-field lower/upper bindings for bounded reads and replay. The selected field may be outside source.watermark_columns; only authored watermark columns are persisted."
            },
            "watermark_to_param": { "type": "string", "description": "API parameter name to receive the current timestamp as upper bound ('to' side of the window)." },
            "watermark_param_location": {
              "type": "string",
              "enum": ["params", "body"],
              "default": "params",
              "description": "Where to inject watermark values: URL query string or request body."
            },
            "watermark_param_format": {
              "type": "string",
              "enum": ["iso", "date", "timestamp", "timestamp_ms", "datetime", "datetime_ms"],
              "default": "iso",
              "description": "Serialisation format for watermark values sent to the API."
            },
            "watermark_to_param_timezone": { "type": "string", "description": "Timezone for the 'now' upper-bound timestamp (IANA name or ±HH:MM offset). Overrides connection-level default." },

            "watermark_range_interval_unit": {
              "type": "string",
              "enum": ["hour", "day", "month", "year"],
              "description": "Splits the watermark window into sub-ranges of this unit and fetches concurrently. Canonical mode uses range_param_mapping with lower and upper bindings; legacy mode requires watermark_to_param and watermark_param_mapping for saved-state continuation. Requires a stored lower watermark or watermark_range_start."
            },
            "watermark_range_interval_amount": { "type": "integer", "default": 1, "description": "Number of units per sub-range interval (e.g. 3 with unit='hour' → 3-hour windows)." },
            "watermark_range_start": { "type": "string", "description": "ISO-8601 fallback lower bound for API split-range reads when no stored lower watermark exists. Omit when a usable lower watermark is already stored." },
            "watermark_range_max_workers": { "type": "integer", "default": 4, "description": "Max parallel HTTP workers for range-split fetching." },
            "watermark_range_to_exclusive_offset": {
              "type": ["string", "null"],
              "enum": ["1ms", "1s", "1day", null],
              "description": "Epsilon subtracted from each range upper-bound before sending to the API. Use when API has inclusive BETWEEN semantics to prevent duplicates."
            }
          },
          "additionalProperties": true
        }
      },
      "additionalProperties": false
    },

    "Destination": {
      "type": "object",
      "description": "Write-side pipeline configuration. References a connection by name.",
      "required": ["table"],
      "oneOf": [
        {
          "required": ["connection_name"],
          "not": { "required": ["connection"] }
        },
        {
          "required": ["connection"],
          "not": { "required": ["connection_name"] }
        }
      ],
      "properties": {
        "connection_name": {
          "type": "string",
          "minLength": 1,
          "description": "Name of the connection to write to."
        },
        "connection": {
          "oneOf": [
            { "type": "string", "minLength": 1 },
            { "$ref": "#/$defs/Connection" }
          ],
          "description": "Named connection reference or an inline connection definition."
        },
        "table": {
          "type": "string",
          "minLength": 1,
          "description": "Target table or object name."
        },
        "schema_name": {
          "type": ["string", "null"],
          "description": "Schema namespace within the connection."
        },
        "load_type": {
          "type": "string",
          "enum": ["full_load", "overwrite", "append", "merge_upsert", "merge_overwrite", "scd2"],
          "default": "append",
          "description": "Write strategy."
        },
        "merge_keys": {
          "type": "array",
          "items": { "type": "string" },
          "default": [],
          "description": "Key columns for merge operations (merge_upsert, merge_overwrite, scd2)."
        },
        "partition_columns": {
          "type": "array",
          "items": { "$ref": "#/$defs/PartitionColumn" },
          "default": [],
          "description": "Partition columns for the destination table."
        },
        "configure": {
          "type": "object",
          "description": "Destination-level configuration.",
          "properties": {
            "scd2_effective_column": { "type": "string", "description": "SQL expression used as __valid_from for SCD2 loads." },
            "replace_by_watermark": { "type": "boolean", "default": false, "description": "Use range-based window replace instead of key-based delete for merge_overwrite." },
            "write_options": { "type": "object", "additionalProperties": true },
            "merge_options": { "type": "object", "additionalProperties": true, "description": "Engine-specific merge/upsert options." },
            "partition_columns": {
              "type": "array",
              "items": { "$ref": "#/$defs/PartitionColumn" },
              "description": "Alternative location for partition columns (lifted to top level)."
            }
          },
          "additionalProperties": true
        }
      },
      "additionalProperties": false
    },

    "Transform": {
      "type": "object",
      "description": "Transformation rules applied between source read and destination write.",
      "properties": {
        "deduplicate_columns": {
          "type": "array",
          "items": { "type": "string" },
          "default": [],
          "description": "Key columns for deduplication."
        },
        "latest_data_columns": {
          "type": "array",
          "items": { "type": "string" },
          "default": [],
          "description": "Columns used to pick the latest row when deduplicating."
        },
        "filter_expression": {
          "type": ["string", "null"],
          "description": "SQL WHERE predicate applied after computed columns (transformer order 35)."
        },
        "additional_columns": {
          "type": "array",
          "items": { "$ref": "#/$defs/AdditionalColumn" },
          "default": [],
          "description": "Computed columns added during transform phase."
        },
        "schema_hints": {
          "type": "array",
          "items": { "$ref": "#/$defs/SchemaHint" },
          "default": [],
          "description": "Column-level type hints for schema conversion."
        },
        "select_columns": {
          "type": "array",
          "items": { "type": "string", "minLength": 1 },
          "uniqueItems": true,
          "default": [],
          "description": "Business columns to keep, resolved before rename_columns. Mutually exclusive with drop_columns."
        },
        "drop_columns": {
          "type": "array",
          "items": { "type": "string", "minLength": 1 },
          "uniqueItems": true,
          "default": [],
          "description": "Business columns to remove, resolved before rename_columns. Mutually exclusive with select_columns."
        },
        "rename_columns": {
          "type": "object",
          "additionalProperties": { "type": "string", "minLength": 1 },
          "default": {},
          "description": "Atomic old-name to new-name mapping. Chains, cycles, and overwrites are invalid."
        },
        "value_rules": {
          "type": "array",
          "items": { "$ref": "#/$defs/ValueRule" },
          "default": [],
          "description": "Typed normalization rules applied before schema casting."
        },
        "hash_columns": {
          "type": "array",
          "items": { "$ref": "#/$defs/HashColumn" },
          "default": [],
          "description": "Stable SHA-256 string or signed XXHash64 BIGINT columns generated from canonical scalar payloads."
        },
        "masking_rules": {
          "type": "array",
          "items": { "$ref": "#/$defs/MaskingRule" },
          "default": [],
          "description": "Structured scalar masking rules applied before projection."
        },
        "configure": {
          "type": "object",
          "description": "Transform-level configuration.",
          "properties": {
            "convert_timestamp_ntz": { "type": "boolean", "default": false, "description": "Convert timestamp_ntz columns to timestamp using timestamp_timezone. Disabled by default to preserve wall-clock semantics." },
            "timestamp_timezone": { "type": ["string", "null"], "description": "Timezone used for an explicit timestamp_ntz-to-instant conversion." },
            "deduplicate_by_rank": { "type": "boolean", "default": false, "description": "Use RANK-based dedup instead of ROW_NUMBER." },
            "missing_column_policy": { "type": "string", "enum": ["error", "ignore"], "default": "error", "description": "Behavior for absent columns in typed value/hash/masking rules and projection. Missing schema hints warn and skip; deduplication remains strict." }
          },
          "additionalProperties": true
        }
      },
      "allOf": [
        {
          "not": {
            "properties": {
              "select_columns": { "minItems": 1 },
              "drop_columns": { "minItems": 1 }
            },
            "required": ["select_columns", "drop_columns"]
          }
        }
      ],
      "additionalProperties": false
    },

    "ValueRule": {
      "type": "object",
      "required": ["operation", "columns"],
      "properties": {
        "operation": { "type": "string", "enum": ["trim", "case", "regex_replace", "empty_to_null", "fill_null", "map"] },
        "columns": { "type": "array", "items": { "type": "string", "minLength": 1 }, "minItems": 1, "uniqueItems": true },
        "order": { "type": "integer", "minimum": 0, "default": 100 },
        "mode": { "type": ["string", "null"], "enum": ["lower", "upper", null] },
        "pattern": {
          "type": ["string", "null"],
          "maxLength": 4096,
          "description": "DataCoolie portable regex v1; engine-specific constructs such as lookaround, backreferences, named groups, inline flags, and Unicode shorthand classes are rejected by runtime validation."
        },
        "replacement": {
          "type": "string",
          "default": "",
          "description": "Literal replacement text; capture-group expansion is not supported."
        },
        "value": { "type": ["string", "number", "integer", "boolean", "null"] },
        "mapping": { "type": "object", "additionalProperties": { "type": "string" }, "default": {} },
        "on_unmapped": { "type": "string", "enum": ["keep", "null"], "default": "keep" }
      },
      "allOf": [
        {
          "if": { "properties": { "operation": { "const": "case" } }, "required": ["operation"] },
          "then": {
            "required": ["mode"],
            "properties": { "mode": { "type": "string", "enum": ["lower", "upper"] } }
          }
        },
        {
          "if": { "properties": { "operation": { "const": "regex_replace" } }, "required": ["operation"] },
          "then": {
            "required": ["pattern"],
            "properties": { "pattern": { "type": "string", "maxLength": 4096 } }
          }
        },
        {
          "if": { "properties": { "operation": { "const": "fill_null" } }, "required": ["operation"] },
          "then": {
            "required": ["value"],
            "properties": { "value": { "type": ["string", "number", "integer", "boolean"] } }
          }
        },
        {
          "if": { "properties": { "operation": { "const": "map" } }, "required": ["operation"] },
          "then": {
            "required": ["mapping"],
            "properties": { "mapping": { "minProperties": 1 } }
          }
        }
      ],
      "additionalProperties": false
    },

    "MaskingRule": {
      "type": "object",
      "required": ["method", "columns"],
      "properties": {
        "method": { "type": "string", "enum": ["redact", "nullify", "partial", "numeric_bucket", "date_truncate"] },
        "columns": { "type": "array", "items": { "type": "string", "minLength": 1 }, "minItems": 1, "uniqueItems": true },
        "value": { "type": ["string", "number", "integer", "boolean", "null"] },
        "keep_start": { "type": "integer", "minimum": 0, "default": 0 },
        "keep_end": { "type": "integer", "minimum": 0, "default": 0 },
        "mask_char": { "type": "string", "minLength": 1, "maxLength": 1, "default": "*" },
        "bucket_size": { "type": ["number", "null"], "exclusiveMinimum": 0 },
        "unit": { "type": ["string", "null"], "enum": ["year", "month", "day", "hour", null] }
      },
      "allOf": [
        {
          "if": { "properties": { "method": { "const": "redact" } }, "required": ["method"] },
          "then": {
            "required": ["value"],
            "properties": { "value": { "type": ["string", "number", "integer", "boolean"] } }
          }
        },
        {
          "if": { "properties": { "method": { "const": "numeric_bucket" } }, "required": ["method"] },
          "then": {
            "required": ["bucket_size"],
            "properties": { "bucket_size": { "type": "number", "exclusiveMinimum": 0 } }
          }
        },
        {
          "if": { "properties": { "method": { "const": "date_truncate" } }, "required": ["method"] },
          "then": {
            "required": ["unit"],
            "properties": { "unit": { "type": "string", "enum": ["year", "month", "day", "hour"] } }
          }
        }
      ],
      "additionalProperties": false
    },

    "HashColumn": {
      "type": "object",
      "required": ["target_column", "columns"],
      "properties": {
        "target_column": { "type": "string", "minLength": 1 },
        "columns": { "type": "array", "items": { "type": "string", "minLength": 1 }, "minItems": 1, "uniqueItems": true },
        "algorithm": { "type": "string", "enum": ["sha256", "xxhash64"], "default": "sha256" },
        "serialization": { "type": "string", "enum": ["dc_hash_v1"], "default": "dc_hash_v1" }
      },
      "additionalProperties": false
    },

    "SchemaHint": {
      "type": "object",
      "description": "Column-level type hint for schema conversion.",
      "required": ["column_name", "data_type"],
      "properties": {
        "column_name": {
          "type": "string",
          "minLength": 1,
          "description": "Target column name."
        },
        "data_type": {
          "type": "string",
          "minLength": 1,
          "description": "Target data type (e.g. int, string, decimal, timestamp, date, boolean, float, double, long, short, byte)."
        },
        "format": {
          "type": ["string", "null"],
          "description": "Date/timestamp format pattern (e.g. yyyy-MM-dd)."
        },
        "precision": {
          "type": ["integer", "string", "null"],
          "description": "Decimal precision (total digits)."
        },
        "scale": {
          "type": ["integer", "string", "null"],
          "description": "Decimal scale (digits after decimal point)."
        },
        "default_value": {
          "type": ["string", "null"],
          "description": "Default value for null entries."
        },
        "ordinal_position": {
          "type": ["integer", "string", "null"],
          "default": 0,
          "description": "Column ordering position."
        },
        "is_active": {
          "type": ["boolean", "string"],
          "default": true,
          "description": "Toggle hint on/off."
        }
      },
      "additionalProperties": false
    },

    "PartitionColumn": {
      "type": "object",
      "description": "Partition column definition with optional derived expression.",
      "required": ["column"],
      "properties": {
        "column": {
          "type": "string",
          "minLength": 1,
          "description": "Partition column name."
        },
        "expression": {
          "type": ["string", "null"],
          "description": "SQL expression to derive the partition value (e.g. year(event_date))."
        }
      },
      "additionalProperties": false
    },

    "AdditionalColumn": {
      "type": "object",
      "description": "Computed column added during the transform phase.",
      "required": ["column", "expression"],
      "properties": {
        "column": {
          "type": "string",
          "minLength": 1,
          "description": "Output column name."
        },
        "expression": {
          "type": "string",
          "minLength": 1,
          "description": "SQL expression to compute the column value."
        }
      },
      "additionalProperties": false
    },

    "SharedSchemaHint": {
      "type": "object",
      "description": "Shared schema hints scoped to a connection + table. Applied by the metadata provider to matching dataflows.",
      "required": ["table_name", "hints"],
      "anyOf": [
        { "required": ["connection_name"] },
        { "required": ["connection_id"] }
      ],
      "properties": {
        "connection_name": {
          "type": "string",
          "minLength": 1,
          "description": "Connection name this hint set applies to."
        },
        "connection_id": {
          "type": "string",
          "minLength": 1,
          "description": "Connection identifier this hint set applies to."
        },
        "table_name": {
          "type": "string",
          "minLength": 1,
          "description": "Table name this hint set applies to."
        },
        "schema_name": {
          "type": ["string", "null"],
          "description": "Optional schema namespace to narrow hint matching to a specific schema."
        },
        "hints": {
          "type": "array",
          "items": { "$ref": "#/$defs/SchemaHint" },
          "description": "List of column-level type hints."
        }
      },
      "additionalProperties": false
    }
  }
}
