diff --git a/.cargo/config.toml b/.cargo/config.toml deleted file mode 100644 index c427b4132f..0000000000 --- a/.cargo/config.toml +++ /dev/null @@ -1,8 +0,0 @@ -[build] -rustflags = ["--cfg", "tokio_unstable"] - -[profile.test] -lto = "off" - -[target.aarch64-apple-darwin] -rustflags = ["-C", "link-arg=-ld_classic"] diff --git a/.devcontainer/devcontainer.json b/.devcontainer/devcontainer.json deleted file mode 100644 index d5bae8ff64..0000000000 --- a/.devcontainer/devcontainer.json +++ /dev/null @@ -1,37 +0,0 @@ -// For format details, see https://aka.ms/devcontainer.json. For config options, see the -// README at: https://github.com/devcontainers/templates/tree/main/src/rust -{ - "name": "Rust", - // Or use a Dockerfile or Docker Compose file. More info: https://containers.dev/guide/dockerfile - "image": "mcr.microsoft.com/devcontainers/rust:1-1-bullseye", - "features": { - "ghcr.io/devcontainers/features/rust:1": { - "version": "latest", - "profile": "default" - } - }, - "runArgs": [ - "-m", - "32GB", - "--cpus", - "10" - ] - // Use 'mounts' to make the cargo cache persistent in a Docker Volume. - // "mounts": [ - // { - // "source": "devcontainer-cargo-cache-${devcontainerId}", - // "target": "/usr/local/cargo", - // "type": "volume" - // } - // ] - // Features to add to the dev container. More info: https://containers.dev/features. - // "features": {}, - // Use 'forwardPorts' to make a list of ports inside the container available locally. - // "forwardPorts": [], - // Use 'postCreateCommand' to run commands after the container is created. - // "postCreateCommand": "rustc --version", - // Configure tool-specific properties. - // "customizations": {}, - // Uncomment to connect as root instead. More info: https://aka.ms/dev-containers-non-root. - // "remoteUser": "root" -} \ No newline at end of file diff --git a/.dockerignore b/.dockerignore deleted file mode 100644 index b685045d9d..0000000000 --- a/.dockerignore +++ /dev/null @@ -1,5 +0,0 @@ -db/ -dozer-sql/data -**/**/target/ -target/ -!target/release/dozer \ No newline at end of file diff --git a/.github/ISSUE_TEMPLATE/bug_report.md b/.github/ISSUE_TEMPLATE/bug_report.md deleted file mode 100644 index 41f1b83cb9..0000000000 --- a/.github/ISSUE_TEMPLATE/bug_report.md +++ /dev/null @@ -1,35 +0,0 @@ ---- -name: Bug report -about: Create a report to help us improve -title: '' -labels: '' -assignees: '' - ---- - -## Describe the bug - -A clear and concise description of what the bug is. - -## Environment - -- **Dozer version**: 0.1.2 -- **OS Version**: Ubuntu 20.04 -- **Docker Compose Version(Optional)**: - -## To Reproduce - -Steps to reproduce the behavior: - -1. Share your `dozer-config.yaml` -2. Run cli command `dozer build` or `dozer run app` -... -N. Share you error - -## Expected behavior - -A clear and concise description of what you expected to happen. - -## Logs - -If applicable, add logs to help explain your problem. diff --git a/.github/ISSUE_TEMPLATE/config.yml b/.github/ISSUE_TEMPLATE/config.yml deleted file mode 100644 index 6303df14fe..0000000000 --- a/.github/ISSUE_TEMPLATE/config.yml +++ /dev/null @@ -1,7 +0,0 @@ -contact_links: - - name: Support questions - url: https://github.com/getdozer/dozer/discussions/new?category=q-a - about: Open a discussion to ask for help - - name: Discord - url: https://discord.gg/ChZzfvVH2p - about: Get a real-time support from our Discord community diff --git a/.github/ISSUE_TEMPLATE/documentation_improvement.md b/.github/ISSUE_TEMPLATE/documentation_improvement.md deleted file mode 100644 index 7ae2daae4c..0000000000 --- a/.github/ISSUE_TEMPLATE/documentation_improvement.md +++ /dev/null @@ -1,28 +0,0 @@ ---- -name: Documentation Improvement -about: Suggest improvements or report inaccuracies in the Dozer documentation. -title: "[Docs] " -labels: documentation -assignees: '' - ---- - -**Describe the documentation issue** -A clear and concise description of what the issue is in the documentation. Please specify the topic, section, or page where you found the issue. - -**Suggested improvement or correction** -Describe your suggestion for improving or correcting the documentation. If possible, provide examples or references to support your suggestion. - -**Additional context** -Add any other context or screenshots about the documentation issue here. This can include anything that will help understand and address the issue. - -**Your environment (optional)** -If applicable, provide information about your environment or setup (e.g., operating system, Dozer version, etc.) that may be relevant to the documentation issue. - -**Checklist** -Please ensure you have completed the following before submitting your issue: - -- [ ] I have searched the existing issues to make sure my issue hasn't already been reported. -- [ ] I have provided a clear and concise description of the issue. -- [ ] I have provided any relevant examples or references to support my suggestion. -- [ ] I have filled out all sections of the issue template to the best of my ability. diff --git a/.github/ISSUE_TEMPLATE/feature_request.md b/.github/ISSUE_TEMPLATE/feature_request.md deleted file mode 100644 index bbcbbe7d61..0000000000 --- a/.github/ISSUE_TEMPLATE/feature_request.md +++ /dev/null @@ -1,20 +0,0 @@ ---- -name: Feature request -about: Suggest an idea for this project -title: '' -labels: '' -assignees: '' - ---- - -**Is your feature request related to a problem? Please describe.** -A clear and concise description of what the problem is. Ex. I'm always frustrated when [...] - -**Describe the solution you'd like** -A clear and concise description of what you want to happen. - -**Describe alternatives you've considered** -A clear and concise description of any alternative solutions or features you've considered. - -**Additional context** -Add any other context or screenshots about the feature request here. diff --git a/.github/config/cliff.toml b/.github/config/cliff.toml deleted file mode 100644 index d53f1709c2..0000000000 --- a/.github/config/cliff.toml +++ /dev/null @@ -1,67 +0,0 @@ -# configuration file for git-cliff (0.1.0) - -[changelog] -# changelog header -header = """ -# Changelog\n -All notable changes to this project will be documented in this file.\n -""" -# template for the changelog body -# https://tera.netlify.app/docs/#introduction -body = """ -{% if version %}\ - ## [{{ version | trim_start_matches(pat="v") }}] - {{ timestamp | date(format="%Y-%m-%d") }} -{% else %}\ - ## [unreleased] -{% endif %}\ -{% for group, commits in commits | group_by(attribute="group") %} - ### {{ group | upper_first }} - {% for commit in commits %} - - {% if commit.breaking %}[**breaking**] {% endif %}{{ commit.message | upper_first }}\ - {% endfor %} -{% endfor %}\n -""" -# remove the leading and trailing whitespace from the template -trim = true -# changelog footer -footer = """ -## Support\n -Contact us at https://getdozer.io -""" - -[git] -# parse the commits based on https://www.conventionalcommits.org -conventional_commits = true -# filter out the commits that are not conventional -filter_unconventional = true -# process each line of a commit as an individual commit -split_commits = false -# regex for preprocessing the commit messages -commit_preprocessors = [ - { pattern = '\((\w+\s)?#([0-9]+)\)', replace = "([#${2}](https://github.com/getdozer/dozer/issues/${2}))"}, -] -# regex for parsing and grouping commits -commit_parsers = [ - { message = "^feat", group = "Features"}, - { message = "^fix", group = "Bug Fixes"}, - { message = "^doc", group = "Documentation"}, - { message = "^perf", group = "Performance"}, - { message = "^refactor", group = "Refactor"}, - { message = "^style", group = "Styling"}, - { message = "^test", group = "Testing"}, - { message = "^chore\\(release\\): prepare for", skip = true}, - { message = "^chore", group = "Miscellaneous Tasks"}, - { body = ".*security", group = "Security"}, -] -# filter out the commits that are not matched by commit parsers -filter_commits = false -# glob pattern for matching git tags -tag_pattern = "v[0-9]*" -# regex for skipping tags -skip_tags = "v0.1.0-beta.1" -# regex for ignoring tags -ignore_tags = "" -# sort the tags chronologically -date_order = false -# sort the commits inside sections by oldest/newest order -sort_commits = "oldest" diff --git a/.github/dependabot.yml b/.github/dependabot.yml deleted file mode 100644 index 8ffad3b6f6..0000000000 --- a/.github/dependabot.yml +++ /dev/null @@ -1,9 +0,0 @@ -version: 2 -updates: - - package-ecosystem: cargo - directory: / - schedule: - interval: daily - commit-message: - prefix: '' - labels: [] diff --git a/.github/workflows/coverage.yaml b/.github/workflows/coverage.yaml deleted file mode 100644 index 26b9a34642..0000000000 --- a/.github/workflows/coverage.yaml +++ /dev/null @@ -1,174 +0,0 @@ -name: Dozer Coverage - -on: - workflow_dispatch: - push: - tags: - - "v*.*.*" - -env: - CARGO_TERM_COLOR: always - -concurrency: - group: coverage/${{ github.head_ref || github.run_id }} - cancel-in-progress: true - -permissions: - id-token: write # This is required for requesting the JWT - contents: write # This is required for actions/checkout - -jobs: - # Run coverage - coverage: - timeout-minutes: 60 - runs-on: ubuntu-latest - services: - postgres: - image: debezium/postgres:13 - ports: - - 5434:5432 - env: - POSTGRES_DB: dozer_test - POSTGRES_USER: postgres - POSTGRES_PASSWORD: postgres - ALLOW_IP_RANGE: 0.0.0.0/0 - # command: postgres -c hba_file=/var/lib/stock-sample/pg_hba.conf - options: >- - --health-cmd pg_isready - --health-interval 10s - --health-timeout 5s - --health-retries 5 - - steps: - - name: Configure AWS credentials - uses: aws-actions/configure-aws-credentials@v2 - with: - role-to-assume: ${{ secrets.AWS_ROLE_TO_ASSUME }} - role-session-name: dozer-coverage - aws-region: us-east-2 - - - if: github.event_name == 'pull_request_target' - uses: actions/checkout@v3 - with: - ref: ${{ github.event.pull_request.head.sha }} - submodules: 'recursive' - - - if: github.event_name != 'pull_request_target' - uses: actions/checkout@v3 - with: - submodules: 'recursive' - - - name: Install stable with llvm-tools-preview - uses: actions-rs/toolchain@v1 - with: - profile: minimal - toolchain: stable - components: llvm-tools-preview - - - name: Download grcov - run: | - mkdir target - wget -O target/grcov https://dozer-ci.s3.ap-southeast-1.amazonaws.com/grcov-linux-amd64-v0.8.13 - chmod +x target/grcov - - - name: Install Protoc - uses: arduino/setup-protoc@v1 - with: - repo-token: ${{ secrets.GITHUB_TOKEN }} - - - name: ⚡ Cache - uses: actions/cache@v3 - with: - path: | - ~/.cargo/bin/ - ~/.cargo/.crates.toml - ~/.cargo/.crates2.json - ~/.cargo/.package-cache - ~/.cargo/registry/ - ~/.cargo/git/db/ - target/ - key: coverage-${{ runner.os }}-cargo-${{ hashFiles('Cargo.lock') }} - restore-keys: | - coverage-${{ runner.os }}-cargo-${{ hashFiles('Cargo.lock') }} - coverage-${{ runner.os }}-cargo- - - name: MongoDB in GitHub Actions - uses: supercharge/mongodb-github-action@1.8.0 - - - uses: ./.github/workflows/setup-snowflake-and-kafka - - - uses: ./.github/workflows/setup-mysql-and-mariadb - - - name: Run connectors tests - env: - CARGO_INCREMENTAL: "0" - RUSTFLAGS: "-Cinstrument-coverage" - LLVM_PROFILE_FILE: "cargo-test-%p-%m.profraw" - SN_SERVER: ${{ secrets.SN_SERVER }} - SN_USER: ${{ secrets.SN_USER }} - SN_PASSWORD: ${{ secrets.SN_PASSWORD }} - SN_DATABASE: ${{ secrets.SN_DATABASE }} - SN_WAREHOUSE: ${{ secrets.SN_WAREHOUSE }} - SN_DRIVER: ${{ secrets.SN_DRIVER }} - shell: bash - run: | - cargo test test_connector_ --lib --features snowflake,ethereum,kafka,python --no-fail-fast -- --ignored - - - name: Run tests - env: - CARGO_INCREMENTAL: "0" - RUSTFLAGS: "-Cinstrument-coverage" - LLVM_PROFILE_FILE: "cargo-test-%p-%m.profraw" - shell: bash - run: | - source ./dozer-tests/python_udf/virtualenv.sh - cargo test --features snowflake,ethereum,kafka,python,mongodb --no-fail-fast - - - name: Get current date - id: date - run: echo "::set-output name=date::$(date +'%Y-%m-%d')" - - - id: coverage - run: | - ./target/grcov . --binary-path ./target/debug/deps/ -s . -t lcov --branch --ignore-not-existing --ignore '../*' --ignore "/*" --ignore 'target/*' --ignore 'dozer-tests/*' -o coverage.lcov - echo "::set-output name=report::coverage.lcov" - - - uses: actions/upload-artifact@v3 - with: - name: coverage - path: | - ${{ steps.coverage.outputs.report }} - retention-days: 10 - - - if: github.event_name == 'pull_request_target' - name: Coveralls upload - uses: coverallsapp/github-action@master - with: - github-token: ${{ secrets.GITHUB_TOKEN }} - path-to-lcov: ${{ steps.coverage.outputs.report }} - git-commit: ${{ github.event.pull_request.head.sha }} - - - if: github.event_name != 'pull_request_target' - name: Coveralls upload - uses: coverallsapp/github-action@master - with: - github-token: ${{ secrets.GITHUB_TOKEN }} - path-to-lcov: ${{ steps.coverage.outputs.report }} - - discord_notification: - if: ${{ github.event_name == 'push' }} - runs-on: ubuntu-latest - steps: - - name: Discord notification - env: - DISCORD_WEBHOOK: ${{ secrets.DISCORD_GITHUB_WEBOOK }} - DISCORD_EMBEDS: '[ { - "title": " ${{ github.actor }} pushed to `${{ github.ref_name }}` :rocket:", - "author": { "icon_url": "https://avatars.githubusercontent.com/${{ github.actor }}", "name": "${{ github.actor }}", "url": "https://github.com/${{ github.actor }}" }, - "fields": [ - { "name": "Commit", "value": "[${{ github.event.head_commit.id }}](${{ github.event.head_commit.url }})"}, - { "name": "Repository", "value": "[getdozer/dozer](https://github.com/getdozer/dozer)" }, - { "name": "Message", "value": ${{ toJSON(github.event.head_commit.message) }}} - ], - "color": 990099 - }]' - uses: Ilshidur/action-discord@master diff --git a/.github/workflows/dozer.yaml b/.github/workflows/dozer.yaml deleted file mode 100644 index 156b01b3fe..0000000000 --- a/.github/workflows/dozer.yaml +++ /dev/null @@ -1,50 +0,0 @@ -name: Dozer CI - -on: - workflow_dispatch: - pull_request: - branches: [main] - merge_group: - -env: - CARGO_TERM_COLOR: always - -concurrency: - group: ci/${{ github.ref }} - cancel-in-progress: true - -jobs: - lint: - timeout-minutes: 60 - runs-on: - labels: ubuntu-latest - steps: - - uses: actions/checkout@v3 - with: - submodules: 'recursive' - - - name: Install minimal stable with clippy and rustfmt - uses: actions-rs/toolchain@v1 - with: - profile: minimal - toolchain: stable - components: rustfmt, clippy - - - name: Install Protoc - uses: arduino/setup-protoc@v1 - with: - repo-token: ${{ secrets.GITHUB_TOKEN }} - - - name: Rust cache - uses: swatinem/rust-cache@v2 - - - name: Clippy - run: | - cargo clippy --workspace --all-features --all-targets -- -D warnings - - - name: Lint - run: | - cargo fmt -- --check - - - name: Cargo Deny - uses: EmbarkStudios/cargo-deny-action@v1 diff --git a/.github/workflows/integration/docker-compose.yaml b/.github/workflows/integration/docker-compose.yaml deleted file mode 100644 index 709a0cd8b9..0000000000 --- a/.github/workflows/integration/docker-compose.yaml +++ /dev/null @@ -1,75 +0,0 @@ -version: '3.1' - -services: - dozer-tests-ubuntu-20-amd64: - build: - context: dockerfiles - dockerfile: ubuntu-20-amd64 - volumes: - - ${PWD}:/dozer - deploy: - resources: - limits: - cpus: '2' - memory: 8G - working_dir: /dozer - environment: - - DOZER_VERSION - command: sh /dozer/.github/workflows/integration/test-dozer-ubuntu.sh - dozer-tests-ubuntu-20-arm64: - build: - context: dockerfiles - dockerfile: ubuntu-20-arm64 - volumes: - - ${PWD}:/dozer - deploy: - resources: - limits: - cpus: '2' - memory: 8G - working_dir: /dozer - environment: - - DOZER_VERSION - command: sh /dozer/.github/workflows/integration/test-dozer-ubuntu.sh - dozer-tests-ubuntu-22-amd64: - build: - context: dockerfiles - dockerfile: ubuntu-22-amd64 - volumes: - - ${PWD}:/dozer - deploy: - resources: - limits: - cpus: '2' - memory: 8G - working_dir: /dozer - environment: - - DOZER_VERSION - command: sh /dozer/.github/workflows/integration/test-dozer-ubuntu.sh - dozer-tests-ubuntu-22-arm64: - build: - context: dockerfiles - dockerfile: ubuntu-22-arm64 - volumes: - - ${PWD}:/dozer - deploy: - resources: - limits: - cpus: '2' - memory: 8G - working_dir: /dozer - environment: - - DOZER_VERSION - command: sh /dozer/.github/workflows/integration/test-dozer-ubuntu.sh - run-tests: - image: alpine - command: echo 'All tests passed' - depends_on: - dozer-tests-ubuntu-20-amd64: - condition: service_completed_successfully - dozer-tests-ubuntu-20-arm64: - condition: service_completed_successfully - dozer-tests-ubuntu-22-amd64: - condition: service_completed_successfully - dozer-tests-ubuntu-22-arm64: - condition: service_completed_successfully diff --git a/.github/workflows/integration/dockerfiles/install-curl-ubuntu.sh b/.github/workflows/integration/dockerfiles/install-curl-ubuntu.sh deleted file mode 100644 index 9159738ff6..0000000000 --- a/.github/workflows/integration/dockerfiles/install-curl-ubuntu.sh +++ /dev/null @@ -1,4 +0,0 @@ -set -e - -apt update -apt install -y curl diff --git a/.github/workflows/integration/dockerfiles/install-dozer-ubuntu-amd64.sh b/.github/workflows/integration/dockerfiles/install-dozer-ubuntu-amd64.sh deleted file mode 100644 index bd46233358..0000000000 --- a/.github/workflows/integration/dockerfiles/install-dozer-ubuntu-amd64.sh +++ /dev/null @@ -1,4 +0,0 @@ -set -e - -curl -sLO https://github.com/getdozer/dozer/releases/latest/download/dozer-linux-amd64.deb -dpkg -i dozer-linux-amd64.deb diff --git a/.github/workflows/integration/dockerfiles/install-dozer-ubuntu-arm64.sh b/.github/workflows/integration/dockerfiles/install-dozer-ubuntu-arm64.sh deleted file mode 100644 index ee8d6d1003..0000000000 --- a/.github/workflows/integration/dockerfiles/install-dozer-ubuntu-arm64.sh +++ /dev/null @@ -1,4 +0,0 @@ -set -e - -curl -sLO https://github.com/getdozer/dozer/releases/latest/download/dozer-linux-aarch64.deb -dpkg -i dozer-linux-aarch64.deb diff --git a/.github/workflows/integration/dockerfiles/install-protoc-ubuntu-20-amd64.sh b/.github/workflows/integration/dockerfiles/install-protoc-ubuntu-20-amd64.sh deleted file mode 100644 index 99376b7e26..0000000000 --- a/.github/workflows/integration/dockerfiles/install-protoc-ubuntu-20-amd64.sh +++ /dev/null @@ -1,5 +0,0 @@ -set -e - -curl -sLO https://github.com/protocolbuffers/protobuf/releases/download/v22.2/protoc-22.2-linux-x86_64.zip -apt install -y unzip -unzip protoc-22.2-linux-x86_64.zip -d /usr/local diff --git a/.github/workflows/integration/dockerfiles/install-protoc-ubuntu-20-arm64.sh b/.github/workflows/integration/dockerfiles/install-protoc-ubuntu-20-arm64.sh deleted file mode 100644 index 05bc3386ce..0000000000 --- a/.github/workflows/integration/dockerfiles/install-protoc-ubuntu-20-arm64.sh +++ /dev/null @@ -1,5 +0,0 @@ -set -e - -curl -sLO https://github.com/protocolbuffers/protobuf/releases/download/v22.2/protoc-22.2-linux-aarch_64.zip -apt install -y unzip -unzip protoc-22.2-linux-aarch_64.zip -d /usr/local diff --git a/.github/workflows/integration/dockerfiles/install-protoc-ubuntu-22.sh b/.github/workflows/integration/dockerfiles/install-protoc-ubuntu-22.sh deleted file mode 100644 index 63084a40a3..0000000000 --- a/.github/workflows/integration/dockerfiles/install-protoc-ubuntu-22.sh +++ /dev/null @@ -1,3 +0,0 @@ -set -e - -apt install -y protobuf-compiler diff --git a/.github/workflows/integration/dockerfiles/ubuntu-20-amd64 b/.github/workflows/integration/dockerfiles/ubuntu-20-amd64 deleted file mode 100644 index fe3b719ff0..0000000000 --- a/.github/workflows/integration/dockerfiles/ubuntu-20-amd64 +++ /dev/null @@ -1,13 +0,0 @@ -FROM --platform=amd64 ubuntu:20.04 - -# Install curl -COPY install-curl-ubuntu.sh . -RUN sh install-curl-ubuntu.sh - -# Install Protoc -COPY install-protoc-ubuntu-20-amd64.sh . -RUN sh install-protoc-ubuntu-20-amd64.sh - -# Install Dozer -COPY install-dozer-ubuntu-amd64.sh . -RUN sh install-dozer-ubuntu-amd64.sh diff --git a/.github/workflows/integration/dockerfiles/ubuntu-20-arm64 b/.github/workflows/integration/dockerfiles/ubuntu-20-arm64 deleted file mode 100644 index 9d933fbdb8..0000000000 --- a/.github/workflows/integration/dockerfiles/ubuntu-20-arm64 +++ /dev/null @@ -1,13 +0,0 @@ -FROM --platform=arm64 ubuntu:20.04 - -# Install curl -COPY install-curl-ubuntu.sh . -RUN sh install-curl-ubuntu.sh - -# Install Protoc -COPY install-protoc-ubuntu-20-arm64.sh . -RUN sh install-protoc-ubuntu-20-arm64.sh - -# Install Dozer -COPY install-dozer-ubuntu-arm64.sh . -RUN sh install-dozer-ubuntu-arm64.sh diff --git a/.github/workflows/integration/dockerfiles/ubuntu-22-amd64 b/.github/workflows/integration/dockerfiles/ubuntu-22-amd64 deleted file mode 100644 index 90bc0ebd80..0000000000 --- a/.github/workflows/integration/dockerfiles/ubuntu-22-amd64 +++ /dev/null @@ -1,13 +0,0 @@ -FROM --platform=amd64 ubuntu:22.04 - -# Install curl -COPY install-curl-ubuntu.sh . -RUN sh install-curl-ubuntu.sh - -# Install Protoc -COPY install-protoc-ubuntu-22.sh . -RUN sh install-protoc-ubuntu-22.sh - -# Install Dozer -COPY install-dozer-ubuntu-amd64.sh . -RUN sh install-dozer-ubuntu-amd64.sh diff --git a/.github/workflows/integration/dockerfiles/ubuntu-22-arm64 b/.github/workflows/integration/dockerfiles/ubuntu-22-arm64 deleted file mode 100644 index 7b9db749d5..0000000000 --- a/.github/workflows/integration/dockerfiles/ubuntu-22-arm64 +++ /dev/null @@ -1,13 +0,0 @@ -FROM --platform=arm64 ubuntu:22.04 - -# Install curl -COPY install-curl-ubuntu.sh . -RUN sh install-curl-ubuntu.sh - -# Install Protoc -COPY install-protoc-ubuntu-22.sh . -RUN sh install-protoc-ubuntu-22.sh - -# Install Dozer -COPY install-dozer-ubuntu-arm64.sh . -RUN sh install-dozer-ubuntu-arm64.sh diff --git a/.github/workflows/integration/test-dozer-ubuntu.sh b/.github/workflows/integration/test-dozer-ubuntu.sh deleted file mode 100644 index 0203f8a5eb..0000000000 --- a/.github/workflows/integration/test-dozer-ubuntu.sh +++ /dev/null @@ -1,4 +0,0 @@ -set -e - -apt install -y build-essential -sh .github/workflows/integration/test-dozer.sh diff --git a/.github/workflows/integration/test-dozer.sh b/.github/workflows/integration/test-dozer.sh deleted file mode 100644 index 541c2092a1..0000000000 --- a/.github/workflows/integration/test-dozer.sh +++ /dev/null @@ -1,4 +0,0 @@ -set -e - -# Check if dozer version matches `DOZER_VERSION` -dozer -V | grep "$DOZER_VERSION" diff --git a/.github/workflows/setup-mysql-and-mariadb/action.yaml b/.github/workflows/setup-mysql-and-mariadb/action.yaml deleted file mode 100644 index 3e6495a138..0000000000 --- a/.github/workflows/setup-mysql-and-mariadb/action.yaml +++ /dev/null @@ -1,20 +0,0 @@ -name: Setup MySQL and MariaDB - -runs: - using: "composite" - steps: - - name: Run MySQL - shell: bash - run: | - docker run -d --name mysql \ - -e MYSQL_ROOT_PASSWORD=mysql -e MYSQL_ROOT_HOST=% -e MYSQL_DATABASE=test \ - -p 3306:3306 \ - mysql:8 --log-bin --binlog-format=row - - - name: Run MariaDB - shell: bash - run: | - docker run -d --name mariadb \ - -e MARIADB_ROOT_PASSWORD=mariadb -e MARIADB_ROOT_HOST=% -e MARIADB_DATABASE=test \ - -p 3307:3306 \ - mariadb:11 --log-bin --binlog-format=row diff --git a/.github/workflows/setup-snowflake-and-kafka/action.yaml b/.github/workflows/setup-snowflake-and-kafka/action.yaml deleted file mode 100644 index d92e2a812a..0000000000 --- a/.github/workflows/setup-snowflake-and-kafka/action.yaml +++ /dev/null @@ -1,11 +0,0 @@ -name: Setup snowflake and kafka (debezium) - -runs: - using: "composite" - steps: - - name: Install Snowflake ODBC driver - shell: bash - run: curl ${SNOWFLAKE_DRIVER_URL} -o snowflake_driver.deb && sudo dpkg -i snowflake_driver.deb - env: - SNOWFLAKE_DRIVER_URL: https://sfc-repo.snowflakecomputing.com/odbc/linux/2.25.7/snowflake-odbc-2.25.7.x86_64.deb - diff --git a/.gitignore b/.gitignore deleted file mode 100644 index 7bb4ff0f80..0000000000 --- a/.gitignore +++ /dev/null @@ -1,44 +0,0 @@ -/target -docker/sample_db -docker/pagila/ -.idea/ -db/embedded/ -db/test -dozer-ingestion/db/test/ -dozer-core/.data -.dozer-cloud - -.vscode -*.code-workspace -docker/data/ -dozer-sql/data -*.mdb - - -workspaces/ - -# Local configuration for development -/dozer-config.* -dozer-config.test.* -log4rs.yaml -.DS_Store -queries -dozer.lock - -.dozer/ -logs/ -*.log -register-postgres.test.json -*.db -config/tests/local/*.json -config/tests/local/*.yaml -dozer-ui - -# Integration test generated files -dozer-tests/src/e2e_tests/cases/*/local_runner/ -dozer-tests/src/e2e_tests/cases/*/buildkite_runner*/ - -dozer-ingestion/src/tests/cases/*/local_runner/ -dozer-ingestion/src/tests/cases/*/buildkite_runner*/ -# Connectors Benches YAML -dozer-ingestion/benches/connectors.yaml diff --git a/CODE_OF_CONDUCT.md b/CODE_OF_CONDUCT.md deleted file mode 100644 index 7c87cdcc13..0000000000 --- a/CODE_OF_CONDUCT.md +++ /dev/null @@ -1,76 +0,0 @@ -# Contributor Covenant Code of Conduct - -## Our Pledge - -In the interest of fostering an open and welcoming environment, we as -contributors and maintainers pledge to making participation in our project and -our community a harassment-free experience for everyone, regardless of age, body -size, disability, ethnicity, sex characteristics, gender identity and expression, -level of experience, education, socio-economic status, nationality, personal -appearance, race, religion, or sexual identity and orientation. - -## Our Standards - -Examples of behavior that contributes to creating a positive environment -include: - -* Using welcoming and inclusive language -* Being respectful of differing viewpoints and experiences -* Gracefully accepting constructive criticism -* Focusing on what is best for the community -* Showing empathy towards other community members - -Examples of unacceptable behavior by participants include: - -* The use of sexualized language or imagery and unwelcome sexual attention or - advances -* Trolling, insulting/derogatory comments, and personal or political attacks -* Public or private harassment -* Publishing others' private information, such as a physical or electronic - address, without explicit permission -* Other conduct which could reasonably be considered inappropriate in a - professional setting - -## Our Responsibilities - -Project maintainers are responsible for clarifying the standards of acceptable -behavior and are expected to take appropriate and fair corrective action in -response to any instances of unacceptable behavior. - -Project maintainers have the right and responsibility to remove, edit, or -reject comments, commits, code, wiki edits, issues, and other contributions -that are not aligned to this Code of Conduct, or to ban temporarily or -permanently any contributor for other behaviors that they deem inappropriate, -threatening, offensive, or harmful. - -## Scope - -This Code of Conduct applies both within project spaces and in public spaces -when an individual is representing the project or its community. Examples of -representing a project or community include using an official project e-mail -address, posting via an official social media account, or acting as an appointed -representative at an online or offline event. Representation of a project may be -further defined and clarified by project maintainers. - -## Enforcement - -Instances of abusive, harassing, or otherwise unacceptable behavior may be -reported by contacting the project team at hello@getdozer.io. All -complaints will be reviewed and investigated and will result in a response that -is deemed necessary and appropriate to the circumstances. The project team is -obligated to maintain confidentiality with regard to the reporter of an incident. -Further details of specific enforcement policies may be posted separately. - -Project maintainers who do not follow or enforce the Code of Conduct in good -faith may face temporary or permanent repercussions as determined by other -members of the project's leadership. - -## Attribution - -This Code of Conduct is adapted from the [Contributor Covenant][homepage], version 1.4, -available at https://www.contributor-covenant.org/version/1/4/code-of-conduct.html - -[homepage]: https://www.contributor-covenant.org - -For answers to common questions about this code of conduct, see -https://www.contributor-covenant.org/faq \ No newline at end of file diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md deleted file mode 100644 index ac0c017b62..0000000000 --- a/CONTRIBUTING.md +++ /dev/null @@ -1,4 +0,0 @@ -# Contributing - -Thank you for your interest in contributing to Dozer! Contributions in various ways are most welcome. -Please refer to [Contributing](https://getdozer.io/docs/contributing/overview) for more details. diff --git a/Cargo.lock b/Cargo.lock deleted file mode 100644 index 55a96a2c33..0000000000 --- a/Cargo.lock +++ /dev/null @@ -1,10440 +0,0 @@ -# This file is automatically @generated by Cargo. -# It is not intended for manual editing. -version = 3 - -[[package]] -name = "Inflector" -version = "0.11.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fe438c63458706e03479442743baae6c88256498e6431708f6dfc520a26515d3" -dependencies = [ - "lazy_static", - "regex", -] - -[[package]] -name = "actix-codec" -version = "0.5.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5f7b0a21988c1bf877cf4759ef5ddaac04c1c9fe808c9142ecb78ba97d97a28a" -dependencies = [ - "bitflags 2.5.0", - "bytes", - "futures-core", - "futures-sink", - "memchr", - "pin-project-lite", - "tokio", - "tokio-util", - "tracing", -] - -[[package]] -name = "actix-files" -version = "0.6.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bf0bdd6ff79de7c9a021f5d9ea79ce23e108d8bfc9b49b5b4a2cf6fad5a35212" -dependencies = [ - "actix-http", - "actix-service", - "actix-utils", - "actix-web", - "bitflags 2.5.0", - "bytes", - "derive_more", - "futures-core", - "http-range", - "log", - "mime", - "mime_guess", - "percent-encoding", - "pin-project-lite", - "v_htmlescape", -] - -[[package]] -name = "actix-http" -version = "3.6.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d223b13fd481fc0d1f83bb12659ae774d9e3601814c68a0bc539731698cca743" -dependencies = [ - "actix-codec", - "actix-rt", - "actix-service", - "actix-utils", - "ahash 0.8.11", - "base64 0.21.7", - "bitflags 2.5.0", - "brotli", - "bytes", - "bytestring", - "derive_more", - "encoding_rs", - "flate2", - "futures-core", - "h2 0.3.26", - "http 0.2.12", - "httparse", - "httpdate", - "itoa", - "language-tags", - "local-channel", - "mime", - "percent-encoding", - "pin-project-lite", - "rand", - "sha1", - "smallvec", - "tokio", - "tokio-util", - "tracing", - "zstd", -] - -[[package]] -name = "actix-macros" -version = "0.2.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e01ed3140b2f8d422c68afa1ed2e85d996ea619c988ac834d255db32138655cb" -dependencies = [ - "quote", - "syn 2.0.53", -] - -[[package]] -name = "actix-router" -version = "0.5.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d22475596539443685426b6bdadb926ad0ecaefdfc5fb05e5e3441f15463c511" -dependencies = [ - "bytestring", - "http 0.2.12", - "regex", - "serde", - "tracing", -] - -[[package]] -name = "actix-rt" -version = "2.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "28f32d40287d3f402ae0028a9d54bef51af15c8769492826a69d28f81893151d" -dependencies = [ - "futures-core", - "tokio", -] - -[[package]] -name = "actix-server" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3eb13e7eef0423ea6eab0e59f6c72e7cb46d33691ad56a726b3cd07ddec2c2d4" -dependencies = [ - "actix-rt", - "actix-service", - "actix-utils", - "futures-core", - "futures-util", - "mio", - "socket2 0.5.6", - "tokio", - "tracing", -] - -[[package]] -name = "actix-service" -version = "2.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3b894941f818cfdc7ccc4b9e60fa7e53b5042a2e8567270f9147d5591893373a" -dependencies = [ - "futures-core", - "paste", - "pin-project-lite", -] - -[[package]] -name = "actix-utils" -version = "3.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "88a1dcdff1466e3c2488e1cb5c36a71822750ad43839937f85d2f4d9f8b705d8" -dependencies = [ - "local-waker", - "pin-project-lite", -] - -[[package]] -name = "actix-web" -version = "4.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "43a6556ddebb638c2358714d853257ed226ece6023ef9364f23f0c70737ea984" -dependencies = [ - "actix-codec", - "actix-http", - "actix-macros", - "actix-router", - "actix-rt", - "actix-server", - "actix-service", - "actix-utils", - "actix-web-codegen", - "ahash 0.8.11", - "bytes", - "bytestring", - "cfg-if", - "cookie", - "derive_more", - "encoding_rs", - "futures-core", - "futures-util", - "itoa", - "language-tags", - "log", - "mime", - "once_cell", - "pin-project-lite", - "regex", - "serde", - "serde_json", - "serde_urlencoded", - "smallvec", - "socket2 0.5.6", - "time", - "url", -] - -[[package]] -name = "actix-web-codegen" -version = "4.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eb1f50ebbb30eca122b188319a4398b3f7bb4a8cdf50ecfb73bfc6a3c3ce54f5" -dependencies = [ - "actix-router", - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "addr2line" -version = "0.21.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8a30b2e23b9e17a9f90641c7ab1549cd9b44f296d3ccbf309d2863cfe398a0cb" -dependencies = [ - "gimli", -] - -[[package]] -name = "adler" -version = "1.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f26201604c87b1e01bd3d98f8d5d9a8fcbb815e8cedb41ffccbeb4bf593a35fe" - -[[package]] -name = "adler32" -version = "1.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "aae1277d39aeec15cb388266ecc24b11c80469deae6067e17a1a7aa9e5c1f234" - -[[package]] -name = "aead" -version = "0.5.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d122413f284cf2d62fb1b7db97e02edb8cda96d769b16e443a4f6195e35662b0" -dependencies = [ - "crypto-common", - "generic-array", -] - -[[package]] -name = "aes" -version = "0.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ac1f845298e95f983ff1944b728ae08b8cebab80d684f0a832ed0fc74dfa27e2" -dependencies = [ - "cfg-if", - "cipher", - "cpufeatures", -] - -[[package]] -name = "aes-gcm" -version = "0.10.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "831010a0f742e1209b3bcea8fab6a8e149051ba6099432c8cb2cc117dec3ead1" -dependencies = [ - "aead", - "aes", - "cipher", - "ctr", - "ghash", - "subtle", -] - -[[package]] -name = "aes-kw" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "69fa2b352dcefb5f7f3a5fb840e02665d311d878955380515e4fd50095dd3d8c" -dependencies = [ - "aes", -] - -[[package]] -name = "ahash" -version = "0.7.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "891477e0c6a8957309ee5c45a6368af3ae14bb510732d2684ffa19af310920f9" -dependencies = [ - "getrandom", - "once_cell", - "version_check", -] - -[[package]] -name = "ahash" -version = "0.8.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e89da841a80418a9b391ebaea17f5c112ffaaa96f621d2c285b5174da76b9011" -dependencies = [ - "cfg-if", - "const-random", - "getrandom", - "once_cell", - "version_check", - "zerocopy", -] - -[[package]] -name = "aho-corasick" -version = "1.1.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8e60d3430d3a69478ad0993f19238d2df97c507009a52b3c10addcd7f6bcb916" -dependencies = [ - "memchr", -] - -[[package]] -name = "alloc-no-stdlib" -version = "2.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cc7bb162ec39d46ab1ca8c77bf72e890535becd1751bb45f64c597edb4c8c6b3" - -[[package]] -name = "alloc-stdlib" -version = "0.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "94fb8275041c72129eb51b7d0322c29b8387a0386127718b096429201a5d6ece" -dependencies = [ - "alloc-no-stdlib", -] - -[[package]] -name = "allocator-api2" -version = "0.2.16" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0942ffc6dcaadf03badf6e6a2d0228460359d5e34b57ccdc720b7382dfbd5ec5" - -[[package]] -name = "android-tzdata" -version = "0.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e999941b234f3131b00bc13c22d06e8c5ff726d1b6318ac7eb276997bbb4fef0" - -[[package]] -name = "android_system_properties" -version = "0.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "819e7219dbd41043ac279b19830f2efc897156490d7fd6ea916720117ee66311" -dependencies = [ - "libc", -] - -[[package]] -name = "anes" -version = "0.1.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4b46cbb362ab8752921c97e041f5e366ee6297bd428a31275b9fcf1e380f7299" - -[[package]] -name = "anstream" -version = "0.6.13" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d96bd03f33fe50a863e394ee9718a706f988b9079b20c3784fb726e7678b62fb" -dependencies = [ - "anstyle", - "anstyle-parse", - "anstyle-query", - "anstyle-wincon", - "colorchoice", - "utf8parse", -] - -[[package]] -name = "anstyle" -version = "1.0.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8901269c6307e8d93993578286ac0edf7f195079ffff5ebdeea6a59ffb7e36bc" - -[[package]] -name = "anstyle-parse" -version = "0.2.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c75ac65da39e5fe5ab759307499ddad880d724eed2f6ce5b5e8a26f4f387928c" -dependencies = [ - "utf8parse", -] - -[[package]] -name = "anstyle-query" -version = "1.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e28923312444cdd728e4738b3f9c9cac739500909bb3d3c94b43551b16517648" -dependencies = [ - "windows-sys 0.52.0", -] - -[[package]] -name = "anstyle-wincon" -version = "3.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1cd54b81ec8d6180e24654d0b371ad22fc3dd083b6ff8ba325b72e00c87660a7" -dependencies = [ - "anstyle", - "windows-sys 0.52.0", -] - -[[package]] -name = "anyhow" -version = "1.0.81" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0952808a6c2afd1aa8947271f3a60f1a6763c7b912d210184c5149b5cf147247" - -[[package]] -name = "apache-avro" -version = "0.16.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ceb7c683b2f8f40970b70e39ff8be514c95b96fcb9c4af87e1ed2cb2e10801a0" -dependencies = [ - "digest 0.10.7", - "lazy_static", - "libflate", - "log", - "num-bigint", - "quad-rand", - "rand", - "regex-lite", - "serde", - "serde_json", - "strum", - "strum_macros", - "thiserror", - "typed-builder 0.16.2", - "uuid", -] - -[[package]] -name = "approx" -version = "0.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cab112f0a86d568ea0e627cc1d6be74a1e9cd55214684db5561995f6dad897c6" -dependencies = [ - "num-traits", -] - -[[package]] -name = "arbitrary" -version = "1.3.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7d5a26814d8dcb93b0e5a0ff3c6d80a8843bafb21b39e8e18a6f05471870e110" -dependencies = [ - "derive_arbitrary", -] - -[[package]] -name = "arrayref" -version = "0.3.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6b4930d2cb77ce62f89ee5d5289b4ac049559b1c45539271f5ed4fdc7db34545" - -[[package]] -name = "arrayvec" -version = "0.7.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "96d30a06541fbafbc7f82ed10c06164cfbd2c401138f6addd8404629c4b16711" - -[[package]] -name = "arrow" -version = "50.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "aa285343fba4d829d49985bdc541e3789cf6000ed0e84be7c039438df4a4e78c" -dependencies = [ - "arrow-arith", - "arrow-array", - "arrow-buffer", - "arrow-cast", - "arrow-csv", - "arrow-data", - "arrow-ipc", - "arrow-json", - "arrow-ord", - "arrow-row", - "arrow-schema", - "arrow-select", - "arrow-string", -] - -[[package]] -name = "arrow-arith" -version = "50.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "753abd0a5290c1bcade7c6623a556f7d1659c5f4148b140b5b63ce7bd1a45705" -dependencies = [ - "arrow-array", - "arrow-buffer", - "arrow-data", - "arrow-schema", - "chrono", - "half", - "num", -] - -[[package]] -name = "arrow-array" -version = "50.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d390feeb7f21b78ec997a4081a025baef1e2e0d6069e181939b61864c9779609" -dependencies = [ - "ahash 0.8.11", - "arrow-buffer", - "arrow-data", - "arrow-schema", - "chrono", - "chrono-tz", - "half", - "hashbrown 0.14.3", - "num", -] - -[[package]] -name = "arrow-buffer" -version = "50.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "69615b061701bcdffbc62756bc7e85c827d5290b472b580c972ebbbf690f5aa4" -dependencies = [ - "bytes", - "half", - "num", -] - -[[package]] -name = "arrow-cast" -version = "50.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e448e5dd2f4113bf5b74a1f26531708f5edcacc77335b7066f9398f4bcf4cdef" -dependencies = [ - "arrow-array", - "arrow-buffer", - "arrow-data", - "arrow-schema", - "arrow-select", - "base64 0.21.7", - "chrono", - "comfy-table", - "half", - "lexical-core", - "num", -] - -[[package]] -name = "arrow-csv" -version = "50.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "46af72211f0712612f5b18325530b9ad1bfbdc87290d5fbfd32a7da128983781" -dependencies = [ - "arrow-array", - "arrow-buffer", - "arrow-cast", - "arrow-data", - "arrow-schema", - "chrono", - "csv", - "csv-core", - "lazy_static", - "lexical-core", - "regex", -] - -[[package]] -name = "arrow-data" -version = "50.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "67d644b91a162f3ad3135ce1184d0a31c28b816a581e08f29e8e9277a574c64e" -dependencies = [ - "arrow-buffer", - "arrow-schema", - "half", - "num", -] - -[[package]] -name = "arrow-ipc" -version = "50.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "03dea5e79b48de6c2e04f03f62b0afea7105be7b77d134f6c5414868feefb80d" -dependencies = [ - "arrow-array", - "arrow-buffer", - "arrow-cast", - "arrow-data", - "arrow-schema", - "flatbuffers", - "lz4_flex", -] - -[[package]] -name = "arrow-json" -version = "50.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8950719280397a47d37ac01492e3506a8a724b3fb81001900b866637a829ee0f" -dependencies = [ - "arrow-array", - "arrow-buffer", - "arrow-cast", - "arrow-data", - "arrow-schema", - "chrono", - "half", - "indexmap 2.2.5", - "lexical-core", - "num", - "serde", - "serde_json", -] - -[[package]] -name = "arrow-ord" -version = "50.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ed9630979034077982d8e74a942b7ac228f33dd93a93b615b4d02ad60c260be" -dependencies = [ - "arrow-array", - "arrow-buffer", - "arrow-data", - "arrow-schema", - "arrow-select", - "half", - "num", -] - -[[package]] -name = "arrow-row" -version = "50.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "007035e17ae09c4e8993e4cb8b5b96edf0afb927cd38e2dff27189b274d83dcf" -dependencies = [ - "ahash 0.8.11", - "arrow-array", - "arrow-buffer", - "arrow-data", - "arrow-schema", - "half", - "hashbrown 0.14.3", -] - -[[package]] -name = "arrow-schema" -version = "50.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ff3e9c01f7cd169379d269f926892d0e622a704960350d09d331be3ec9e0029" -dependencies = [ - "serde", -] - -[[package]] -name = "arrow-select" -version = "50.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ce20973c1912de6514348e064829e50947e35977bb9d7fb637dc99ea9ffd78c" -dependencies = [ - "ahash 0.8.11", - "arrow-array", - "arrow-buffer", - "arrow-data", - "arrow-schema", - "num", -] - -[[package]] -name = "arrow-string" -version = "50.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "00f3b37f2aeece31a2636d1b037dabb69ef590e03bdc7eb68519b51ec86932a7" -dependencies = [ - "arrow-array", - "arrow-buffer", - "arrow-data", - "arrow-schema", - "arrow-select", - "num", - "regex", - "regex-syntax 0.8.2", -] - -[[package]] -name = "ast_node" -version = "0.9.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c3e3e06ec6ac7d893a0db7127d91063ad7d9da8988f8a1a256f03729e6eec026" -dependencies = [ - "proc-macro2", - "quote", - "swc_macros_common", - "syn 2.0.53", -] - -[[package]] -name = "async-compression" -version = "0.4.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a116f46a969224200a0a97f29cfd4c50e7534e4b4826bd23ea2c3c533039c82c" -dependencies = [ - "brotli", - "bzip2", - "flate2", - "futures-core", - "futures-io", - "memchr", - "pin-project-lite", - "tokio", - "xz2", - "zstd", - "zstd-safe", -] - -[[package]] -name = "async-recursion" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "30c5ef0ede93efbf733c1a727f3b6b5a1060bbedd5600183e66f6e4be4af0ec5" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "async-stream" -version = "0.3.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cd56dd203fef61ac097dd65721a419ddccb106b2d2b70ba60a6b529f03961a51" -dependencies = [ - "async-stream-impl", - "futures-core", - "pin-project-lite", -] - -[[package]] -name = "async-stream-impl" -version = "0.3.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "16e62a023e7c117e27523144c5d2459f4397fcc3cab0085af8e2224f643a0193" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "async-trait" -version = "0.1.78" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "461abc97219de0eaaf81fe3ef974a540158f3d079c2ab200f891f1a2ef201e85" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "atomic" -version = "0.5.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c59bdb34bc650a32731b31bd8f0829cc15d24a708ee31559e0bb34f2bc320cba" - -[[package]] -name = "atomic-polyfill" -version = "1.0.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8cf2bce30dfe09ef0bfaef228b9d414faaf7e563035494d7fe092dba54b300f4" -dependencies = [ - "critical-section", -] - -[[package]] -name = "autocfg" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d468802bab17cbc0cc575e9b053f41e72aa36bfa6b7f55e3529ffa43161b97fa" - -[[package]] -name = "axum" -version = "0.6.20" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3b829e4e32b91e643de6eafe82b1d90675f5874230191a4ffbc1b336dec4d6bf" -dependencies = [ - "async-trait", - "axum-core", - "bitflags 1.3.2", - "bytes", - "futures-util", - "http 0.2.12", - "http-body 0.4.6", - "hyper 0.14.28", - "itoa", - "matchit", - "memchr", - "mime", - "percent-encoding", - "pin-project-lite", - "rustversion", - "serde", - "sync_wrapper", - "tower", - "tower-layer", - "tower-service", -] - -[[package]] -name = "axum-core" -version = "0.3.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "759fa577a247914fd3f7f76d62972792636412fbfd634cd452f6a385a74d2d2c" -dependencies = [ - "async-trait", - "bytes", - "futures-util", - "http 0.2.12", - "http-body 0.4.6", - "mime", - "rustversion", - "tower-layer", - "tower-service", -] - -[[package]] -name = "backtrace" -version = "0.3.69" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2089b7e3f35b9dd2d0ed921ead4f6d318c27680d4a5bd167b3ee120edb105837" -dependencies = [ - "addr2line", - "cc", - "cfg-if", - "libc", - "miniz_oxide", - "object", - "rustc-demangle", -] - -[[package]] -name = "base16ct" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4c7f02d4ea65f2c1853089ffd8d2787bdbc63de2f0d29dedbcf8ccdfa0ccd4cf" - -[[package]] -name = "base64" -version = "0.13.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9e1b586273c5702936fe7b7d6896644d8be71e6314cfe09d3167c95f712589e8" - -[[package]] -name = "base64" -version = "0.21.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9d297deb1925b89f2ccc13d7635fa0714f12c87adce1c75356b39ca9b7178567" - -[[package]] -name = "base64-simd" -version = "0.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "781dd20c3aff0bd194fe7d2a977dd92f21c173891f3a03b677359e5fa457e5d5" -dependencies = [ - "simd-abstraction", -] - -[[package]] -name = "base64-simd" -version = "0.8.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "339abbe78e73178762e23bea9dfd08e697eb3f3301cd4be981c0f78ba5859195" -dependencies = [ - "outref 0.5.1", - "vsimd", -] - -[[package]] -name = "base64ct" -version = "1.6.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8c3c1a368f70d6cf7302d78f8f7093da241fb8e8807c05cc9e51a125895a6d5b" - -[[package]] -name = "bcder" -version = "0.7.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c627747a6774aab38beb35990d88309481378558875a41da1a4b2e373c906ef0" -dependencies = [ - "bytes", - "smallvec", -] - -[[package]] -name = "beef" -version = "0.5.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3a8241f3ebb85c056b509d4327ad0358fbbba6ffb340bf388f26350aeda225b1" - -[[package]] -name = "better_scoped_tls" -version = "0.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "794edcc9b3fb07bb4aecaa11f093fd45663b4feadb782d68303a2268bc2701de" -dependencies = [ - "scoped-tls", -] - -[[package]] -name = "bigdecimal" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a6773ddc0eafc0e509fb60e48dff7f450f8e674a0686ae8605e8d9901bd5eefa" -dependencies = [ - "num-bigint", - "num-integer", - "num-traits", - "serde", -] - -[[package]] -name = "bigdecimal" -version = "0.4.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9324c8014cd04590682b34f1e9448d38f0674d0f7b2dc553331016ef0e4e9ebc" -dependencies = [ - "autocfg", - "libm", - "num-bigint", - "num-integer", - "num-traits", -] - -[[package]] -name = "bincode" -version = "1.3.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b1f45e9417d87227c7a56d22e471c6206462cba514c7590c09aff4cf6d1ddcad" -dependencies = [ - "serde", -] - -[[package]] -name = "bincode" -version = "2.0.0-rc.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f11ea1a0346b94ef188834a65c068a03aec181c94896d481d7a0a40d85b0ce95" -dependencies = [ - "bincode_derive", - "serde", -] - -[[package]] -name = "bincode_derive" -version = "2.0.0-rc.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e30759b3b99a1b802a7a3aa21c85c3ded5c28e1c83170d82d70f08bbf7f3e4c" -dependencies = [ - "virtue", -] - -[[package]] -name = "bindgen" -version = "0.69.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a00dc851838a2120612785d195287475a3ac45514741da670b735818822129a0" -dependencies = [ - "bitflags 2.5.0", - "cexpr", - "clang-sys", - "itertools 0.12.1", - "lazy_static", - "lazycell", - "proc-macro2", - "quote", - "regex", - "rustc-hash", - "shlex", - "syn 2.0.53", -] - -[[package]] -name = "bit-set" -version = "0.5.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0700ddab506f33b20a03b13996eccd309a48e5ff77d0d95926aa0210fb4e95f1" -dependencies = [ - "bit-vec", -] - -[[package]] -name = "bit-vec" -version = "0.6.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "349f9b6a179ed607305526ca489b34ad0a41aed5f7980fa90eb03160b69598fb" - -[[package]] -name = "bitflags" -version = "1.3.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a" - -[[package]] -name = "bitflags" -version = "2.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cf4b9d6a944f767f8e5e0db018570623c85f3d925ac718db4e06d0187adb21c1" - -[[package]] -name = "bitvec" -version = "1.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1bc2832c24239b0141d5674bb9174f9d68a8b5b3f2753311927c172ca46f7e9c" -dependencies = [ - "funty", - "radium", - "tap", - "wyz", -] - -[[package]] -name = "blake2" -version = "0.10.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "46502ad458c9a52b69d4d4d32775c788b7a1b85e8bc9d482d92250fc0e3f8efe" -dependencies = [ - "digest 0.10.7", -] - -[[package]] -name = "blake3" -version = "1.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "30cca6d3674597c30ddf2c587bf8d9d65c9a84d2326d941cc79c9842dfe0ef52" -dependencies = [ - "arrayref", - "arrayvec", - "cc", - "cfg-if", - "constant_time_eq", -] - -[[package]] -name = "block-buffer" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4152116fd6e9dadb291ae18fc1ec3575ed6d84c29642d97890f4b4a3417297e4" -dependencies = [ - "generic-array", -] - -[[package]] -name = "block-buffer" -version = "0.10.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71" -dependencies = [ - "generic-array", -] - -[[package]] -name = "block-padding" -version = "0.3.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a8894febbff9f758034a5b8e12d87918f56dfc64a8e1fe757d65e29041538d93" -dependencies = [ - "generic-array", -] - -[[package]] -name = "borsh" -version = "1.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f58b559fd6448c6e2fd0adb5720cd98a2506594cafa4737ff98c396f3e82f667" -dependencies = [ - "borsh-derive", - "cfg_aliases", -] - -[[package]] -name = "borsh-derive" -version = "1.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7aadb5b6ccbd078890f6d7003694e33816e6b784358f18e15e7e6d9f065a57cd" -dependencies = [ - "once_cell", - "proc-macro-crate 3.1.0", - "proc-macro2", - "quote", - "syn 2.0.53", - "syn_derive", -] - -[[package]] -name = "brotli" -version = "3.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d640d25bc63c50fb1f0b545ffd80207d2e10a4c965530809b40ba3386825c391" -dependencies = [ - "alloc-no-stdlib", - "alloc-stdlib", - "brotli-decompressor", -] - -[[package]] -name = "brotli-decompressor" -version = "2.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4e2e4afe60d7dd600fdd3de8d0f08c2b7ec039712e3b6137ff98b7004e82de4f" -dependencies = [ - "alloc-no-stdlib", - "alloc-stdlib", -] - -[[package]] -name = "bson" -version = "2.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ce21468c1c9c154a85696bb25c20582511438edb6ad67f846ba1378ffdd80222" -dependencies = [ - "ahash 0.8.11", - "base64 0.13.1", - "bitvec", - "hex", - "indexmap 2.2.5", - "js-sys", - "once_cell", - "rand", - "serde", - "serde_bytes", - "serde_json", - "time", - "uuid", -] - -[[package]] -name = "btoi" -version = "0.4.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9dd6407f73a9b8b6162d8a2ef999fe6afd7cc15902ebf42c5cd296addf17e0ad" -dependencies = [ - "num-traits", -] - -[[package]] -name = "bumpalo" -version = "3.15.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7ff69b9dd49fd426c69a0db9fc04dd934cdb6645ff000864d98f7e2af8830eaa" - -[[package]] -name = "byte-slice-cast" -version = "1.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c3ac9f8b63eca6fd385229b3675f6cc0dc5c8a5c8a54a59d4f52ffd670d87b0c" - -[[package]] -name = "bytecheck" -version = "0.6.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "23cdc57ce23ac53c931e88a43d06d070a6fd142f2617be5855eb75efc9beb1c2" -dependencies = [ - "bytecheck_derive", - "ptr_meta", - "simdutf8", -] - -[[package]] -name = "bytecheck_derive" -version = "0.6.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3db406d29fbcd95542e92559bed4d8ad92636d1ca8b3b72ede10b4bcc010e659" -dependencies = [ - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "bytemuck" -version = "1.15.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5d6d68c57235a3a081186990eca2867354726650f42f7516ca50c28d6281fd15" - -[[package]] -name = "byteorder" -version = "1.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b" - -[[package]] -name = "bytes" -version = "1.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a2bd12c1caf447e69cd4528f47f94d203fd2582878ecb9e9465484c4148a8223" - -[[package]] -name = "bytestring" -version = "1.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "74d80203ea6b29df88012294f62733de21cfeab47f17b41af3a38bc30a03ee72" -dependencies = [ - "bytes", -] - -[[package]] -name = "bzip2" -version = "0.4.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bdb116a6ef3f6c3698828873ad02c3014b3c85cadb88496095628e3ef1e347f8" -dependencies = [ - "bzip2-sys", - "libc", -] - -[[package]] -name = "bzip2-sys" -version = "0.1.11+1.0.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "736a955f3fa7875102d57c82b8cac37ec45224a07fd32d58f9f7a186b6cd4cdc" -dependencies = [ - "cc", - "libc", - "pkg-config", -] - -[[package]] -name = "camino" -version = "1.1.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c59e92b5a388f549b863a7bea62612c09f24c8393560709a54558a9abdfb3b9c" - -[[package]] -name = "cast" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "37b2a672a2cb129a2e41c10b1224bb368f9f37a2b16b612598138befd7b37eb5" - -[[package]] -name = "cbc" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "26b52a9543ae338f279b96b0b9fed9c8093744685043739079ce85cd58f289a6" -dependencies = [ - "cipher", -] - -[[package]] -name = "cc" -version = "1.0.90" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8cd6604a82acf3039f1144f54b8eb34e91ffba622051189e71b781822d5ee1f5" -dependencies = [ - "jobserver", - "libc", -] - -[[package]] -name = "cesu8" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6d43a04d8753f35258c91f8ec639f792891f748a1edbd759cf1dcea3382ad83c" - -[[package]] -name = "cexpr" -version = "0.6.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6fac387a98bb7c37292057cffc56d62ecb629900026402633ae9160df93a8766" -dependencies = [ - "nom", -] - -[[package]] -name = "cfg-if" -version = "1.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "baf1de4339761588bc0619e3cbc0120ee582ebb74b53b4efbf79117bd2da40fd" - -[[package]] -name = "cfg_aliases" -version = "0.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fd16c4719339c4530435d38e511904438d07cce7950afa3718a84ac36c10e89e" - -[[package]] -name = "chrono" -version = "0.4.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5bc015644b92d5890fab7489e49d21f879d5c990186827d42ec511919404f38b" -dependencies = [ - "android-tzdata", - "arbitrary", - "iana-time-zone", - "js-sys", - "num-traits", - "serde", - "wasm-bindgen", - "windows-targets 0.52.4", -] - -[[package]] -name = "chrono-tz" -version = "0.8.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d59ae0466b83e838b81a54256c39d5d7c20b9d7daa10510a242d9b75abd5936e" -dependencies = [ - "chrono", - "chrono-tz-build", - "phf", -] - -[[package]] -name = "chrono-tz-build" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "433e39f13c9a060046954e0592a8d0a4bcb1040125cbf91cb8ee58964cfb350f" -dependencies = [ - "parse-zoneinfo", - "phf", - "phf_codegen", -] - -[[package]] -name = "ciborium" -version = "0.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "42e69ffd6f0917f5c029256a24d0161db17cea3997d185db0d35926308770f0e" -dependencies = [ - "ciborium-io", - "ciborium-ll", - "serde", -] - -[[package]] -name = "ciborium-io" -version = "0.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "05afea1e0a06c9be33d539b876f1ce3692f4afea2cb41f740e7743225ed1c757" - -[[package]] -name = "ciborium-ll" -version = "0.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "57663b653d948a338bfb3eeba9bb2fd5fcfaecb9e199e87e1eda4d9e8b240fd9" -dependencies = [ - "ciborium-io", - "half", -] - -[[package]] -name = "cipher" -version = "0.4.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "773f3b9af64447d2ce9850330c473515014aa235e6a783b02db81ff39e4a3dad" -dependencies = [ - "crypto-common", - "inout", -] - -[[package]] -name = "clang-sys" -version = "1.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "67523a3b4be3ce1989d607a828d036249522dd9c1c8de7f4dd2dae43a37369d1" -dependencies = [ - "glob", - "libc", - "libloading 0.8.3", -] - -[[package]] -name = "clap" -version = "4.5.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "949626d00e063efc93b6dca932419ceb5432f99769911c0b995f7e884c778813" -dependencies = [ - "clap_builder", - "clap_derive", -] - -[[package]] -name = "clap_builder" -version = "4.5.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ae129e2e766ae0ec03484e609954119f123cc1fe650337e155d03b022f24f7b4" -dependencies = [ - "anstream", - "anstyle", - "clap_lex", - "strsim 0.11.0", -] - -[[package]] -name = "clap_derive" -version = "4.5.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "90239a040c80f5e14809ca132ddc4176ab33d5e17e49691793296e3fcb34d72f" -dependencies = [ - "heck 0.5.0", - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "clap_lex" -version = "0.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "98cc8fbded0c607b7ba9dd60cd98df59af97e84d24e49c8557331cfc26d301ce" - -[[package]] -name = "clickhouse-rs" -version = "1.1.0-alpha.1" -source = "git+https://github.com/getdozer/clickhouse-rs#d85b606a1d924c2d2ec033f8cac557799204b46d" -dependencies = [ - "byteorder", - "cfg-if", - "chrono", - "chrono-tz", - "clickhouse-rs-cityhash-sys", - "combine", - "crossbeam", - "either", - "futures-core", - "futures-sink", - "futures-util", - "hostname", - "lazy_static", - "log", - "lz4", - "percent-encoding", - "pin-project", - "thiserror", - "tokio", - "url", - "uuid", -] - -[[package]] -name = "clickhouse-rs-cityhash-sys" -version = "0.1.2" -source = "git+https://github.com/getdozer/clickhouse-rs#d85b606a1d924c2d2ec033f8cac557799204b46d" -dependencies = [ - "cc", -] - -[[package]] -name = "clipboard-win" -version = "5.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d517d4b86184dbb111d3556a10f1c8a04da7428d2987bf1081602bf11c3aa9ee" -dependencies = [ - "error-code", -] - -[[package]] -name = "cmake" -version = "0.1.50" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a31c789563b815f77f4250caee12365734369f942439b7defd71e18a48197130" -dependencies = [ - "cc", -] - -[[package]] -name = "colorchoice" -version = "1.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "acbf1af155f9b9ef647e42cdc158db4b64a1b61f743629225fde6f3e0be2a7c7" - -[[package]] -name = "combine" -version = "4.6.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "35ed6e9d84f0b51a7f52daf1c7d71dd136fd7a3f41a8462b8cdb8c78d920fad4" -dependencies = [ - "bytes", - "memchr", -] - -[[package]] -name = "comfy-table" -version = "7.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7c64043d6c7b7a4c58e39e7efccfdea7b93d885a795d0c054a69dbbf4dd52686" -dependencies = [ - "strum", - "strum_macros", - "unicode-width", -] - -[[package]] -name = "console" -version = "0.15.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0e1f83fc076bd6dd27517eacdf25fef6c4dfe5f1d7448bafaaf3a26f13b5e4eb" -dependencies = [ - "encode_unicode 0.3.6", - "lazy_static", - "libc", - "unicode-width", - "windows-sys 0.52.0", -] - -[[package]] -name = "console-api" -version = "0.6.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fd326812b3fd01da5bb1af7d340d0d555fd3d4b641e7f1dfcf5962a902952787" -dependencies = [ - "futures-core", - "prost", - "prost-types", - "tonic 0.10.2", - "tracing-core", -] - -[[package]] -name = "console-subscriber" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7481d4c57092cd1c19dd541b92bdce883de840df30aa5d03fd48a3935c01842e" -dependencies = [ - "console-api", - "crossbeam-channel", - "crossbeam-utils", - "futures-task", - "hdrhistogram", - "humantime", - "prost-types", - "serde", - "serde_json", - "thread_local", - "tokio", - "tokio-stream", - "tonic 0.10.2", - "tracing", - "tracing-core", - "tracing-subscriber", -] - -[[package]] -name = "console_static_text" -version = "0.8.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f4be93df536dfbcbd39ff7c129635da089901116b88bfc29ec1acb9b56f8ff35" -dependencies = [ - "unicode-width", - "vte", -] - -[[package]] -name = "const-oid" -version = "0.9.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c2459377285ad874054d797f3ccebf984978aa39129f6eafde5cdc8315b612f8" - -[[package]] -name = "const-random" -version = "0.1.18" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "87e00182fe74b066627d63b85fd550ac2998d4b0bd86bfed477a0ae4c7c71359" -dependencies = [ - "const-random-macro", -] - -[[package]] -name = "const-random-macro" -version = "0.1.16" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f9d839f2a20b0aee515dc581a6172f2321f96cab76c1a38a4c584a194955390e" -dependencies = [ - "getrandom", - "once_cell", - "tiny-keccak", -] - -[[package]] -name = "constant_time_eq" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f7144d30dcf0fafbce74250a3963025d8d52177934239851c917d29f1df280c2" - -[[package]] -name = "convert_case" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6245d59a3e82a7fc217c5828a6692dbc6dfb63a0c8c90495621f7b9d79704a0e" - -[[package]] -name = "cooked-waker" -version = "5.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "147be55d677052dabc6b22252d5dd0fd4c29c8c27aa4f2fbef0f94aa003b406f" - -[[package]] -name = "cookie" -version = "0.16.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e859cd57d0710d9e06c381b550c06e76992472a8c6d527aecd2fc673dcc231fb" -dependencies = [ - "percent-encoding", - "time", - "version_check", -] - -[[package]] -name = "cookie_store" -version = "0.16.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d606d0fba62e13cf04db20536c05cb7f13673c161cb47a47a82b9b9e7d3f1daa" -dependencies = [ - "cookie", - "idna 0.2.3", - "log", - "publicsuffix", - "serde", - "serde_derive", - "serde_json", - "time", - "url", -] - -[[package]] -name = "core-foundation" -version = "0.9.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "91e195e091a93c46f7102ec7818a2aa394e1e1771c3ab4825963fa03e45afb8f" -dependencies = [ - "core-foundation-sys", - "libc", -] - -[[package]] -name = "core-foundation-sys" -version = "0.8.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "06ea2b9bc92be3c2baa9334a323ebca2d6f074ff852cd1d7b11064035cd3868f" - -[[package]] -name = "core2" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b49ba7ef1ad6107f8824dbe97de947cbaac53c44e7f9756a1fba0d37c1eec505" -dependencies = [ - "memchr", -] - -[[package]] -name = "cpufeatures" -version = "0.2.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53fe5e26ff1b7aef8bca9c6080520cfb8d9333c7568e1829cef191a9723e5504" -dependencies = [ - "libc", -] - -[[package]] -name = "crc32fast" -version = "1.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b3855a8a784b474f333699ef2bbca9db2c4a1f6d9088a90a2d25b1eb53111eaa" -dependencies = [ - "cfg-if", -] - -[[package]] -name = "criterion" -version = "0.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2b12d017a929603d80db1831cd3a24082f8137ce19c69e6447f54f5fc8d692f" -dependencies = [ - "anes", - "cast", - "ciborium", - "clap", - "criterion-plot", - "is-terminal", - "itertools 0.10.5", - "num-traits", - "once_cell", - "oorandom", - "plotters", - "rayon", - "regex", - "serde", - "serde_derive", - "serde_json", - "tinytemplate", - "walkdir", -] - -[[package]] -name = "criterion-plot" -version = "0.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6b50826342786a51a89e2da3a28f1c32b06e387201bc2d19791f622c673706b1" -dependencies = [ - "cast", - "itertools 0.10.5", -] - -[[package]] -name = "critical-section" -version = "1.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7059fff8937831a9ae6f0fe4d658ffabf58f2ca96aa9dec1c889f936f705f216" - -[[package]] -name = "crossbeam" -version = "0.8.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1137cd7e7fc0fb5d3c5a8678be38ec56e819125d8d7907411fe24ccb943faca8" -dependencies = [ - "crossbeam-channel", - "crossbeam-deque", - "crossbeam-epoch", - "crossbeam-queue", - "crossbeam-utils", -] - -[[package]] -name = "crossbeam-channel" -version = "0.5.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ab3db02a9c5b5121e1e42fbdb1aeb65f5e02624cc58c43f2884c6ccac0b82f95" -dependencies = [ - "crossbeam-utils", -] - -[[package]] -name = "crossbeam-deque" -version = "0.8.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "613f8cc01fe9cf1a3eb3d7f488fd2fa8388403e97039e2f73692932e291a770d" -dependencies = [ - "crossbeam-epoch", - "crossbeam-utils", -] - -[[package]] -name = "crossbeam-epoch" -version = "0.9.18" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5b82ac4a3c2ca9c3460964f020e1402edd5753411d7737aa39c3714ad1b5420e" -dependencies = [ - "crossbeam-utils", -] - -[[package]] -name = "crossbeam-queue" -version = "0.3.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "df0346b5d5e76ac2fe4e327c5fd1118d6be7c51dfb18f9b7922923f287471e35" -dependencies = [ - "crossbeam-utils", -] - -[[package]] -name = "crossbeam-utils" -version = "0.8.19" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "248e3bacc7dc6baa3b21e405ee045c3047101a49145e7e9eca583ab4c2ca5345" - -[[package]] -name = "crunchy" -version = "0.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7a81dae078cea95a014a339291cec439d2f232ebe854a9d672b796c6afafa9b7" - -[[package]] -name = "crypto-bigint" -version = "0.5.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0dc92fb57ca44df6db8059111ab3af99a63d5d0f8375d9972e319a379c6bab76" -dependencies = [ - "generic-array", - "rand_core", - "subtle", - "zeroize", -] - -[[package]] -name = "crypto-common" -version = "0.1.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1bfb12502f3fc46cca1bb51ac28df9d618d813cdc3d2f25b9fe775a34af26bb3" -dependencies = [ - "generic-array", - "rand_core", - "typenum", -] - -[[package]] -name = "csv" -version = "1.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ac574ff4d437a7b5ad237ef331c17ccca63c46479e5b5453eb8e10bb99a759fe" -dependencies = [ - "csv-core", - "itoa", - "ryu", - "serde", -] - -[[package]] -name = "csv-core" -version = "0.1.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5efa2b3d7902f4b634a20cae3c9c4e6209dc4779feb6863329607560143efa70" -dependencies = [ - "memchr", -] - -[[package]] -name = "ctr" -version = "0.9.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0369ee1ad671834580515889b80f2ea915f23b8be8d0daa4bbaf2ac5c7590835" -dependencies = [ - "cipher", -] - -[[package]] -name = "curve25519-dalek" -version = "4.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0a677b8922c94e01bdbb12126b0bc852f00447528dee1782229af9c720c3f348" -dependencies = [ - "cfg-if", - "cpufeatures", - "curve25519-dalek-derive", - "fiat-crypto", - "platforms", - "rustc_version 0.4.0", - "subtle", - "zeroize", -] - -[[package]] -name = "curve25519-dalek-derive" -version = "0.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f46882e17999c6cc590af592290432be3bce0428cb0d5f8b6715e4dc7b383eb3" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "daggy" -version = "0.8.0" -source = "git+https://github.com/getdozer/daggy?branch=feat/try_map#4ff8ebbdba979ecd7f638b6636c33ec9c0d27ccc" -dependencies = [ - "petgraph 0.6.3", - "serde", -] - -[[package]] -name = "darling" -version = "0.13.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a01d95850c592940db9b8194bc39f4bc0e89dee5c4265e4b1807c34a9aba453c" -dependencies = [ - "darling_core 0.13.4", - "darling_macro 0.13.4", -] - -[[package]] -name = "darling" -version = "0.20.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "54e36fcd13ed84ffdfda6f5be89b31287cbb80c439841fe69e04841435464391" -dependencies = [ - "darling_core 0.20.8", - "darling_macro 0.20.8", -] - -[[package]] -name = "darling_core" -version = "0.13.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "859d65a907b6852c9361e3185c862aae7fafd2887876799fa55f5f99dc40d610" -dependencies = [ - "fnv", - "ident_case", - "proc-macro2", - "quote", - "strsim 0.10.0", - "syn 1.0.109", -] - -[[package]] -name = "darling_core" -version = "0.20.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9c2cf1c23a687a1feeb728783b993c4e1ad83d99f351801977dd809b48d0a70f" -dependencies = [ - "fnv", - "ident_case", - "proc-macro2", - "quote", - "strsim 0.10.0", - "syn 2.0.53", -] - -[[package]] -name = "darling_macro" -version = "0.13.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9c972679f83bdf9c42bd905396b6c3588a843a17f0f16dfcfa3e2c5d57441835" -dependencies = [ - "darling_core 0.13.4", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "darling_macro" -version = "0.20.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a668eda54683121533a393014d8692171709ff57a7d61f187b6e782719f8933f" -dependencies = [ - "darling_core 0.20.8", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "dary_heap" -version = "0.3.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7762d17f1241643615821a8455a0b2c3e803784b058693d990b11f2dce25a0ca" - -[[package]] -name = "dashmap" -version = "4.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e77a43b28d0668df09411cb0bc9a8c2adc40f9a048afe863e05fd43251e8e39c" -dependencies = [ - "cfg-if", - "num_cpus", -] - -[[package]] -name = "dashmap" -version = "5.5.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "978747c1d849a7d2ee5e8adc0159961c48fb7e5db2f06af6723b80123bb53856" -dependencies = [ - "cfg-if", - "hashbrown 0.14.3", - "lock_api", - "once_cell", - "parking_lot_core", -] - -[[package]] -name = "data-encoding" -version = "2.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e962a19be5cfc3f3bf6dd8f61eb50107f356ad6270fbb3ed41476571db78be5" - -[[package]] -name = "data-url" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "41b319d1b62ffbd002e057f36bebd1f42b9f97927c9577461d855f3513c4289f" - -[[package]] -name = "datafusion" -version = "35.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4328f5467f76d890fe3f924362dbc3a838c6a733f762b32d87f9e0b7bef5fb49" -dependencies = [ - "ahash 0.8.11", - "arrow", - "arrow-array", - "arrow-ipc", - "arrow-schema", - "async-compression", - "async-trait", - "bytes", - "bzip2", - "chrono", - "dashmap 5.5.3", - "datafusion-common", - "datafusion-execution", - "datafusion-expr", - "datafusion-optimizer", - "datafusion-physical-expr", - "datafusion-physical-plan", - "datafusion-sql", - "flate2", - "futures", - "glob", - "half", - "hashbrown 0.14.3", - "indexmap 2.2.5", - "itertools 0.12.1", - "log", - "num_cpus", - "object_store", - "parking_lot", - "parquet", - "pin-project-lite", - "rand", - "sqlparser 0.41.0", - "tempfile", - "tokio", - "tokio-util", - "url", - "uuid", - "xz2", - "zstd", -] - -[[package]] -name = "datafusion-common" -version = "35.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d29a7752143b446db4a2cccd9a6517293c6b97e8c39e520ca43ccd07135a4f7e" -dependencies = [ - "ahash 0.8.11", - "arrow", - "arrow-array", - "arrow-buffer", - "arrow-schema", - "chrono", - "half", - "libc", - "num_cpus", - "object_store", - "parquet", - "sqlparser 0.41.0", -] - -[[package]] -name = "datafusion-execution" -version = "35.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2d447650af16e138c31237f53ddaef6dd4f92f0e2d3f2f35d190e16c214ca496" -dependencies = [ - "arrow", - "chrono", - "dashmap 5.5.3", - "datafusion-common", - "datafusion-expr", - "futures", - "hashbrown 0.14.3", - "log", - "object_store", - "parking_lot", - "rand", - "tempfile", - "url", -] - -[[package]] -name = "datafusion-expr" -version = "35.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d8d19598e48a498850fb79f97a9719b1f95e7deb64a7a06f93f313e8fa1d524b" -dependencies = [ - "ahash 0.8.11", - "arrow", - "arrow-array", - "datafusion-common", - "paste", - "sqlparser 0.41.0", - "strum", - "strum_macros", -] - -[[package]] -name = "datafusion-optimizer" -version = "35.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b7feb0391f1fc75575acb95b74bfd276903dc37a5409fcebe160bc7ddff2010" -dependencies = [ - "arrow", - "async-trait", - "chrono", - "datafusion-common", - "datafusion-expr", - "datafusion-physical-expr", - "hashbrown 0.14.3", - "itertools 0.12.1", - "log", - "regex-syntax 0.8.2", -] - -[[package]] -name = "datafusion-physical-expr" -version = "35.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e911bca609c89a54e8f014777449d8290327414d3e10c57a3e3c2122e38878d0" -dependencies = [ - "ahash 0.8.11", - "arrow", - "arrow-array", - "arrow-buffer", - "arrow-ord", - "arrow-schema", - "base64 0.21.7", - "blake2", - "blake3", - "chrono", - "datafusion-common", - "datafusion-expr", - "half", - "hashbrown 0.14.3", - "hex", - "indexmap 2.2.5", - "itertools 0.12.1", - "log", - "md-5", - "paste", - "petgraph 0.6.4", - "rand", - "regex", - "sha2", - "unicode-segmentation", - "uuid", -] - -[[package]] -name = "datafusion-physical-plan" -version = "35.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e96b546b8a02e9c2ab35ac6420d511f12a4701950c1eb2e568c122b4fefb0be3" -dependencies = [ - "ahash 0.8.11", - "arrow", - "arrow-array", - "arrow-buffer", - "arrow-schema", - "async-trait", - "chrono", - "datafusion-common", - "datafusion-execution", - "datafusion-expr", - "datafusion-physical-expr", - "futures", - "half", - "hashbrown 0.14.3", - "indexmap 2.2.5", - "itertools 0.12.1", - "log", - "once_cell", - "parking_lot", - "pin-project-lite", - "rand", - "tokio", - "uuid", -] - -[[package]] -name = "datafusion-proto" -version = "35.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5742f993d1812d6bb3cdc4ce2a0aa99e10f6dc0daa11dd69b0ff57f2d8e7518c" -dependencies = [ - "arrow", - "chrono", - "datafusion", - "datafusion-common", - "datafusion-expr", - "object_store", - "prost", -] - -[[package]] -name = "datafusion-sql" -version = "35.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2d18d36f260bbbd63aafdb55339213a23d540d3419810575850ef0a798a6b768" -dependencies = [ - "arrow", - "arrow-schema", - "datafusion-common", - "datafusion-expr", - "log", - "sqlparser 0.41.0", -] - -[[package]] -name = "debugid" -version = "0.8.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bef552e6f588e446098f6ba40d89ac146c8c7b64aade83c051ee00bb5d2bc18d" -dependencies = [ - "serde", - "uuid", -] - -[[package]] -name = "deltalake" -version = "0.17.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4e5fa38c2d00cc8a96789eaf3e7a2b27f0bd4c4c7a23e1e59bd5b7b4ceb796fe" -dependencies = [ - "deltalake-core", -] - -[[package]] -name = "deltalake-core" -version = "0.17.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "607be097d9bf5998bfbde3c5e6364f775e5adde0be55843130e5e50f2a2a4387" -dependencies = [ - "arrow", - "arrow-arith", - "arrow-array", - "arrow-buffer", - "arrow-cast", - "arrow-ipc", - "arrow-json", - "arrow-ord", - "arrow-row", - "arrow-schema", - "arrow-select", - "async-trait", - "bytes", - "cfg-if", - "chrono", - "dashmap 5.5.3", - "datafusion", - "datafusion-common", - "datafusion-expr", - "datafusion-physical-expr", - "datafusion-proto", - "datafusion-sql", - "either", - "errno", - "fix-hidden-lifetime-bug", - "futures", - "hashbrown 0.14.3", - "indexmap 2.2.5", - "itertools 0.12.1", - "lazy_static", - "libc", - "maplit", - "num-bigint", - "num-traits", - "num_cpus", - "object_store", - "once_cell", - "parking_lot", - "parquet", - "percent-encoding", - "pin-project-lite", - "rand", - "regex", - "roaring", - "serde", - "serde_json", - "sqlparser 0.41.0", - "thiserror", - "tokio", - "tracing", - "url", - "uuid", - "z85", -] - -[[package]] -name = "deno_ast" -version = "1.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1d87c67f73e749f78096f517cbb57967d98a8c713b39cf88b1f0b8750a84aa29" -dependencies = [ - "anyhow", - "base64 0.21.7", - "deno_media_type", - "dprint-swc-ext", - "serde", - "swc_atoms", - "swc_common", - "swc_config", - "swc_config_macro", - "swc_ecma_ast", - "swc_ecma_codegen", - "swc_ecma_codegen_macros", - "swc_ecma_loader", - "swc_ecma_parser", - "swc_ecma_transforms_base", - "swc_ecma_transforms_classes", - "swc_ecma_transforms_macros", - "swc_ecma_transforms_proposal", - "swc_ecma_transforms_react", - "swc_ecma_transforms_typescript", - "swc_ecma_utils", - "swc_ecma_visit", - "swc_eq_ignore_macros", - "swc_macros_common", - "swc_visit", - "swc_visit_macros", - "text_lines", - "url", -] - -[[package]] -name = "deno_broadcast_channel" -version = "0.136.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eabffd397f07309b53463257c9fc06c8942cc3cc00cdf9e8ee39cb98ad9b4f54" -dependencies = [ - "async-trait", - "deno_core", - "tokio", - "uuid", -] - -[[package]] -name = "deno_cache" -version = "0.74.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "942802e70c5dd03dd26e7175610a21dfc6a8966687c40caf07774206d23b0e46" -dependencies = [ - "async-trait", - "deno_core", - "rusqlite", - "serde", - "sha2", - "tokio", -] - -[[package]] -name = "deno_cache_dir" -version = "0.7.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6cf517bddfd22d79d0f284500318e3f9aea193536c2b61cbf6ce7b50a85f1b6a" -dependencies = [ - "anyhow", - "deno_media_type", - "indexmap 2.2.5", - "log", - "once_cell", - "parking_lot", - "serde", - "serde_json", - "sha2", - "thiserror", - "url", -] - -[[package]] -name = "deno_console" -version = "0.142.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1712de085595861d173f1cdaa05ef59521c63f78db9c9b7cd7b693ccde7e329b" -dependencies = [ - "deno_core", -] - -[[package]] -name = "deno_core" -version = "0.270.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2af854955a06a4bde79c68600a78d2269f5a783417f5adc1d2d1fd410b6cc434" -dependencies = [ - "anyhow", - "bincode 1.3.3", - "bit-set", - "bit-vec", - "bytes", - "cooked-waker", - "deno_core_icudata", - "deno_ops", - "deno_unsync", - "futures", - "libc", - "log", - "memoffset 0.9.0", - "parking_lot", - "pin-project", - "serde", - "serde_json", - "serde_v8", - "smallvec", - "sourcemap 7.1.1", - "static_assertions", - "tokio", - "url", - "v8", -] - -[[package]] -name = "deno_core_icudata" -version = "0.0.73" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a13951ea98c0a4c372f162d669193b4c9d991512de9f2381dd161027f34b26b1" - -[[package]] -name = "deno_crypto" -version = "0.156.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "505ce77750e138dd29e0bd51f944f3f52c27d1fbbdf2cfd388d848d05efe7000" -dependencies = [ - "aes", - "aes-gcm", - "aes-kw", - "base64 0.21.7", - "cbc", - "const-oid", - "ctr", - "curve25519-dalek", - "deno_core", - "deno_web", - "elliptic-curve", - "num-traits", - "once_cell", - "p256", - "p384", - "p521", - "rand", - "ring", - "rsa", - "serde", - "serde_bytes", - "sha1", - "sha2", - "signature", - "spki", - "tokio", - "uuid", - "x25519-dalek", -] - -[[package]] -name = "deno_fetch" -version = "0.166.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "aea276018532007aa3c45dc5bdff9c707fb8747e879b16d7a7782138128738e1" -dependencies = [ - "bytes", - "data-url", - "deno_core", - "deno_tls", - "dyn-clone", - "http 0.2.12", - "pin-project", - "reqwest", - "serde", - "serde_json", - "tokio", - "tokio-util", -] - -[[package]] -name = "deno_media_type" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a798670c20308e5770cc0775de821424ff9e85665b602928509c8c70430b3ee0" -dependencies = [ - "data-url", - "serde", - "url", -] - -[[package]] -name = "deno_napi" -version = "0.72.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "13553afb6cc3292b96c989abd3ee9e9cdb58ba703c2b69941c1b9392839372f9" -dependencies = [ - "deno_core", - "libloading 0.7.4", -] - -[[package]] -name = "deno_native_certs" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f4785d0bdc13819b665b71e4fb7e119d859568471e4c245ec5610857e70c9345" -dependencies = [ - "dlopen2", - "dlopen2_derive", - "once_cell", - "rustls-native-certs 0.6.3", - "rustls-pemfile 1.0.4", -] - -[[package]] -name = "deno_net" -version = "0.134.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b508e548550610f9555b438ec4acceb829d3d1c921e125c5eb4f8a9831c3fd3" -dependencies = [ - "deno_core", - "deno_tls", - "enum-as-inner 0.5.1", - "log", - "pin-project", - "rustls-tokio-stream", - "serde", - "socket2 0.5.6", - "tokio", - "trust-dns-proto 0.22.0", - "trust-dns-resolver 0.22.0", -] - -[[package]] -name = "deno_ops" -version = "0.146.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "13689abbb2af68c19b949a8852d9612f063fdc68a446a9c9d2b7b1e340f8516c" -dependencies = [ - "proc-macro-rules", - "proc-macro2", - "quote", - "strum", - "strum_macros", - "syn 2.0.53", - "thiserror", -] - -[[package]] -name = "deno_permissions" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "13947fbfb119f27f54f9d839d0484d610600b6b76b207ad544d134a36bd730ad" -dependencies = [ - "console_static_text", - "deno_core", - "deno_terminal", - "libc", - "log", - "once_cell", - "serde", - "termcolor", - "which 4.4.2", - "winapi", -] - -[[package]] -name = "deno_terminal" -version = "0.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e6337d4e7f375f8b986409a76fbeecfa4bd8a1343e63355729ae4befa058eaf" -dependencies = [ - "once_cell", - "termcolor", -] - -[[package]] -name = "deno_tls" -version = "0.129.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "515e99b099065defd3e1039b50b386dec85d0a6621371ce9fc9b67cc4d945912" -dependencies = [ - "deno_core", - "deno_native_certs", - "once_cell", - "rustls 0.21.10", - "rustls-pemfile 1.0.4", - "rustls-tokio-stream", - "rustls-webpki 0.101.7", - "serde", - "webpki-roots 0.25.4", -] - -[[package]] -name = "deno_unsync" -version = "0.3.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "30dff7e03584dbae188dae96a0f1876740054809b2ad0cf7c9fc5d361f20e739" -dependencies = [ - "tokio", -] - -[[package]] -name = "deno_url" -version = "0.142.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6bd5afe3567e4dba73bd1bf50153bbff6bcb2781d8c7e0468794ddfc4e8e96a6" -dependencies = [ - "deno_core", - "serde", - "urlpattern", -] - -[[package]] -name = "deno_web" -version = "0.173.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ff579a91d7c82a1bb9e71e3dba48c5b5fdb92953ac927a6f37d46bce49d3bdcd" -dependencies = [ - "async-trait", - "base64-simd 0.8.0", - "bytes", - "deno_core", - "encoding_rs", - "flate2", - "futures", - "serde", - "tokio", - "uuid", - "windows-sys 0.48.0", -] - -[[package]] -name = "deno_webidl" -version = "0.142.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c4e1377b1548eb05421de51d533878425c200de952dcb45798af963a67fc76be" -dependencies = [ - "deno_core", -] - -[[package]] -name = "deno_websocket" -version = "0.147.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "de022c3ed606c0a0df47bfd6d62b4f1fd0c4d57edfeba6c1202f7d6dec037658" -dependencies = [ - "bytes", - "deno_core", - "deno_net", - "deno_tls", - "fastwebsockets", - "h2 0.4.4", - "http 1.1.0", - "http-body-util", - "hyper 1.1.0", - "hyper-util", - "once_cell", - "rustls-tokio-stream", - "serde", - "tokio", -] - -[[package]] -name = "deno_webstorage" -version = "0.137.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cd538d9a6b8f45a105e9f8c2cd3b345d8a437c563f42d85aea368a322d05ab51" -dependencies = [ - "deno_core", - "deno_web", - "rusqlite", - "serde", -] - -[[package]] -name = "der" -version = "0.7.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fffa369a668c8af7dbf8b5e56c9f744fbd399949ed171606040001947de40b1c" -dependencies = [ - "const-oid", - "pem-rfc7468", - "zeroize", -] - -[[package]] -name = "deranged" -version = "0.3.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b42b6fa04a440b495c8b04d0e71b707c585f83cb9cb28cf8cd0d976c315e31b4" -dependencies = [ - "powerfmt", -] - -[[package]] -name = "derivative" -version = "2.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fcc3dd5e9e9c0b295d6e1e4d811fb6f157d5ffd784b8d202fc62eac8035a770b" -dependencies = [ - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "derive_arbitrary" -version = "1.3.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "67e77553c4162a157adbf834ebae5b415acbecbeafc7a74b0e886657506a7611" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "derive_more" -version = "0.99.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4fb810d30a7c1953f91334de7244731fc3f3c10d7fe163338a35b9f640960321" -dependencies = [ - "convert_case", - "proc-macro2", - "quote", - "rustc_version 0.4.0", - "syn 1.0.109", -] - -[[package]] -name = "digest" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3dd60d1080a57a05ab032377049e0591415d2b31afd7028356dbf3cc6dcb066" -dependencies = [ - "generic-array", -] - -[[package]] -name = "digest" -version = "0.10.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" -dependencies = [ - "block-buffer 0.10.4", - "const-oid", - "crypto-common", - "subtle", -] - -[[package]] -name = "dirs-next" -version = "2.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b98cf8ebf19c3d1b223e151f99a4f9f0690dca41414773390fc824184ac833e1" -dependencies = [ - "cfg-if", - "dirs-sys-next", -] - -[[package]] -name = "dirs-sys-next" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4ebda144c4fe02d1f7ea1a7d9641b6fc6b580adcfa024ae48797ecdeb6825b4d" -dependencies = [ - "libc", - "redox_users", - "winapi", -] - -[[package]] -name = "dlopen2" -version = "0.6.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6bc2c7ed06fd72a8513ded8d0d2f6fd2655a85d6885c48cae8625d80faf28c03" -dependencies = [ - "dlopen2_derive", - "libc", - "once_cell", - "winapi", -] - -[[package]] -name = "dlopen2_derive" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2b99bf03862d7f545ebc28ddd33a665b50865f4dfd84031a393823879bd4c54" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "doc-comment" -version = "0.3.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fea41bba32d969b513997752735605054bc0dfa92b4c56bf1189f2e174be7a10" - -[[package]] -name = "dozer-cli" -version = "0.4.0" -dependencies = [ - "actix-files", - "actix-web", - "async-trait", - "camino", - "clap", - "dozer-core", - "dozer-ingestion", - "dozer-sink-clickhouse", - "dozer-sql", - "dozer-tracing", - "dozer-types", - "futures", - "glob", - "handlebars", - "include_dir", - "notify", - "notify-debouncer-full", - "page_size", - "prometheus-parse", - "prost-reflect", - "reqwest", - "rustyline", - "rustyline-derive", - "tempfile", - "tokio", - "tokio-stream", - "tonic 0.11.0", - "tonic-reflection", - "tonic-web", - "tower", - "tower-http", - "uuid", - "webbrowser", - "zip", -] - -[[package]] -name = "dozer-core" -version = "0.4.0" -dependencies = [ - "async-stream", - "bincode 2.0.0-rc.3", - "crossbeam", - "daggy", - "deno_core", - "dozer-tracing", - "dozer-types", - "futures", - "futures-util", - "tokio", - "uuid", -] - -[[package]] -name = "dozer-deno" -version = "0.4.0" -dependencies = [ - "deno_ast", - "deno_broadcast_channel", - "deno_cache", - "deno_cache_dir", - "deno_console", - "deno_core", - "deno_crypto", - "deno_fetch", - "deno_napi", - "deno_permissions", - "deno_terminal", - "deno_tls", - "deno_url", - "deno_web", - "deno_webidl", - "deno_websocket", - "deno_webstorage", - "dozer-types", - "encoding_rs", - "once_cell", - "tempfile", - "tokio", -] - -[[package]] -name = "dozer-ingestion" -version = "0.4.0" -dependencies = [ - "bytes", - "chrono", - "criterion", - "dozer-ingestion-connector", - "dozer-ingestion-deltalake", - "dozer-ingestion-ethereum", - "dozer-ingestion-grpc", - "dozer-ingestion-javascript", - "dozer-ingestion-kafka", - "dozer-ingestion-mongodb", - "dozer-ingestion-mysql", - "dozer-ingestion-object-store", - "dozer-ingestion-postgres", - "dozer-ingestion-snowflake", - "dozer-ingestion-webhook", - "dozer-tracing", - "dozer-utils", - "env_logger 0.10.2", - "futures", - "hex", - "parquet", - "prost-reflect", - "rand", - "serial_test 2.0.0", - "tempfile", - "tokio", - "url", -] - -[[package]] -name = "dozer-ingestion-connector" -version = "0.4.0" -dependencies = [ - "dozer-types", - "futures", - "tokio", -] - -[[package]] -name = "dozer-ingestion-deltalake" -version = "0.4.0" -dependencies = [ - "deltalake", - "dozer-ingestion-connector", - "dozer-ingestion-object-store", -] - -[[package]] -name = "dozer-ingestion-ethereum" -version = "0.4.0" -dependencies = [ - "dozer-ingestion-connector", - "dozer-tracing", - "hex-literal", - "web3", -] - -[[package]] -name = "dozer-ingestion-grpc" -version = "0.4.0" -dependencies = [ - "dozer-ingestion-connector", - "tonic-reflection", - "tonic-web", - "tower-http", -] - -[[package]] -name = "dozer-ingestion-javascript" -version = "0.4.0" -dependencies = [ - "camino", - "deno_core", - "dozer-deno", - "dozer-ingestion-connector", -] - -[[package]] -name = "dozer-ingestion-kafka" -version = "0.4.0" -dependencies = [ - "base64 0.21.7", - "dozer-ingestion-connector", - "rdkafka", - "schema_registry_converter", -] - -[[package]] -name = "dozer-ingestion-mongodb" -version = "0.4.0" -dependencies = [ - "bson", - "dozer-ingestion-connector", - "mongodb", -] - -[[package]] -name = "dozer-ingestion-mysql" -version = "0.4.0" -dependencies = [ - "dozer-ingestion-connector", - "geozero", - "hex", - "mysql_async", - "mysql_common", - "rand", - "serial_test 1.0.0", - "sqlparser 0.41.0", -] - -[[package]] -name = "dozer-ingestion-object-store" -version = "0.4.0" -dependencies = [ - "datafusion", - "dozer-ingestion-connector", - "object_store", - "url", -] - -[[package]] -name = "dozer-ingestion-postgres" -version = "0.4.0" -dependencies = [ - "dozer-ingestion-connector", - "postgres-protocol", - "postgres-types", - "rand", - "regex", - "rustls 0.22.2", - "rustls-native-certs 0.7.0", - "serial_test 1.0.0", - "tokio", - "tokio-postgres", - "tokio-postgres-rustls", - "uuid", -] - -[[package]] -name = "dozer-ingestion-snowflake" -version = "0.4.0" -dependencies = [ - "base64 0.21.7", - "dozer-ingestion-connector", - "genawaiter", - "include_dir", - "memchr", - "odbc", - "rand", -] - -[[package]] -name = "dozer-ingestion-webhook" -version = "0.1.0" -dependencies = [ - "actix-web", - "dozer-ingestion-connector", - "env_logger 0.11.2", - "reqwest", - "tokio", -] - -[[package]] -name = "dozer-sink-clickhouse" -version = "0.1.0" -dependencies = [ - "chrono-tz", - "clickhouse-rs", - "dozer-core", - "dozer-types", - "either", - "serde", -] - -[[package]] -name = "dozer-sql" -version = "0.4.0" -dependencies = [ - "ahash 0.8.11", - "bincode 2.0.0-rc.3", - "dozer-core", - "dozer-sql-expression", - "dozer-tracing", - "dozer-types", - "enum_dispatch", - "linked-hash-map", - "multimap 0.9.1", - "proptest", - "regex", - "tokio", -] - -[[package]] -name = "dozer-sql-expression" -version = "0.4.0" -dependencies = [ - "async-recursion", - "bigdecimal 0.3.1", - "bincode 2.0.0-rc.3", - "deno_core", - "dozer-core", - "dozer-deno", - "dozer-types", - "half", - "jsonpath", - "like", - "ndarray", - "num-traits", - "ort", - "proptest", - "sqlparser 0.35.0", - "tokio", -] - -[[package]] -name = "dozer-tests" -version = "0.4.0" -dependencies = [ - "ahash 0.8.11", - "async-trait", - "clap", - "crossbeam", - "csv", - "dozer-cli", - "dozer-core", - "dozer-sql", - "dozer-tracing", - "dozer-types", - "dozer-utils", - "env_logger 0.10.2", - "futures", - "libtest-mimic", - "rusqlite", - "sqllogictest", - "sqlparser 0.35.0", - "tokio", - "tonic 0.10.2", - "url", - "walkdir", -] - -[[package]] -name = "dozer-tracing" -version = "0.4.0" -dependencies = [ - "console-subscriber", - "crossbeam", - "dozer-types", - "futures-util", - "hyper 0.14.28", - "once_cell", - "opentelemetry", - "opentelemetry-aws", - "opentelemetry-otlp", - "opentelemetry-prometheus", - "opentelemetry_sdk", - "parking_lot", - "prometheus", - "tokio", - "tracing-opentelemetry", - "tracing-subscriber", -] - -[[package]] -name = "dozer-types" -version = "0.4.0" -dependencies = [ - "ahash 0.8.11", - "arbitrary", - "arrow", - "arrow-cast", - "arrow-schema", - "bincode 2.0.0-rc.3", - "bytes", - "chrono", - "geo", - "ijson", - "indexmap 2.2.5", - "indicatif", - "log", - "ordered-float 3.9.2", - "parking_lot", - "prettytable-rs", - "prost", - "prost-types", - "pyo3", - "regex", - "rmp-serde", - "rust_decimal", - "schemars", - "serde", - "serde_bytes", - "serde_json", - "serde_yaml", - "thiserror", - "tokio", - "tokio-postgres", - "tonic 0.11.0", - "tonic-build", - "tracing", -] - -[[package]] -name = "dozer-utils" -version = "0.4.0" -dependencies = [ - "dozer-types", -] - -[[package]] -name = "dprint-swc-ext" -version = "0.13.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7b2f24ce6b89a06ae3eb08d5d4f88c05d0aef1fa58e2eba8dd92c97b84210c25" -dependencies = [ - "bumpalo", - "num-bigint", - "rustc-hash", - "swc_atoms", - "swc_common", - "swc_ecma_ast", - "swc_ecma_parser", - "text_lines", -] - -[[package]] -name = "dyn-clone" -version = "1.0.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0d6ef0072f8a535281e4876be788938b528e9a1d43900b82c2569af7da799125" - -[[package]] -name = "earcutr" -version = "0.4.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "79127ed59a85d7687c409e9978547cffb7dc79675355ed22da6b66fd5f6ead01" -dependencies = [ - "itertools 0.11.0", - "num-traits", -] - -[[package]] -name = "ecdsa" -version = "0.16.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ee27f32b5c5292967d2d4a9d7f1e0b0aed2c15daded5a60300e4abb9d8020bca" -dependencies = [ - "der", - "digest 0.10.7", - "elliptic-curve", - "rfc6979", - "signature", - "spki", -] - -[[package]] -name = "educe" -version = "0.4.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0f0042ff8246a363dbe77d2ceedb073339e85a804b9a47636c6e016a9a32c05f" -dependencies = [ - "enum-ordinalize", - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "either" -version = "1.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "11157ac094ffbdde99aa67b23417ebdd801842852b500e395a45a9c0aac03e4a" - -[[package]] -name = "elliptic-curve" -version = "0.13.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b5e6043086bf7973472e0c7dff2142ea0b680d30e18d9cc40f267efbf222bd47" -dependencies = [ - "base16ct", - "crypto-bigint", - "digest 0.10.7", - "ff", - "generic-array", - "group", - "hkdf", - "pem-rfc7468", - "pkcs8", - "rand_core", - "sec1", - "subtle", - "zeroize", -] - -[[package]] -name = "encode_unicode" -version = "0.3.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a357d28ed41a50f9c765dbfe56cbc04a64e53e5fc58ba79fbc34c10ef3df831f" - -[[package]] -name = "encode_unicode" -version = "1.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "34aa73646ffb006b8f5147f3dc182bd4bcb190227ce861fc4a4844bf8e3cb2c0" - -[[package]] -name = "encoding_rs" -version = "0.8.33" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7268b386296a025e474d5140678f75d6de9493ae55a5d709eeb9dd08149945e1" -dependencies = [ - "cfg-if", -] - -[[package]] -name = "endian-type" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c34f04666d835ff5d62e058c3995147c06f42fe86ff053337632bca83e42702d" - -[[package]] -name = "enum-as-inner" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "21cdad81446a7f7dc43f6a77409efeb9733d2fa65553efef6018ef257c959b73" -dependencies = [ - "heck 0.4.1", - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "enum-as-inner" -version = "0.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c9720bba047d567ffc8a3cba48bf19126600e249ab7f128e9233e6376976a116" -dependencies = [ - "heck 0.4.1", - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "enum-ordinalize" -version = "3.1.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1bf1fa3f06bbff1ea5b1a9c7b14aa992a39657db60a2759457328d7e058f49ee" -dependencies = [ - "num-bigint", - "num-traits", - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "enum_dispatch" -version = "0.3.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8f33313078bb8d4d05a2733a94ac4c2d8a0df9a2b84424ebf4f33bfc224a890e" -dependencies = [ - "once_cell", - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "env_filter" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a009aa4810eb158359dda09d0c87378e4bbb89b5a801f016885a4707ba24f7ea" -dependencies = [ - "log", - "regex", -] - -[[package]] -name = "env_logger" -version = "0.10.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4cd405aab171cb85d6735e5c8d9db038c17d3ca007a4d2c25f337935c3d90580" -dependencies = [ - "humantime", - "is-terminal", - "log", - "regex", - "termcolor", -] - -[[package]] -name = "env_logger" -version = "0.11.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6c012a26a7f605efc424dd53697843a72be7dc86ad2d01f7814337794a12231d" -dependencies = [ - "anstream", - "anstyle", - "env_filter", - "humantime", - "log", -] - -[[package]] -name = "equivalent" -version = "1.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5443807d6dff69373d433ab9ef5378ad8df50ca6298caf15de6e52e24aaf54d5" - -[[package]] -name = "errno" -version = "0.3.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a258e46cdc063eb8519c00b9fc845fc47bcfca4130e2f08e88665ceda8474245" -dependencies = [ - "libc", - "windows-sys 0.52.0", -] - -[[package]] -name = "error-code" -version = "3.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a0474425d51df81997e2f90a21591180b38eccf27292d755f3e30750225c175b" - -[[package]] -name = "ethabi" -version = "18.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7413c5f74cc903ea37386a8965a936cbeb334bd270862fdece542c1b2dcbc898" -dependencies = [ - "ethereum-types", - "hex", - "once_cell", - "regex", - "serde", - "serde_json", - "sha3", - "thiserror", - "uint", -] - -[[package]] -name = "ethbloom" -version = "0.13.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c22d4b5885b6aa2fe5e8b9329fb8d232bf739e434e6b87347c63bdd00c120f60" -dependencies = [ - "crunchy", - "fixed-hash", - "impl-rlp", - "impl-serde", - "tiny-keccak", -] - -[[package]] -name = "ethereum-types" -version = "0.14.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "02d215cbf040552efcbe99a38372fe80ab9d00268e20012b79fcd0f073edd8ee" -dependencies = [ - "ethbloom", - "fixed-hash", - "impl-rlp", - "impl-serde", - "primitive-types", - "uint", -] - -[[package]] -name = "fallible-iterator" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4443176a9f2c162692bd3d352d745ef9413eec5782a80d8fd6f8a1ac692a07f7" - -[[package]] -name = "fallible-streaming-iterator" -version = "0.1.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7360491ce676a36bf9bb3c56c1aa791658183a54d2744120f27285738d90465a" - -[[package]] -name = "fastrand" -version = "2.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "25cbce373ec4653f1a01a31e8a5e5ec0c622dc27ff9c4e6606eefef5cbbed4a5" - -[[package]] -name = "fastwebsockets" -version = "0.6.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f63dd7b57f9b33b1741fa631c9522eb35d43e96dcca4a6a91d5e4ca7c93acdc1" -dependencies = [ - "base64 0.21.7", - "http-body-util", - "hyper 1.1.0", - "hyper-util", - "pin-project", - "rand", - "sha1", - "simdutf8", - "thiserror", - "tokio", - "utf-8", -] - -[[package]] -name = "fd-lock" -version = "4.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e5768da2206272c81ef0b5e951a41862938a6070da63bcea197899942d3b947" -dependencies = [ - "cfg-if", - "rustix", - "windows-sys 0.52.0", -] - -[[package]] -name = "ff" -version = "0.13.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ded41244b729663b1e574f1b4fb731469f69f79c17667b5d776b16cda0479449" -dependencies = [ - "rand_core", - "subtle", -] - -[[package]] -name = "fiat-crypto" -version = "0.2.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c007b1ae3abe1cb6f85a16305acd418b7ca6343b953633fee2b76d8f108b830f" - -[[package]] -name = "file-id" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e13be71e6ca82e91bc0cb862bebaac0b2d1924a5a1d970c822b2f98b63fda8c3" -dependencies = [ - "winapi-util", -] - -[[package]] -name = "filetime" -version = "0.2.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ee447700ac8aa0b2f2bd7bc4462ad686ba06baa6727ac149a2d6277f0d240fd" -dependencies = [ - "cfg-if", - "libc", - "redox_syscall", - "windows-sys 0.52.0", -] - -[[package]] -name = "finl_unicode" -version = "1.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fcfdc7a0362c9f4444381a9e697c79d435fe65b52a37466fc2c1184cee9edc6" - -[[package]] -name = "fix-hidden-lifetime-bug" -version = "0.2.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d4ae9c2016a663983d4e40a9ff967d6dcac59819672f0b47f2b17574e99c33c8" -dependencies = [ - "fix-hidden-lifetime-bug-proc_macros", -] - -[[package]] -name = "fix-hidden-lifetime-bug-proc_macros" -version = "0.2.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e4c81935e123ab0741c4c4f0d9b8377e5fb21d3de7e062fa4b1263b1fbcba1ea" -dependencies = [ - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "fixed-hash" -version = "0.8.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "835c052cb0c08c1acf6ffd71c022172e18723949c8282f2b9f27efbc51e64534" -dependencies = [ - "byteorder", - "rand", - "rustc-hex", - "static_assertions", -] - -[[package]] -name = "fixedbitset" -version = "0.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ce7134b9999ecaf8bcd65542e436736ef32ddca1b3e06094cb6ec5755203b80" - -[[package]] -name = "flatbuffers" -version = "23.5.26" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4dac53e22462d78c16d64a1cd22371b54cc3fe94aa15e7886a2fa6e5d1ab8640" -dependencies = [ - "bitflags 1.3.2", - "rustc_version 0.4.0", -] - -[[package]] -name = "flate2" -version = "1.0.28" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "46303f565772937ffe1d394a4fac6f411c6013172fadde9dcdb1e147a086940e" -dependencies = [ - "crc32fast", - "miniz_oxide", -] - -[[package]] -name = "float_next_after" -version = "1.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8bf7cc16383c4b8d58b9905a8509f02926ce3058053c056376248d958c9df1e8" - -[[package]] -name = "fnv" -version = "1.0.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3f9eec918d3f24069decb9af1554cad7c880e2da24a9afd88aca000531ab82c1" - -[[package]] -name = "foreign-types" -version = "0.3.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f6f339eb8adc052cd2ca78910fda869aefa38d22d5cb648e6485e4d3fc06f3b1" -dependencies = [ - "foreign-types-shared", -] - -[[package]] -name = "foreign-types-shared" -version = "0.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "00b0228411908ca8685dba7fc2cdd70ec9990a6e753e89b6ac91a84c40fbaf4b" - -[[package]] -name = "form_urlencoded" -version = "1.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e13624c2627564efccf4934284bdd98cbaa14e79b0b5a141218e507b3a823456" -dependencies = [ - "percent-encoding", -] - -[[package]] -name = "from_variant" -version = "0.1.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3a0b11eeb173ce52f84ebd943d42e58813a2ebb78a6a3ff0a243b71c5199cd7b" -dependencies = [ - "proc-macro2", - "swc_macros_common", - "syn 2.0.53", -] - -[[package]] -name = "frunk" -version = "0.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "11a351b59e12f97b4176ee78497dff72e4276fb1ceb13e19056aca7fa0206287" -dependencies = [ - "frunk_core", - "frunk_derives", - "frunk_proc_macros", -] - -[[package]] -name = "frunk_core" -version = "0.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "af2469fab0bd07e64ccf0ad57a1438f63160c69b2e57f04a439653d68eb558d6" - -[[package]] -name = "frunk_derives" -version = "0.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b0fa992f1656e1707946bbba340ad244f0814009ef8c0118eb7b658395f19a2e" -dependencies = [ - "frunk_proc_macro_helpers", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "frunk_proc_macro_helpers" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "35b54add839292b743aeda6ebedbd8b11e93404f902c56223e51b9ec18a13d2c" -dependencies = [ - "frunk_core", - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "frunk_proc_macros" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "71b85a1d4a9a6b300b41c05e8e13ef2feca03e0334127f29eca9506a7fe13a93" -dependencies = [ - "frunk_core", - "frunk_proc_macro_helpers", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "fs-err" -version = "2.11.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "88a41f105fe1d5b6b34b2055e3dc59bb79b46b48b2040b9e6c7b4b5de097aa41" -dependencies = [ - "autocfg", -] - -[[package]] -name = "fsevent-sys" -version = "4.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "76ee7a02da4d231650c7cea31349b889be2f45ddb3ef3032d2ec8185f6313fd2" -dependencies = [ - "libc", -] - -[[package]] -name = "fslock" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "04412b8935272e3a9bae6f48c7bfff74c2911f60525404edfdd28e49884c3bfb" -dependencies = [ - "libc", - "winapi", -] - -[[package]] -name = "funty" -version = "2.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e6d5a32815ae3f33302d95fdcb2ce17862f8c65363dcfd29360480ba1001fc9c" - -[[package]] -name = "futures" -version = "0.3.30" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "645c6916888f6cb6350d2550b80fb63e734897a8498abe35cfb732b6487804b0" -dependencies = [ - "futures-channel", - "futures-core", - "futures-executor", - "futures-io", - "futures-sink", - "futures-task", - "futures-util", -] - -[[package]] -name = "futures-channel" -version = "0.3.30" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eac8f7d7865dcb88bd4373ab671c8cf4508703796caa2b1985a9ca867b3fcb78" -dependencies = [ - "futures-core", - "futures-sink", -] - -[[package]] -name = "futures-core" -version = "0.3.30" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dfc6580bb841c5a68e9ef15c77ccc837b40a7504914d52e47b8b0e9bbda25a1d" - -[[package]] -name = "futures-executor" -version = "0.3.30" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a576fc72ae164fca6b9db127eaa9a9dda0d61316034f33a0a0d4eda41f02b01d" -dependencies = [ - "futures-core", - "futures-task", - "futures-util", -] - -[[package]] -name = "futures-io" -version = "0.3.30" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a44623e20b9681a318efdd71c299b6b222ed6f231972bfe2f224ebad6311f0c1" - -[[package]] -name = "futures-macro" -version = "0.3.30" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "87750cf4b7a4c0625b1529e4c543c2182106e4dedc60a2a6455e00d212c489ac" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "futures-sink" -version = "0.3.30" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9fb8e00e87438d937621c1c6269e53f536c14d3fbd6a042bb24879e57d474fb5" - -[[package]] -name = "futures-task" -version = "0.3.30" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "38d84fa142264698cdce1a9f9172cf383a0c82de1bddcf3092901442c4097004" - -[[package]] -name = "futures-timer" -version = "3.0.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f288b0a4f20f9a56b5d1da57e2227c661b7b16168e2f72365f57b63326e29b24" - -[[package]] -name = "futures-util" -version = "0.3.30" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3d6401deb83407ab3da39eba7e33987a73c3df0c82b4bb5813ee871c19c41d48" -dependencies = [ - "futures-channel", - "futures-core", - "futures-io", - "futures-macro", - "futures-sink", - "futures-task", - "memchr", - "pin-project-lite", - "pin-utils", - "slab", -] - -[[package]] -name = "genawaiter" -version = "0.99.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c86bd0361bcbde39b13475e6e36cb24c329964aa2611be285289d1e4b751c1a0" -dependencies = [ - "genawaiter-macro", - "genawaiter-proc-macro", - "proc-macro-hack", -] - -[[package]] -name = "genawaiter-macro" -version = "0.99.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b32dfe1fdfc0bbde1f22a5da25355514b5e450c33a6af6770884c8750aedfbc" - -[[package]] -name = "genawaiter-proc-macro" -version = "0.99.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "784f84eebc366e15251c4a8c3acee82a6a6f427949776ecb88377362a9621738" -dependencies = [ - "proc-macro-error 0.4.12", - "proc-macro-hack", - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "generic-array" -version = "0.14.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" -dependencies = [ - "typenum", - "version_check", - "zeroize", -] - -[[package]] -name = "geo" -version = "0.26.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1645cf1d7fea7dac1a66f7357f3df2677ada708b8d9db8e9b043878930095a96" -dependencies = [ - "earcutr", - "float_next_after", - "geo-types", - "geographiclib-rs", - "log", - "num-traits", - "robust", - "rstar", - "serde", -] - -[[package]] -name = "geo-types" -version = "0.7.13" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ff16065e5720f376fbced200a5ae0f47ace85fd70b7e54269790281353b6d61" -dependencies = [ - "approx", - "num-traits", - "rstar", - "serde", -] - -[[package]] -name = "geographiclib-rs" -version = "0.2.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e6e5ed84f8089c70234b0a8e0aedb6dc733671612ddc0d37c6066052f9781960" -dependencies = [ - "libm", -] - -[[package]] -name = "geozero" -version = "0.11.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4b1b9a1eeae9ad09e12ec50243956105184b26440f81f978cd3aae009b214d4d" -dependencies = [ - "log", - "scroll", - "serde_json", - "thiserror", - "wkt", -] - -[[package]] -name = "getrandom" -version = "0.2.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "190092ea657667030ac6a35e305e62fc4dd69fd98ac98631e5d3a2b1575a12b5" -dependencies = [ - "cfg-if", - "libc", - "wasi", -] - -[[package]] -name = "ghash" -version = "0.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f0d8a4362ccb29cb0b265253fb0a2728f592895ee6854fd9bc13f2ffda266ff1" -dependencies = [ - "opaque-debug", - "polyval", -] - -[[package]] -name = "gimli" -version = "0.28.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4271d37baee1b8c7e4b708028c57d816cf9d2434acb33a549475f78c181f6253" - -[[package]] -name = "glob" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d2fabcfbdc87f4758337ca535fb41a6d701b65693ce38287d856d1674551ec9b" - -[[package]] -name = "group" -version = "0.13.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f0f9ef7462f7c099f518d754361858f86d8a07af53ba9af0fe635bbccb151a63" -dependencies = [ - "ff", - "rand_core", - "subtle", -] - -[[package]] -name = "h2" -version = "0.3.26" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "81fe527a889e1532da5c525686d96d4c2e74cdd345badf8dfef9f6b39dd5f5e8" -dependencies = [ - "bytes", - "fnv", - "futures-core", - "futures-sink", - "futures-util", - "http 0.2.12", - "indexmap 2.2.5", - "slab", - "tokio", - "tokio-util", - "tracing", -] - -[[package]] -name = "h2" -version = "0.4.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "816ec7294445779408f36fe57bc5b7fc1cf59664059096c65f905c1c61f58069" -dependencies = [ - "bytes", - "fnv", - "futures-core", - "futures-sink", - "futures-util", - "http 1.1.0", - "indexmap 2.2.5", - "slab", - "tokio", - "tokio-util", - "tracing", -] - -[[package]] -name = "half" -version = "2.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b5eceaaeec696539ddaf7b333340f1af35a5aa87ae3e4f3ead0532f72affab2e" -dependencies = [ - "cfg-if", - "crunchy", - "num-traits", -] - -[[package]] -name = "handlebars" -version = "4.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "faa67bab9ff362228eb3d00bd024a4965d8231bbb7921167f0cfa66c6626b225" -dependencies = [ - "log", - "pest", - "pest_derive", - "serde", - "serde_json", - "thiserror", -] - -[[package]] -name = "hash32" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b0c35f58762feb77d74ebe43bdbc3210f09be9fe6742234d573bacc26ed92b67" -dependencies = [ - "byteorder", -] - -[[package]] -name = "hashbrown" -version = "0.12.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8a9ee70c43aaf417c914396645a0fa852624801b24ebb7ae78fe8272889ac888" -dependencies = [ - "ahash 0.7.8", -] - -[[package]] -name = "hashbrown" -version = "0.13.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "43a3c133739dddd0d2990f9a4bdf8eb4b21ef50e4851ca85ab661199821d510e" -dependencies = [ - "ahash 0.8.11", -] - -[[package]] -name = "hashbrown" -version = "0.14.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "290f1a1d9242c78d09ce40a5e87e7554ee637af1351968159f4952f028f75604" -dependencies = [ - "ahash 0.8.11", - "allocator-api2", -] - -[[package]] -name = "hashlink" -version = "0.8.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e8094feaf31ff591f651a2664fb9cfd92bba7a60ce3197265e9482ebe753c8f7" -dependencies = [ - "hashbrown 0.14.3", -] - -[[package]] -name = "hdrhistogram" -version = "7.5.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "765c9198f173dd59ce26ff9f95ef0aafd0a0fe01fb9d72841bc5066a4c06511d" -dependencies = [ - "base64 0.21.7", - "byteorder", - "flate2", - "nom", - "num-traits", -] - -[[package]] -name = "headers" -version = "0.3.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "06683b93020a07e3dbcf5f8c0f6d40080d725bea7936fc01ad345c01b97dc270" -dependencies = [ - "base64 0.21.7", - "bytes", - "headers-core", - "http 0.2.12", - "httpdate", - "mime", - "sha1", -] - -[[package]] -name = "headers-core" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e7f66481bfee273957b1f20485a4ff3362987f85b2c236580d81b4eb7a326429" -dependencies = [ - "http 0.2.12", -] - -[[package]] -name = "heapless" -version = "0.7.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cdc6457c0eb62c71aac4bc17216026d8410337c4126773b9c5daba343f17964f" -dependencies = [ - "atomic-polyfill", - "hash32", - "rustc_version 0.4.0", - "spin 0.9.8", - "stable_deref_trait", -] - -[[package]] -name = "heck" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "95505c38b4572b2d910cecb0281560f54b440a19336cbbcb27bf6ce6adc6f5a8" - -[[package]] -name = "heck" -version = "0.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" - -[[package]] -name = "hermit-abi" -version = "0.3.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d231dfb89cfffdbc30e7fc41579ed6066ad03abda9e567ccafae602b97ec5024" - -[[package]] -name = "hex" -version = "0.4.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7f24254aa9a54b5c858eaee2f5bccdb46aaf0e486a595ed5fd8f86ba55232a70" - -[[package]] -name = "hex-literal" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6fe2267d4ed49bc07b63801559be28c718ea06c4738b7a03c94df7386d2cde46" - -[[package]] -name = "hkdf" -version = "0.12.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7b5f8eb2ad728638ea2c7d47a21db23b7b58a72ed6a38256b8a1849f15fbbdf7" -dependencies = [ - "hmac", -] - -[[package]] -name = "hmac" -version = "0.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6c49c37c09c17a53d937dfbb742eb3a961d65a994e6bcdcf37e7399d0cc8ab5e" -dependencies = [ - "digest 0.10.7", -] - -[[package]] -name = "home" -version = "0.5.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e3d1354bf6b7235cb4a0576c2619fd4ed18183f689b12b006a0ee7329eeff9a5" -dependencies = [ - "windows-sys 0.52.0", -] - -[[package]] -name = "hostname" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3c731c3e10504cc8ed35cfe2f1db4c9274c3d35fa486e3b31df46f068ef3e867" -dependencies = [ - "libc", - "match_cfg", - "winapi", -] - -[[package]] -name = "hstr" -version = "0.2.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "17fafeca18cf0927e23ea44d7a5189c10536279dfe9094e0dfa953053fbb5377" -dependencies = [ - "new_debug_unreachable", - "once_cell", - "phf", - "rustc-hash", - "smallvec", -] - -[[package]] -name = "http" -version = "0.2.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "601cbb57e577e2f5ef5be8e7b83f0f63994f25aa94d673e54a92d5c516d101f1" -dependencies = [ - "bytes", - "fnv", - "itoa", -] - -[[package]] -name = "http" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "21b9ddb458710bc376481b842f5da65cdf31522de232c1ca8146abce2a358258" -dependencies = [ - "bytes", - "fnv", - "itoa", -] - -[[package]] -name = "http-body" -version = "0.4.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7ceab25649e9960c0311ea418d17bee82c0dcec1bd053b5f9a66e265a693bed2" -dependencies = [ - "bytes", - "http 0.2.12", - "pin-project-lite", -] - -[[package]] -name = "http-body" -version = "1.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1cac85db508abc24a2e48553ba12a996e87244a0395ce011e62b37158745d643" -dependencies = [ - "bytes", - "http 1.1.0", -] - -[[package]] -name = "http-body-util" -version = "0.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0475f8b2ac86659c21b64320d5d653f9efe42acd2a4e560073ec61a155a34f1d" -dependencies = [ - "bytes", - "futures-core", - "http 1.1.0", - "http-body 1.0.0", - "pin-project-lite", -] - -[[package]] -name = "http-range" -version = "0.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "21dec9db110f5f872ed9699c3ecf50cf16f423502706ba5c72462e28d3157573" - -[[package]] -name = "http-range-header" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "add0ab9360ddbd88cfeb3bd9574a1d85cfdfa14db10b3e21d3700dbc4328758f" - -[[package]] -name = "httparse" -version = "1.8.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d897f394bad6a705d5f4104762e116a75639e470d80901eed05a860a95cb1904" - -[[package]] -name = "httpdate" -version = "1.0.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "df3b46402a9d5adb4c86a0cf463f42e19994e3ee891101b1841f30a545cb49a9" - -[[package]] -name = "humantime" -version = "2.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a3a5bfb195931eeb336b2a7b4d761daec841b97f947d34394601737a7bba5e4" - -[[package]] -name = "hyper" -version = "0.14.28" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bf96e135eb83a2a8ddf766e426a841d8ddd7449d5f00d34ea02b41d2f19eef80" -dependencies = [ - "bytes", - "futures-channel", - "futures-core", - "futures-util", - "h2 0.3.26", - "http 0.2.12", - "http-body 0.4.6", - "httparse", - "httpdate", - "itoa", - "pin-project-lite", - "socket2 0.4.10", - "tokio", - "tower-service", - "tracing", - "want", -] - -[[package]] -name = "hyper" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fb5aa53871fc917b1a9ed87b683a5d86db645e23acb32c2e0785a353e522fb75" -dependencies = [ - "bytes", - "futures-channel", - "futures-util", - "h2 0.4.4", - "http 1.1.0", - "http-body 1.0.0", - "httparse", - "httpdate", - "itoa", - "pin-project-lite", - "tokio", - "want", -] - -[[package]] -name = "hyper-rustls" -version = "0.24.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ec3efd23720e2049821a693cbc7e65ea87c72f1c58ff2f9522ff332b1491e590" -dependencies = [ - "futures-util", - "http 0.2.12", - "hyper 0.14.28", - "rustls 0.21.10", - "tokio", - "tokio-rustls 0.24.1", -] - -[[package]] -name = "hyper-timeout" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bbb958482e8c7be4bc3cf272a766a2b0bf1a6755e7a6ae777f017a31d11b13b1" -dependencies = [ - "hyper 0.14.28", - "pin-project-lite", - "tokio", - "tokio-io-timeout", -] - -[[package]] -name = "hyper-tls" -version = "0.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d6183ddfa99b85da61a140bea0efc93fdf56ceaa041b37d553518030827f9905" -dependencies = [ - "bytes", - "hyper 0.14.28", - "native-tls", - "tokio", - "tokio-native-tls", -] - -[[package]] -name = "hyper-util" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bdea9aac0dbe5a9240d68cfd9501e2db94222c6dc06843e06640b9e07f0fdc67" -dependencies = [ - "bytes", - "futures-channel", - "futures-util", - "http 1.1.0", - "http-body 1.0.0", - "hyper 1.1.0", - "pin-project-lite", - "socket2 0.5.6", - "tokio", - "tracing", -] - -[[package]] -name = "iana-time-zone" -version = "0.1.60" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e7ffbb5a1b541ea2561f8c41c087286cc091e21e556a4f09a8f6cbf17b69b141" -dependencies = [ - "android_system_properties", - "core-foundation-sys", - "iana-time-zone-haiku", - "js-sys", - "wasm-bindgen", - "windows-core", -] - -[[package]] -name = "iana-time-zone-haiku" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f31827a206f56af32e590ba56d5d2d085f558508192593743f16b2306495269f" -dependencies = [ - "cc", -] - -[[package]] -name = "ident_case" -version = "1.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b9e0384b61958566e926dc50660321d12159025e767c18e043daf26b70104c39" - -[[package]] -name = "idna" -version = "0.2.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "418a0a6fab821475f634efe3ccc45c013f742efe03d853e8d3355d5cb850ecf8" -dependencies = [ - "matches", - "unicode-bidi", - "unicode-normalization", -] - -[[package]] -name = "idna" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e14ddfc70884202db2244c223200c204c2bda1bc6e0998d11b5e024d657209e6" -dependencies = [ - "unicode-bidi", - "unicode-normalization", -] - -[[package]] -name = "idna" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7d20d6b07bfbc108882d88ed8e37d39636dcc260e15e30c45e6ba089610b917c" -dependencies = [ - "unicode-bidi", - "unicode-normalization", -] - -[[package]] -name = "idna" -version = "0.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "634d9b1461af396cad843f47fdba5597a4f9e6ddd4bfb6ff5d85028c25cb12f6" -dependencies = [ - "unicode-bidi", - "unicode-normalization", -] - -[[package]] -name = "if_chain" -version = "1.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cb56e1aa765b4b4f3aadfab769793b7087bb03a4ea4920644a6d238e2df5b9ed" - -[[package]] -name = "ijson" -version = "0.1.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b96214564d1f12875bd9661b183d8494dd10e373cb693629536fe2f3125e254b" -dependencies = [ - "dashmap 4.0.2", - "lazy_static", - "serde", - "serde_json", -] - -[[package]] -name = "impl-codec" -version = "0.6.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ba6a270039626615617f3f36d15fc827041df3b78c439da2cadfa47455a77f2f" -dependencies = [ - "parity-scale-codec", -] - -[[package]] -name = "impl-rlp" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f28220f89297a075ddc7245cd538076ee98b01f2a9c23a53a4f1105d5a322808" -dependencies = [ - "rlp", -] - -[[package]] -name = "impl-serde" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ebc88fc67028ae3db0c853baa36269d398d5f45b6982f95549ff5def78c935cd" -dependencies = [ - "serde", -] - -[[package]] -name = "impl-trait-for-tuples" -version = "0.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "11d7a9f6330b71fea57921c9b61c47ee6e84f72d394754eff6163ae67e7395eb" -dependencies = [ - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "include_dir" -version = "0.7.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "18762faeff7122e89e0857b02f7ce6fcc0d101d5e9ad2ad7846cc01d61b7f19e" -dependencies = [ - "include_dir_macros", -] - -[[package]] -name = "include_dir_macros" -version = "0.7.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b139284b5cf57ecfa712bcc66950bb635b31aff41c188e8a4cfc758eca374a3f" -dependencies = [ - "proc-macro2", - "quote", -] - -[[package]] -name = "indexmap" -version = "1.9.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bd070e393353796e801d209ad339e89596eb4c8d430d18ede6a1cced8fafbd99" -dependencies = [ - "autocfg", - "hashbrown 0.12.3", -] - -[[package]] -name = "indexmap" -version = "2.2.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7b0b929d511467233429c45a44ac1dcaa21ba0f5ba11e4879e6ed28ddb4f9df4" -dependencies = [ - "equivalent", - "hashbrown 0.14.3", - "serde", -] - -[[package]] -name = "indicatif" -version = "0.17.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "763a5a8f45087d6bcea4222e7b72c291a054edf80e4ef6efd2a4979878c7bea3" -dependencies = [ - "console", - "instant", - "number_prefix", - "portable-atomic", - "unicode-width", -] - -[[package]] -name = "indoc" -version = "1.0.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bfa799dd5ed20a7e349f3b4639aa80d74549c81716d9ec4f994c9b5815598306" - -[[package]] -name = "inotify" -version = "0.9.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8069d3ec154eb856955c1c0fbffefbf5f3c40a104ec912d4797314c1801abff" -dependencies = [ - "bitflags 1.3.2", - "inotify-sys", - "libc", -] - -[[package]] -name = "inotify-sys" -version = "0.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e05c02b5e89bff3b946cedeca278abc628fe811e604f027c45a8aa3cf793d0eb" -dependencies = [ - "libc", -] - -[[package]] -name = "inout" -version = "0.1.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a0c10553d664a4d0bcff9f4215d0aac67a639cc68ef660840afe309b807bc9f5" -dependencies = [ - "block-padding", - "generic-array", -] - -[[package]] -name = "instant" -version = "0.1.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7a5bbe824c507c5da5956355e86a746d82e0e1464f65d862cc5e71da70e94b2c" -dependencies = [ - "cfg-if", -] - -[[package]] -name = "integer-encoding" -version = "3.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8bb03732005da905c88227371639bf1ad885cc712789c011c31c5fb3ab3ccf02" - -[[package]] -name = "ipconfig" -version = "0.3.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b58db92f96b720de98181bbbe63c831e87005ab460c1bf306eb2622b4707997f" -dependencies = [ - "socket2 0.5.6", - "widestring", - "windows-sys 0.48.0", - "winreg", -] - -[[package]] -name = "ipnet" -version = "2.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8f518f335dce6725a761382244631d86cf0ccb2863413590b31338feb467f9c3" - -[[package]] -name = "iri-string" -version = "0.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "21859b667d66a4c1dacd9df0863b3efb65785474255face87f5bca39dd8407c0" -dependencies = [ - "memchr", - "serde", -] - -[[package]] -name = "is-macro" -version = "0.3.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "59a85abdc13717906baccb5a1e435556ce0df215f242892f721dff62bf25288f" -dependencies = [ - "Inflector", - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "is-terminal" -version = "0.4.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f23ff5ef2b80d608d61efee834934d862cd92461afc0560dedf493e4c033738b" -dependencies = [ - "hermit-abi", - "libc", - "windows-sys 0.52.0", -] - -[[package]] -name = "itertools" -version = "0.10.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b0fd2260e829bddf4cb6ea802289de2f86d6a7a690192fbe91b3f46e0f2c8473" -dependencies = [ - "either", -] - -[[package]] -name = "itertools" -version = "0.11.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b1c173a5686ce8bfa551b3563d0c2170bf24ca44da99c7ca4bfdab5418c3fe57" -dependencies = [ - "either", -] - -[[package]] -name = "itertools" -version = "0.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ba291022dbbd398a455acf126c1e341954079855bc60dfdda641363bd6922569" -dependencies = [ - "either", -] - -[[package]] -name = "itoa" -version = "1.0.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b1a46d1a171d865aa5f83f92695765caa047a9b4cbae2cbf37dbd613a793fd4c" - -[[package]] -name = "jni" -version = "0.21.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1a87aa2bb7d2af34197c04845522473242e1aa17c12f4935d5856491a7fb8c97" -dependencies = [ - "cesu8", - "cfg-if", - "combine", - "jni-sys", - "log", - "thiserror", - "walkdir", - "windows-sys 0.45.0", -] - -[[package]] -name = "jni-sys" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8eaf4bc02d17cbdd7ff4c7438cafcdf7fb9a4613313ad11b4f8fefe7d3fa0130" - -[[package]] -name = "jobserver" -version = "0.1.28" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ab46a6e9526ddef3ae7f787c06f0f2600639ba80ea3eade3d8e670a2230f51d6" -dependencies = [ - "libc", -] - -[[package]] -name = "js-sys" -version = "0.3.69" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "29c15563dc2726973df627357ce0c9ddddbea194836909d655df6a75d2cf296d" -dependencies = [ - "wasm-bindgen", -] - -[[package]] -name = "jsonpath" -version = "0.2.6" -dependencies = [ - "dozer-types", - "lazy_static", - "pest", - "pest_derive", - "regex", -] - -[[package]] -name = "jsonrpc-core" -version = "18.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "14f7f76aef2d054868398427f6c54943cf3d1caa9a7ec7d0c38d69df97a965eb" -dependencies = [ - "futures", - "futures-executor", - "futures-util", - "log", - "serde", - "serde_derive", - "serde_json", -] - -[[package]] -name = "keccak" -version = "0.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ecc2af9a1119c51f12a14607e783cb977bde58bc069ff0c3da1095e635d70654" -dependencies = [ - "cpufeatures", -] - -[[package]] -name = "keyed_priority_queue" -version = "0.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4ee7893dab2e44ae5f9d0173f26ff4aa327c10b01b06a72b52dd9405b628640d" -dependencies = [ - "indexmap 2.2.5", -] - -[[package]] -name = "kqueue" -version = "1.0.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7447f1ca1b7b563588a205fe93dea8df60fd981423a768bc1c0ded35ed147d0c" -dependencies = [ - "kqueue-sys", - "libc", -] - -[[package]] -name = "kqueue-sys" -version = "1.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ed9625ffda8729b85e45cf04090035ac368927b8cebc34898e7c120f52e4838b" -dependencies = [ - "bitflags 1.3.2", - "libc", -] - -[[package]] -name = "language-tags" -version = "0.3.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d4345964bb142484797b161f473a503a434de77149dd8c7427788c6e13379388" - -[[package]] -name = "lazy_static" -version = "1.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e2abad23fbc42b3700f2f279844dc832adb2b2eb069b2df918f455c4e18cc646" -dependencies = [ - "spin 0.5.2", -] - -[[package]] -name = "lazycell" -version = "1.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "830d08ce1d1d941e6b30645f1a0eb5643013d835ce3779a5fc208261dbe10f55" - -[[package]] -name = "lexical-core" -version = "0.8.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2cde5de06e8d4c2faabc400238f9ae1c74d5412d03a7bd067645ccbc47070e46" -dependencies = [ - "lexical-parse-float", - "lexical-parse-integer", - "lexical-util", - "lexical-write-float", - "lexical-write-integer", -] - -[[package]] -name = "lexical-parse-float" -version = "0.8.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "683b3a5ebd0130b8fb52ba0bdc718cc56815b6a097e28ae5a6997d0ad17dc05f" -dependencies = [ - "lexical-parse-integer", - "lexical-util", - "static_assertions", -] - -[[package]] -name = "lexical-parse-integer" -version = "0.8.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6d0994485ed0c312f6d965766754ea177d07f9c00c9b82a5ee62ed5b47945ee9" -dependencies = [ - "lexical-util", - "static_assertions", -] - -[[package]] -name = "lexical-util" -version = "0.8.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5255b9ff16ff898710eb9eb63cb39248ea8a5bb036bea8085b1a767ff6c4e3fc" -dependencies = [ - "static_assertions", -] - -[[package]] -name = "lexical-write-float" -version = "0.8.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "accabaa1c4581f05a3923d1b4cfd124c329352288b7b9da09e766b0668116862" -dependencies = [ - "lexical-util", - "lexical-write-integer", - "static_assertions", -] - -[[package]] -name = "lexical-write-integer" -version = "0.8.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e1b6f3d1f4422866b68192d62f77bc5c700bee84f3069f2469d7bc8c77852446" -dependencies = [ - "lexical-util", - "static_assertions", -] - -[[package]] -name = "libc" -version = "0.2.153" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9c198f91728a82281a64e1f4f9eeb25d82cb32a5de251c6bd1b5154d63a8e7bd" - -[[package]] -name = "libflate" -version = "2.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9f7d5654ae1795afc7ff76f4365c2c8791b0feb18e8996a96adad8ffd7c3b2bf" -dependencies = [ - "adler32", - "core2", - "crc32fast", - "dary_heap", - "libflate_lz77", -] - -[[package]] -name = "libflate_lz77" -version = "2.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "be5f52fb8c451576ec6b79d3f4deb327398bc05bbdbd99021a6e77a4c855d524" -dependencies = [ - "core2", - "hashbrown 0.13.2", - "rle-decode-fast", -] - -[[package]] -name = "libloading" -version = "0.7.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b67380fd3b2fbe7527a606e18729d21c6f3951633d0500574c4dc22d2d638b9f" -dependencies = [ - "cfg-if", - "winapi", -] - -[[package]] -name = "libloading" -version = "0.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c2a198fb6b0eada2a8df47933734e6d35d350665a33a3593d7164fa52c75c19" -dependencies = [ - "cfg-if", - "windows-targets 0.52.4", -] - -[[package]] -name = "libm" -version = "0.2.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4ec2a862134d2a7d32d7983ddcdd1c4923530833c9f2ea1a44fc5fa473989058" - -[[package]] -name = "libredox" -version = "0.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "85c833ca1e66078851dba29046874e38f08b2c883700aa29a03ddd3b23814ee8" -dependencies = [ - "bitflags 2.5.0", - "libc", - "redox_syscall", -] - -[[package]] -name = "libsqlite3-sys" -version = "0.26.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "afc22eff61b133b115c6e8c74e818c628d6d5e7a502afea6f64dee076dd94326" -dependencies = [ - "cc", - "pkg-config", - "vcpkg", -] - -[[package]] -name = "libtest-mimic" -version = "0.6.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6d8de370f98a6cb8a4606618e53e802f93b094ddec0f96988eaec2c27e6e9ce7" -dependencies = [ - "clap", - "termcolor", - "threadpool", -] - -[[package]] -name = "libz-sys" -version = "1.1.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "037731f5d3aaa87a5675e895b63ddff1a87624bc29f77004ea829809654e48f6" -dependencies = [ - "cc", - "libc", - "pkg-config", - "vcpkg", -] - -[[package]] -name = "like" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fc7281e4b2b1a1fae03463a7c49dd21464de50251a450f6da9715c40c7b21a70" - -[[package]] -name = "linked-hash-map" -version = "0.5.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0717cef1bc8b636c6e1c1bbdefc09e6322da8a9321966e8928ef80d20f7f770f" -dependencies = [ - "serde", -] - -[[package]] -name = "linux-raw-sys" -version = "0.4.13" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "01cda141df6706de531b6c46c3a33ecca755538219bd484262fa09410c13539c" - -[[package]] -name = "local-channel" -version = "0.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b6cbc85e69b8df4b8bb8b89ec634e7189099cea8927a276b7384ce5488e53ec8" -dependencies = [ - "futures-core", - "futures-sink", - "local-waker", -] - -[[package]] -name = "local-waker" -version = "0.1.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4d873d7c67ce09b42110d801813efbc9364414e356be9935700d368351657487" - -[[package]] -name = "lock_api" -version = "0.4.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3c168f8615b12bc01f9c17e2eb0cc07dcae1940121185446edc3744920e8ef45" -dependencies = [ - "autocfg", - "scopeguard", -] - -[[package]] -name = "log" -version = "0.4.20" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b5e6163cb8c49088c2c36f57875e58ccd8c87c7427f7fbd50ea6710b2f3f2e8f" -dependencies = [ - "serde", -] - -[[package]] -name = "logos" -version = "0.13.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c000ca4d908ff18ac99b93a062cb8958d331c3220719c52e77cb19cc6ac5d2c1" -dependencies = [ - "logos-derive", -] - -[[package]] -name = "logos-codegen" -version = "0.13.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc487311295e0002e452025d6b580b77bb17286de87b57138f3b5db711cded68" -dependencies = [ - "beef", - "fnv", - "proc-macro2", - "quote", - "regex-syntax 0.6.29", - "syn 2.0.53", -] - -[[package]] -name = "logos-derive" -version = "0.13.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dbfc0d229f1f42d790440136d941afd806bc9e949e2bcb8faa813b0f00d1267e" -dependencies = [ - "logos-codegen", -] - -[[package]] -name = "lru" -version = "0.12.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3262e75e648fce39813cb56ac41f3c3e3f65217ebf3844d818d1f9398cfb0dc" -dependencies = [ - "hashbrown 0.14.3", -] - -[[package]] -name = "lru-cache" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "31e24f1ad8321ca0e8a1e0ac13f23cb668e6f5466c2c57319f6a5cf1cc8e3b1c" -dependencies = [ - "linked-hash-map", -] - -[[package]] -name = "lz4" -version = "1.24.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e9e2dd86df36ce760a60f6ff6ad526f7ba1f14ba0356f8254fb6905e6494df1" -dependencies = [ - "libc", - "lz4-sys", -] - -[[package]] -name = "lz4-sys" -version = "1.9.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "57d27b317e207b10f69f5e75494119e391a96f48861ae870d1da6edac98ca900" -dependencies = [ - "cc", - "libc", -] - -[[package]] -name = "lz4_flex" -version = "0.11.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "912b45c753ff5f7f5208307e8ace7d2a2e30d024e26d3509f3dce546c044ce15" -dependencies = [ - "twox-hash", -] - -[[package]] -name = "lzma-sys" -version = "0.1.20" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5fda04ab3764e6cde78b9974eec4f779acaba7c4e84b36eca3cf77c581b85d27" -dependencies = [ - "cc", - "libc", - "pkg-config", -] - -[[package]] -name = "malloc_buf" -version = "0.0.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "62bb907fe88d54d8d9ce32a3cceab4218ed2f6b7d35617cafe9adf84e43919cb" -dependencies = [ - "libc", -] - -[[package]] -name = "maplit" -version = "1.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3e2e65a1a2e43cfcb47a895c4c8b10d1f4a61097f9f254f183aee60cad9c651d" - -[[package]] -name = "match_cfg" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ffbee8634e0d45d258acb448e7eaab3fce7a0a467395d4d9f228e3c1f01fb2e4" - -[[package]] -name = "matchers" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8263075bb86c5a1b1427b5ae862e8889656f126e9f77c484496e8b47cf5c5558" -dependencies = [ - "regex-automata 0.1.10", -] - -[[package]] -name = "matches" -version = "0.1.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2532096657941c2fea9c289d370a250971c689d4f143798ff67113ec042024a5" - -[[package]] -name = "matchit" -version = "0.7.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0e7465ac9959cc2b1404e8e2367b43684a6d13790fe23056cc8c6c5a6b7bcb94" - -[[package]] -name = "matrixmultiply" -version = "0.3.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7574c1cf36da4798ab73da5b215bbf444f50718207754cb522201d78d1cd0ff2" -dependencies = [ - "autocfg", - "rawpointer", -] - -[[package]] -name = "md-5" -version = "0.10.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d89e7ee0cfbedfc4da3340218492196241d89eefb6dab27de5df917a6d2e78cf" -dependencies = [ - "cfg-if", - "digest 0.10.7", -] - -[[package]] -name = "memchr" -version = "2.7.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "523dc4f511e55ab87b694dc30d0f820d60906ef06413f93d4d7a1385599cc149" - -[[package]] -name = "memoffset" -version = "0.8.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d61c719bcfbcf5d62b3a09efa6088de8c54bc0bfcd3ea7ae39fcc186108b8de1" -dependencies = [ - "autocfg", -] - -[[package]] -name = "memoffset" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5a634b1c61a95585bd15607c6ab0c4e5b226e695ff2800ba0cdccddf208c406c" -dependencies = [ - "autocfg", -] - -[[package]] -name = "mime" -version = "0.3.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6877bb514081ee2a7ff5ef9de3281f14a4dd4bceac4c09388074a6b5df8a139a" - -[[package]] -name = "mime_guess" -version = "2.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4192263c238a5f0d0c6bfd21f336a313a4ce1c450542449ca191bb657b4642ef" -dependencies = [ - "mime", - "unicase", -] - -[[package]] -name = "minimal-lexical" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "68354c5c6bd36d73ff3feceb05efa59b6acb7626617f4962be322a825e61f79a" - -[[package]] -name = "miniz_oxide" -version = "0.7.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9d811f3e15f28568be3407c8e7fdb6514c1cda3cb30683f15b6a1a1dc4ea14a7" -dependencies = [ - "adler", -] - -[[package]] -name = "mio" -version = "0.8.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a4a650543ca06a924e8b371db273b2756685faae30f8487da1b56505a8f78b0c" -dependencies = [ - "libc", - "log", - "wasi", - "windows-sys 0.48.0", -] - -[[package]] -name = "mongodb" -version = "2.8.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ef206acb1b72389b49bc9985efe7eb1f8a9bb18e5680d262fac26c07f44025f1" -dependencies = [ - "async-trait", - "base64 0.13.1", - "bitflags 1.3.2", - "bson", - "chrono", - "derivative", - "derive_more", - "futures-core", - "futures-executor", - "futures-io", - "futures-util", - "hex", - "hmac", - "lazy_static", - "md-5", - "pbkdf2", - "percent-encoding", - "rand", - "rustc_version_runtime", - "rustls 0.21.10", - "rustls-pemfile 1.0.4", - "serde", - "serde_bytes", - "serde_with", - "sha-1 0.10.0", - "sha2", - "socket2 0.4.10", - "stringprep", - "strsim 0.10.0", - "take_mut", - "thiserror", - "tokio", - "tokio-rustls 0.24.1", - "tokio-util", - "trust-dns-proto 0.21.2", - "trust-dns-resolver 0.21.2", - "typed-builder 0.10.0", - "uuid", - "webpki-roots 0.25.4", -] - -[[package]] -name = "multimap" -version = "0.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e5ce46fe64a9d73be07dcbe690a38ce1b293be448fd8ce1e6c1b8062c9f72c6a" - -[[package]] -name = "multimap" -version = "0.9.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e1a5d38b9b352dbd913288736af36af41c48d61b1a8cd34bcecd727561b7d511" -dependencies = [ - "serde", -] - -[[package]] -name = "mysql-common-derive" -version = "0.31.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c60492b5eb751e55b42d716b6b26dceb66767996cd7a5560a842fbf613ca2e92" -dependencies = [ - "darling 0.20.8", - "heck 0.4.1", - "num-bigint", - "proc-macro-crate 3.1.0", - "proc-macro-error 1.0.4", - "proc-macro2", - "quote", - "syn 2.0.53", - "termcolor", - "thiserror", -] - -[[package]] -name = "mysql_async" -version = "0.34.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fbfe87d7e35cb72363326216cc1712b865d8d4f70abf3b2d2e6b251fb6b2f427" -dependencies = [ - "bytes", - "crossbeam", - "flate2", - "futures-core", - "futures-sink", - "futures-util", - "keyed_priority_queue", - "lazy_static", - "lru", - "mio", - "mysql_common", - "once_cell", - "pem", - "percent-encoding", - "pin-project", - "rand", - "rustls 0.22.2", - "rustls-pemfile 2.1.1", - "serde", - "serde_json", - "socket2 0.5.6", - "thiserror", - "tokio", - "tokio-rustls 0.25.0", - "tokio-util", - "twox-hash", - "url", - "webpki", - "webpki-roots 0.26.1", -] - -[[package]] -name = "mysql_common" -version = "0.32.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8a60cb978c0a1d654edcc1460f8d6092dacf21346ed6017d81fb76a23ef5a8de" -dependencies = [ - "base64 0.21.7", - "bigdecimal 0.4.3", - "bindgen", - "bitflags 2.5.0", - "bitvec", - "btoi", - "byteorder", - "bytes", - "cc", - "chrono", - "cmake", - "crc32fast", - "flate2", - "frunk", - "lazy_static", - "mysql-common-derive", - "num-bigint", - "num-traits", - "rand", - "regex", - "rust_decimal", - "saturating", - "serde", - "serde_json", - "sha1", - "sha2", - "smallvec", - "subprocess", - "thiserror", - "time", - "uuid", - "zstd", -] - -[[package]] -name = "native-tls" -version = "0.2.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "07226173c32f2926027b63cce4bcd8076c3552846cbe7925f3aaffeac0a3b92e" -dependencies = [ - "lazy_static", - "libc", - "log", - "openssl", - "openssl-probe", - "openssl-sys", - "schannel", - "security-framework", - "security-framework-sys", - "tempfile", -] - -[[package]] -name = "ndarray" -version = "0.15.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "adb12d4e967ec485a5f71c6311fe28158e9d6f4bc4a447b474184d0f91a8fa32" -dependencies = [ - "matrixmultiply", - "num-complex", - "num-integer", - "num-traits", - "rawpointer", -] - -[[package]] -name = "ndk-context" -version = "0.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "27b02d87554356db9e9a873add8782d4ea6e3e58ea071a9adb9a2e8ddb884a8b" - -[[package]] -name = "new_debug_unreachable" -version = "1.0.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "650eef8c711430f1a879fdd01d4745a7deea475becfb90269c06775983bbf086" - -[[package]] -name = "nibble_vec" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "77a5d83df9f36fe23f0c3648c6bbb8b0298bb5f1939c8f2704431371f4b84d43" -dependencies = [ - "smallvec", -] - -[[package]] -name = "nix" -version = "0.27.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2eb04e9c688eff1c89d72b407f168cf79bb9e867a9d3323ed6c01519eb9cc053" -dependencies = [ - "bitflags 2.5.0", - "cfg-if", - "libc", -] - -[[package]] -name = "nom" -version = "7.1.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d273983c5a657a70a3e8f2a01329822f3b8c8172b73826411a55751e404a0a4a" -dependencies = [ - "memchr", - "minimal-lexical", -] - -[[package]] -name = "notify" -version = "6.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6205bd8bb1e454ad2e27422015fb5e4f2bcc7e08fa8f27058670d208324a4d2d" -dependencies = [ - "bitflags 2.5.0", - "crossbeam-channel", - "filetime", - "fsevent-sys", - "inotify", - "kqueue", - "libc", - "log", - "mio", - "walkdir", - "windows-sys 0.48.0", -] - -[[package]] -name = "notify-debouncer-full" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "416969970ec751a5d702a88c6cd19ac1332abe997fce43f96db0418550426241" -dependencies = [ - "crossbeam-channel", - "file-id", - "notify", - "parking_lot", - "walkdir", -] - -[[package]] -name = "nu-ansi-term" -version = "0.46.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "77a8165726e8236064dbb45459242600304b42a5ea24ee2948e18e023bf7ba84" -dependencies = [ - "overload", - "winapi", -] - -[[package]] -name = "num" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b05180d69e3da0e530ba2a1dae5110317e49e3b7f3d41be227dc5f92e49ee7af" -dependencies = [ - "num-bigint", - "num-complex", - "num-integer", - "num-iter", - "num-rational", - "num-traits", -] - -[[package]] -name = "num-bigint" -version = "0.4.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "608e7659b5c3d7cba262d894801b9ec9d00de989e8a82bd4bef91d08da45cdc0" -dependencies = [ - "autocfg", - "num-integer", - "num-traits", - "rand", - "serde", -] - -[[package]] -name = "num-bigint-dig" -version = "0.8.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc84195820f291c7697304f3cbdadd1cb7199c0efc917ff5eafd71225c136151" -dependencies = [ - "byteorder", - "lazy_static", - "libm", - "num-integer", - "num-iter", - "num-traits", - "rand", - "smallvec", - "zeroize", -] - -[[package]] -name = "num-complex" -version = "0.4.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "23c6602fda94a57c990fe0df199a035d83576b496aa29f4e634a8ac6004e68a6" -dependencies = [ - "num-traits", -] - -[[package]] -name = "num-conv" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "51d515d32fb182ee37cda2ccdcb92950d6a3c2893aa280e540671c2cd0f3b1d9" - -[[package]] -name = "num-integer" -version = "0.1.46" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7969661fd2958a5cb096e56c8e1ad0444ac2bbcd0061bd28660485a44879858f" -dependencies = [ - "num-traits", -] - -[[package]] -name = "num-iter" -version = "0.1.44" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d869c01cc0c455284163fd0092f1f93835385ccab5a98a0dcc497b2f8bf055a9" -dependencies = [ - "autocfg", - "num-integer", - "num-traits", -] - -[[package]] -name = "num-rational" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0638a1c9d0a3c0914158145bc76cff373a75a627e6ecbfb71cbe6f453a5a19b0" -dependencies = [ - "autocfg", - "num-bigint", - "num-integer", - "num-traits", -] - -[[package]] -name = "num-traits" -version = "0.2.18" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "da0df0e5185db44f69b44f26786fe401b6c293d1907744beaa7fa62b2e5a517a" -dependencies = [ - "autocfg", - "libm", -] - -[[package]] -name = "num_cpus" -version = "1.16.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4161fcb6d602d4d2081af7c3a45852d875a03dd337a6bfdd6e06407b61342a43" -dependencies = [ - "hermit-abi", - "libc", -] - -[[package]] -name = "num_enum" -version = "0.5.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1f646caf906c20226733ed5b1374287eb97e3c2a5c227ce668c1f2ce20ae57c9" -dependencies = [ - "num_enum_derive", -] - -[[package]] -name = "num_enum_derive" -version = "0.5.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dcbff9bc912032c62bf65ef1d5aea88983b420f4f839db1e9b0c281a25c9c799" -dependencies = [ - "proc-macro-crate 1.3.1", - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "number_prefix" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "830b246a0e5f20af87141b25c173cd1b609bd7779a4617d6ec582abaf90870f3" - -[[package]] -name = "objc" -version = "0.2.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "915b1b472bc21c53464d6c8461c9d3af805ba1ef837e1cac254428f4a77177b1" -dependencies = [ - "malloc_buf", -] - -[[package]] -name = "object" -version = "0.32.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a6a622008b6e321afc04970976f62ee297fdbaa6f95318ca343e3eebb9648441" -dependencies = [ - "memchr", -] - -[[package]] -name = "object_store" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d139f545f64630e2e3688fd9f81c470888ab01edeb72d13b4e86c566f1130000" -dependencies = [ - "async-trait", - "base64 0.21.7", - "bytes", - "chrono", - "futures", - "humantime", - "hyper 0.14.28", - "itertools 0.12.1", - "parking_lot", - "percent-encoding", - "quick-xml", - "rand", - "reqwest", - "ring", - "serde", - "serde_json", - "snafu", - "tokio", - "tracing", - "url", - "walkdir", -] - -[[package]] -name = "odbc" -version = "0.17.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1a2056ebb920918e743ccd0ceb6bc9d8e02e4d0d48c28550d1887e7490f6f298" -dependencies = [ - "doc-comment", - "encoding_rs", - "log", - "odbc-safe", - "odbc-sys", -] - -[[package]] -name = "odbc-safe" -version = "0.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f45de02ae2d07b38a7ef0e64139d971b1590d834c2ec089132d0eb49678e7e5a" -dependencies = [ - "odbc-sys", -] - -[[package]] -name = "odbc-sys" -version = "0.8.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dec5c5490e13c423d25508b13cd2cc432490043f5ed5b46b61a75857642f4f37" - -[[package]] -name = "once_cell" -version = "1.19.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3fdb12b2476b595f9358c5161aa467c2438859caa136dec86c26fdd2efe17b92" - -[[package]] -name = "oorandom" -version = "11.1.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ab1bc2a289d34bd04a330323ac98a1b4bc82c9d9fcb1e66b63caa84da26b575" - -[[package]] -name = "opaque-debug" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c08d65885ee38876c4f86fa503fb49d7b507c2b62552df7c70b2fce627e06381" - -[[package]] -name = "openssl" -version = "0.10.64" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "95a0481286a310808298130d22dd1fef0fa571e05a8f44ec801801e84b216b1f" -dependencies = [ - "bitflags 2.5.0", - "cfg-if", - "foreign-types", - "libc", - "once_cell", - "openssl-macros", - "openssl-sys", -] - -[[package]] -name = "openssl-macros" -version = "0.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a948666b637a0f465e8564c73e89d4dde00d72d4d473cc972f390fc3dcee7d9c" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "openssl-probe" -version = "0.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ff011a302c396a5197692431fc1948019154afc178baf7d8e37367442a4601cf" - -[[package]] -name = "openssl-sys" -version = "0.9.101" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dda2b0f344e78efc2facf7d195d098df0dd72151b26ab98da807afc26c198dff" -dependencies = [ - "cc", - "libc", - "pkg-config", - "vcpkg", -] - -[[package]] -name = "opentelemetry" -version = "0.22.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "900d57987be3f2aeb70d385fff9b27fb74c5723cc9a52d904d4f9c807a0667bf" -dependencies = [ - "futures-core", - "futures-sink", - "js-sys", - "once_cell", - "pin-project-lite", - "thiserror", - "urlencoding", -] - -[[package]] -name = "opentelemetry-aws" -version = "0.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "42c915961c059be65af3be9aeaedb3f8930e1e9590e26e5f23b1919e90d1ed7d" -dependencies = [ - "once_cell", - "opentelemetry", - "opentelemetry_sdk", -] - -[[package]] -name = "opentelemetry-otlp" -version = "0.15.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1a016b8d9495c639af2145ac22387dcb88e44118e45320d9238fbf4e7889abcb" -dependencies = [ - "async-trait", - "futures-core", - "http 0.2.12", - "opentelemetry", - "opentelemetry-proto", - "opentelemetry-semantic-conventions", - "opentelemetry_sdk", - "prost", - "thiserror", - "tokio", - "tonic 0.11.0", -] - -[[package]] -name = "opentelemetry-prometheus" -version = "0.15.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "30bbcf6341cab7e2193e5843f0ac36c446a5b3fccb28747afaeda17996dcd02e" -dependencies = [ - "once_cell", - "opentelemetry", - "opentelemetry_sdk", - "prometheus", - "protobuf", -] - -[[package]] -name = "opentelemetry-proto" -version = "0.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3a8fddc9b68f5b80dae9d6f510b88e02396f006ad48cac349411fbecc80caae4" -dependencies = [ - "opentelemetry", - "opentelemetry_sdk", - "prost", - "tonic 0.11.0", -] - -[[package]] -name = "opentelemetry-semantic-conventions" -version = "0.14.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f9ab5bd6c42fb9349dcf28af2ba9a0667f697f9bdcca045d39f2cec5543e2910" - -[[package]] -name = "opentelemetry_sdk" -version = "0.22.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9e90c7113be649e31e9a0f8b5ee24ed7a16923b322c3c5ab6367469c049d6b7e" -dependencies = [ - "async-trait", - "crossbeam-channel", - "futures-channel", - "futures-executor", - "futures-util", - "glob", - "once_cell", - "opentelemetry", - "ordered-float 4.2.0", - "percent-encoding", - "rand", - "thiserror", - "tokio", - "tokio-stream", -] - -[[package]] -name = "ordered-float" -version = "2.10.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "68f19d67e5a2795c94e73e0bb1cc1a7edeb2e28efd39e2e1c9b7a40c1108b11c" -dependencies = [ - "num-traits", -] - -[[package]] -name = "ordered-float" -version = "3.9.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f1e1c390732d15f1d48471625cd92d154e66db2c56645e29a9cd26f4699f72dc" -dependencies = [ - "num-traits", - "rand", - "serde", -] - -[[package]] -name = "ordered-float" -version = "4.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a76df7075c7d4d01fdcb46c912dd17fba5b60c78ea480b475f2b6ab6f666584e" -dependencies = [ - "num-traits", -] - -[[package]] -name = "ort" -version = "1.16.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "889dca4c98efa21b1ba54ddb2bde44fd4920d910f492b618351f839d8428d79d" -dependencies = [ - "flate2", - "half", - "lazy_static", - "libc", - "ndarray", - "tar", - "thiserror", - "tracing", - "ureq", - "vswhom", - "winapi", - "zip", -] - -[[package]] -name = "outref" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7f222829ae9293e33a9f5e9f440c6760a3d450a64affe1846486b140db81c1f4" - -[[package]] -name = "outref" -version = "0.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4030760ffd992bef45b0ae3f10ce1aba99e33464c90d14dd7c039884963ddc7a" - -[[package]] -name = "overload" -version = "0.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b15813163c1d831bf4a13c3610c05c0d03b39feb07f7e09fa234dac9b15aaf39" - -[[package]] -name = "owo-colors" -version = "3.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c1b04fb49957986fdce4d6ee7a65027d55d4b6d2265e5848bbb507b58ccfdb6f" - -[[package]] -name = "p256" -version = "0.13.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c9863ad85fa8f4460f9c48cb909d38a0d689dba1f6f6988a5e3e0d31071bcd4b" -dependencies = [ - "ecdsa", - "elliptic-curve", - "primeorder", - "sha2", -] - -[[package]] -name = "p384" -version = "0.13.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "70786f51bcc69f6a4c0360e063a4cac5419ef7c5cd5b3c99ad70f3be5ba79209" -dependencies = [ - "ecdsa", - "elliptic-curve", - "primeorder", - "sha2", -] - -[[package]] -name = "p521" -version = "0.13.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0fc9e2161f1f215afdfce23677034ae137bbd45016a880c2eb3ba8eb95f085b2" -dependencies = [ - "base16ct", - "ecdsa", - "elliptic-curve", - "primeorder", - "rand_core", - "sha2", -] - -[[package]] -name = "page_size" -version = "0.6.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "30d5b2194ed13191c1999ae0704b7839fb18384fa22e49b57eeaa97d79ce40da" -dependencies = [ - "libc", - "winapi", -] - -[[package]] -name = "parity-scale-codec" -version = "3.6.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "881331e34fa842a2fb61cc2db9643a8fedc615e47cfcc52597d1af0db9a7e8fe" -dependencies = [ - "arrayvec", - "bitvec", - "byte-slice-cast", - "impl-trait-for-tuples", - "parity-scale-codec-derive", - "serde", -] - -[[package]] -name = "parity-scale-codec-derive" -version = "3.6.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "be30eaf4b0a9fba5336683b38de57bb86d179a35862ba6bfcf57625d006bde5b" -dependencies = [ - "proc-macro-crate 2.0.0", - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "parking_lot" -version = "0.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3742b2c103b9f06bc9fff0a37ff4912935851bee6d36f3c02bcc755bcfec228f" -dependencies = [ - "lock_api", - "parking_lot_core", -] - -[[package]] -name = "parking_lot_core" -version = "0.9.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4c42a9226546d68acdd9c0a280d17ce19bfe27a46bf68784e4066115788d008e" -dependencies = [ - "cfg-if", - "libc", - "redox_syscall", - "smallvec", - "windows-targets 0.48.5", -] - -[[package]] -name = "parquet" -version = "50.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "547b92ebf0c1177e3892f44c8f79757ee62e678d564a9834189725f2c5b7a750" -dependencies = [ - "ahash 0.8.11", - "arrow-array", - "arrow-buffer", - "arrow-cast", - "arrow-data", - "arrow-ipc", - "arrow-schema", - "arrow-select", - "base64 0.21.7", - "brotli", - "bytes", - "chrono", - "flate2", - "futures", - "half", - "hashbrown 0.14.3", - "lz4_flex", - "num", - "num-bigint", - "object_store", - "paste", - "seq-macro", - "snap", - "thrift", - "tokio", - "twox-hash", - "zstd", -] - -[[package]] -name = "parse-zoneinfo" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c705f256449c60da65e11ff6626e0c16a0a0b96aaa348de61376b249bc340f41" -dependencies = [ - "regex", -] - -[[package]] -name = "paste" -version = "1.0.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "de3145af08024dea9fa9914f381a17b8fc6034dfb00f3a84013f7ff43f29ed4c" - -[[package]] -name = "pathdiff" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8835116a5c179084a830efb3adc117ab007512b535bc1a21c991d3b32a6b44dd" - -[[package]] -name = "pbkdf2" -version = "0.11.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "83a0692ec44e4cf1ef28ca317f14f8f07da2d95ec3fa01f86e4467b725e60917" -dependencies = [ - "digest 0.10.7", -] - -[[package]] -name = "pem" -version = "3.0.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1b8fcc794035347fb64beda2d3b462595dd2753e3f268d89c5aae77e8cf2c310" -dependencies = [ - "base64 0.21.7", - "serde", -] - -[[package]] -name = "pem-rfc7468" -version = "0.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "88b39c9bfcfc231068454382784bb460aae594343fb030d46e9f50a645418412" -dependencies = [ - "base64ct", -] - -[[package]] -name = "percent-encoding" -version = "2.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e3148f5046208a5d56bcfc03053e3ca6334e51da8dfb19b6cdc8b306fae3283e" - -[[package]] -name = "pest" -version = "2.7.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "56f8023d0fb78c8e03784ea1c7f3fa36e68a723138990b8d5a47d916b651e7a8" -dependencies = [ - "memchr", - "thiserror", - "ucd-trie", -] - -[[package]] -name = "pest_derive" -version = "2.7.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b0d24f72393fd16ab6ac5738bc33cdb6a9aa73f8b902e8fe29cf4e67d7dd1026" -dependencies = [ - "pest", - "pest_generator", -] - -[[package]] -name = "pest_generator" -version = "2.7.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fdc17e2a6c7d0a492f0158d7a4bd66cc17280308bbaff78d5bef566dca35ab80" -dependencies = [ - "pest", - "pest_meta", - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "pest_meta" -version = "2.7.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "934cd7631c050f4674352a6e835d5f6711ffbfb9345c2fc0107155ac495ae293" -dependencies = [ - "once_cell", - "pest", - "sha2", -] - -[[package]] -name = "petgraph" -version = "0.6.3" -source = "git+https://github.com/getdozer/petgraph?branch=feat/try_map#2422a3a21d92e7a6446f0d0f172c8aa24258846e" -dependencies = [ - "fixedbitset", - "indexmap 1.9.3", - "serde", - "serde_derive", -] - -[[package]] -name = "petgraph" -version = "0.6.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e1d3afd2628e69da2be385eb6f2fd57c8ac7977ceeff6dc166ff1657b0e386a9" -dependencies = [ - "fixedbitset", - "indexmap 2.2.5", -] - -[[package]] -name = "phf" -version = "0.11.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ade2d8b8f33c7333b51bcf0428d37e217e9f32192ae4772156f65063b8ce03dc" -dependencies = [ - "phf_macros", - "phf_shared", -] - -[[package]] -name = "phf_codegen" -version = "0.11.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e8d39688d359e6b34654d328e262234662d16cc0f60ec8dcbe5e718709342a5a" -dependencies = [ - "phf_generator", - "phf_shared", -] - -[[package]] -name = "phf_generator" -version = "0.11.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "48e4cc64c2ad9ebe670cb8fd69dd50ae301650392e81c05f9bfcb2d5bdbc24b0" -dependencies = [ - "phf_shared", - "rand", -] - -[[package]] -name = "phf_macros" -version = "0.11.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3444646e286606587e49f3bcf1679b8cef1dc2c5ecc29ddacaffc305180d464b" -dependencies = [ - "phf_generator", - "phf_shared", - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "phf_shared" -version = "0.11.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "90fcb95eef784c2ac79119d1dd819e162b5da872ce6f3c3abe1e8ca1c082f72b" -dependencies = [ - "siphasher", -] - -[[package]] -name = "pin-project" -version = "1.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b6bf43b791c5b9e34c3d182969b4abb522f9343702850a2e57f460d00d09b4b3" -dependencies = [ - "pin-project-internal", -] - -[[package]] -name = "pin-project-internal" -version = "1.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2f38a4412a78282e09a2cf38d195ea5420d15ba0602cb375210efbc877243965" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "pin-project-lite" -version = "0.2.13" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8afb450f006bf6385ca15ef45d71d2288452bc3683ce2e2cacc0d18e4be60b58" - -[[package]] -name = "pin-utils" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b870d8c151b6f2fb93e84a13146138f05d02ed11c7e7c54f8826aaaf7c9f184" - -[[package]] -name = "pkcs1" -version = "0.7.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c8ffb9f10fa047879315e6625af03c164b16962a5368d724ed16323b68ace47f" -dependencies = [ - "der", - "pkcs8", - "spki", -] - -[[package]] -name = "pkcs8" -version = "0.10.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f950b2377845cebe5cf8b5165cb3cc1a5e0fa5cfa3e1f7f55707d8fd82e0a7b7" -dependencies = [ - "der", - "spki", -] - -[[package]] -name = "pkg-config" -version = "0.3.30" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d231b230927b5e4ad203db57bbcbee2802f6bce620b1e4a9024a07d94e2907ec" - -[[package]] -name = "platforms" -version = "3.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "626dec3cac7cc0e1577a2ec3fc496277ec2baa084bebad95bb6fdbfae235f84c" - -[[package]] -name = "plotters" -version = "0.3.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d2c224ba00d7cadd4d5c660deaf2098e5e80e07846537c51f9cfa4be50c1fd45" -dependencies = [ - "num-traits", - "plotters-backend", - "plotters-svg", - "wasm-bindgen", - "web-sys", -] - -[[package]] -name = "plotters-backend" -version = "0.3.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9e76628b4d3a7581389a35d5b6e2139607ad7c75b17aed325f210aa91f4a9609" - -[[package]] -name = "plotters-svg" -version = "0.3.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "38f6d39893cca0701371e3c27294f09797214b86f1fb951b89ade8ec04e2abab" -dependencies = [ - "plotters-backend", -] - -[[package]] -name = "pmutil" -version = "0.6.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "52a40bc70c2c58040d2d8b167ba9a5ff59fc9dab7ad44771cfde3dcfde7a09c6" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "polyval" -version = "0.6.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9d1fe60d06143b2430aa532c94cfe9e29783047f06c0d7fd359a9a51b729fa25" -dependencies = [ - "cfg-if", - "cpufeatures", - "opaque-debug", - "universal-hash", -] - -[[package]] -name = "portable-atomic" -version = "1.6.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7170ef9988bc169ba16dd36a7fa041e5c4cbeb6a35b76d4c03daded371eae7c0" - -[[package]] -name = "postgres" -version = "0.19.5" -source = "git+https://github.com/MaterializeInc/rust-postgres#b759caa33610403aa74b1cfdd37f45eb3100c9af" -dependencies = [ - "bytes", - "fallible-iterator", - "futures-util", - "log", - "tokio", - "tokio-postgres", -] - -[[package]] -name = "postgres-protocol" -version = "0.6.5" -source = "git+https://github.com/MaterializeInc/rust-postgres#b759caa33610403aa74b1cfdd37f45eb3100c9af" -dependencies = [ - "base64 0.21.7", - "byteorder", - "bytes", - "fallible-iterator", - "hmac", - "md-5", - "memchr", - "rand", - "sha2", - "stringprep", -] - -[[package]] -name = "postgres-types" -version = "0.2.5" -source = "git+https://github.com/MaterializeInc/rust-postgres#b759caa33610403aa74b1cfdd37f45eb3100c9af" -dependencies = [ - "bytes", - "chrono", - "fallible-iterator", - "geo-types", - "postgres-protocol", - "serde", - "serde_json", - "uuid", -] - -[[package]] -name = "powerfmt" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "439ee305def115ba05938db6eb1644ff94165c5ab5e9420d1c1bcedbba909391" - -[[package]] -name = "ppv-lite86" -version = "0.2.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5b40af805b3121feab8a3c29f04d8ad262fa8e0561883e7653e024ae4479e6de" - -[[package]] -name = "prettyplease" -version = "0.2.16" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a41cf62165e97c7f814d2221421dbb9afcbcdb0a88068e5ea206e19951c2cbb5" -dependencies = [ - "proc-macro2", - "syn 2.0.53", -] - -[[package]] -name = "prettytable-rs" -version = "0.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eea25e07510aa6ab6547308ebe3c036016d162b8da920dbb079e3ba8acf3d95a" -dependencies = [ - "csv", - "encode_unicode 1.0.0", - "is-terminal", - "lazy_static", - "term", - "unicode-width", -] - -[[package]] -name = "primeorder" -version = "0.13.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "353e1ca18966c16d9deb1c69278edbc5f194139612772bd9537af60ac231e1e6" -dependencies = [ - "elliptic-curve", -] - -[[package]] -name = "primitive-types" -version = "0.12.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b34d9fd68ae0b74a41b21c03c2f62847aa0ffea044eee893b4c140b37e244e2" -dependencies = [ - "fixed-hash", - "impl-codec", - "impl-rlp", - "impl-serde", - "uint", -] - -[[package]] -name = "proc-macro-crate" -version = "1.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7f4c021e1093a56626774e81216a4ce732a735e5bad4868a03f3ed65ca0c3919" -dependencies = [ - "once_cell", - "toml_edit 0.19.15", -] - -[[package]] -name = "proc-macro-crate" -version = "2.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e8366a6159044a37876a2b9817124296703c586a5c92e2c53751fa06d8d43e8" -dependencies = [ - "toml_edit 0.20.7", -] - -[[package]] -name = "proc-macro-crate" -version = "3.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6d37c51ca738a55da99dc0c4a34860fd675453b8b36209178c2249bb13651284" -dependencies = [ - "toml_edit 0.21.1", -] - -[[package]] -name = "proc-macro-error" -version = "0.4.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "18f33027081eba0a6d8aba6d1b1c3a3be58cbb12106341c2d5759fcd9b5277e7" -dependencies = [ - "proc-macro-error-attr 0.4.12", - "proc-macro2", - "quote", - "syn 1.0.109", - "version_check", -] - -[[package]] -name = "proc-macro-error" -version = "1.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "da25490ff9892aab3fcf7c36f08cfb902dd3e71ca0f9f9517bea02a73a5ce38c" -dependencies = [ - "proc-macro-error-attr 1.0.4", - "proc-macro2", - "quote", - "syn 1.0.109", - "version_check", -] - -[[package]] -name = "proc-macro-error-attr" -version = "0.4.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8a5b4b77fdb63c1eca72173d68d24501c54ab1269409f6b672c85deb18af69de" -dependencies = [ - "proc-macro2", - "quote", - "syn 1.0.109", - "syn-mid", - "version_check", -] - -[[package]] -name = "proc-macro-error-attr" -version = "1.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a1be40180e52ecc98ad80b184934baf3d0d29f979574e439af5a55274b35f869" -dependencies = [ - "proc-macro2", - "quote", - "version_check", -] - -[[package]] -name = "proc-macro-hack" -version = "0.5.20+deprecated" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc375e1527247fe1a97d8b7156678dfe7c1af2fc075c9a4db3690ecd2a148068" - -[[package]] -name = "proc-macro-rules" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "07c277e4e643ef00c1233393c673f655e3672cf7eb3ba08a00bdd0ea59139b5f" -dependencies = [ - "proc-macro-rules-macros", - "proc-macro2", - "syn 2.0.53", -] - -[[package]] -name = "proc-macro-rules-macros" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "207fffb0fe655d1d47f6af98cc2793405e85929bdbc420d685554ff07be27ac7" -dependencies = [ - "once_cell", - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "proc-macro2" -version = "1.0.79" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e835ff2298f5721608eb1a980ecaee1aef2c132bf95ecc026a11b7bf3c01c02e" -dependencies = [ - "unicode-ident", -] - -[[package]] -name = "prometheus" -version = "0.13.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "449811d15fbdf5ceb5c1144416066429cf82316e2ec8ce0c1f6f8a02e7bbcf8c" -dependencies = [ - "cfg-if", - "fnv", - "lazy_static", - "memchr", - "parking_lot", - "protobuf", - "thiserror", -] - -[[package]] -name = "prometheus-parse" -version = "0.2.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "811031bea65e5a401fb2e1f37d802cca6601e204ac463809a3189352d13b78a5" -dependencies = [ - "chrono", - "itertools 0.12.1", - "once_cell", - "regex", -] - -[[package]] -name = "proptest" -version = "1.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "31b476131c3c86cb68032fdc5cb6d5a1045e3e42d96b69fa599fd77701e1f5bf" -dependencies = [ - "bit-set", - "bit-vec", - "bitflags 2.5.0", - "lazy_static", - "num-traits", - "rand", - "rand_chacha", - "rand_xorshift", - "regex-syntax 0.8.2", - "rusty-fork", - "tempfile", - "unarray", -] - -[[package]] -name = "prost" -version = "0.12.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "146c289cda302b98a28d40c8b3b90498d6e526dd24ac2ecea73e4e491685b94a" -dependencies = [ - "bytes", - "prost-derive", -] - -[[package]] -name = "prost-build" -version = "0.12.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c55e02e35260070b6f716a2423c2ff1c3bb1642ddca6f99e1f26d06268a0e2d2" -dependencies = [ - "bytes", - "heck 0.4.1", - "itertools 0.11.0", - "log", - "multimap 0.8.3", - "once_cell", - "petgraph 0.6.4", - "prettyplease", - "prost", - "prost-types", - "regex", - "syn 2.0.53", - "tempfile", - "which 4.4.2", -] - -[[package]] -name = "prost-derive" -version = "0.12.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "efb6c9a1dd1def8e2124d17e83a20af56f1570d6c2d2bd9e266ccb768df3840e" -dependencies = [ - "anyhow", - "itertools 0.11.0", - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "prost-reflect" -version = "0.12.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "057237efdb71cf4b3f9396302a3d6599a92fa94063ba537b66130980ea9909f3" -dependencies = [ - "base64 0.21.7", - "logos", - "once_cell", - "prost", - "prost-types", - "serde", - "serde-value", -] - -[[package]] -name = "prost-types" -version = "0.12.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "193898f59edcf43c26227dcd4c8427f00d99d61e95dcde58dabd49fa291d470e" -dependencies = [ - "prost", -] - -[[package]] -name = "protobuf" -version = "2.28.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "106dd99e98437432fed6519dedecfade6a06a73bb7b2a1e019fdd2bee5778d94" - -[[package]] -name = "psl-types" -version = "2.0.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "33cb294fe86a74cbcf50d4445b37da762029549ebeea341421c7c70370f86cac" - -[[package]] -name = "psm" -version = "0.1.21" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5787f7cda34e3033a72192c018bc5883100330f362ef279a8cbccfce8bb4e874" -dependencies = [ - "cc", -] - -[[package]] -name = "ptr_meta" -version = "0.1.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0738ccf7ea06b608c10564b31debd4f5bc5e197fc8bfe088f68ae5ce81e7a4f1" -dependencies = [ - "ptr_meta_derive", -] - -[[package]] -name = "ptr_meta_derive" -version = "0.1.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "16b845dbfca988fa33db069c0e230574d15a3088f147a87b64c7589eb662c9ac" -dependencies = [ - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "publicsuffix" -version = "2.2.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "96a8c1bda5ae1af7f99a2962e49df150414a43d62404644d98dd5c3a93d07457" -dependencies = [ - "idna 0.3.0", - "psl-types", -] - -[[package]] -name = "pyo3" -version = "0.18.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e3b1ac5b3731ba34fdaa9785f8d74d17448cd18f30cf19e0c7e7b1fdb5272109" -dependencies = [ - "cfg-if", - "indoc", - "libc", - "memoffset 0.8.0", - "parking_lot", - "pyo3-build-config", - "pyo3-ffi", - "pyo3-macros", - "unindent", -] - -[[package]] -name = "pyo3-build-config" -version = "0.18.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9cb946f5ac61bb61a5014924910d936ebd2b23b705f7a4a3c40b05c720b079a3" -dependencies = [ - "once_cell", - "target-lexicon", -] - -[[package]] -name = "pyo3-ffi" -version = "0.18.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fd4d7c5337821916ea2a1d21d1092e8443cf34879e53a0ac653fbb98f44ff65c" -dependencies = [ - "libc", - "pyo3-build-config", -] - -[[package]] -name = "pyo3-macros" -version = "0.18.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a9d39c55dab3fc5a4b25bbd1ac10a2da452c4aca13bb450f22818a002e29648d" -dependencies = [ - "proc-macro2", - "pyo3-macros-backend", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "pyo3-macros-backend" -version = "0.18.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "97daff08a4c48320587b5224cc98d609e3c27b6d437315bd40b605c98eeb5918" -dependencies = [ - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "quad-rand" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "658fa1faf7a4cc5f057c9ee5ef560f717ad9d8dc66d975267f709624d6e1ab88" - -[[package]] -name = "quick-error" -version = "1.2.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a1d01941d82fa2ab50be1e79e6714289dd7cde78eba4c074bc5a4374f650dfe0" - -[[package]] -name = "quick-xml" -version = "0.31.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1004a344b30a54e2ee58d66a71b32d2db2feb0a31f9a2d302bf0536f15de2a33" -dependencies = [ - "memchr", - "serde", -] - -[[package]] -name = "quote" -version = "1.0.35" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "291ec9ab5efd934aaf503a6466c5d5251535d108ee747472c3977cc5acc868ef" -dependencies = [ - "proc-macro2", -] - -[[package]] -name = "radium" -version = "0.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc33ff2d4973d518d823d61aa239014831e521c75da58e3df4840d3f47749d09" - -[[package]] -name = "radix_trie" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c069c179fcdc6a2fe24d8d18305cf085fdbd4f922c041943e203685d6a1c58fd" -dependencies = [ - "endian-type", - "nibble_vec", -] - -[[package]] -name = "rand" -version = "0.8.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "34af8d1a0e25924bc5b7c43c079c942339d8f0a8b57c39049bef581b46327404" -dependencies = [ - "libc", - "rand_chacha", - "rand_core", - "serde", -] - -[[package]] -name = "rand_chacha" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e6c10a63a0fa32252be49d21e7709d4d4baf8d231c2dbce1eaa8141b9b127d88" -dependencies = [ - "ppv-lite86", - "rand_core", -] - -[[package]] -name = "rand_core" -version = "0.6.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ec0be4795e2f6a28069bec0b5ff3e2ac9bafc99e6a9a7dc3547996c5c816922c" -dependencies = [ - "getrandom", - "serde", -] - -[[package]] -name = "rand_xorshift" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d25bf25ec5ae4a3f1b92f929810509a2f53d7dca2f50b794ff57e3face536c8f" -dependencies = [ - "rand_core", -] - -[[package]] -name = "raw-window-handle" -version = "0.5.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2ff9a1f06a88b01621b7ae906ef0211290d1c8a168a15542486a8f61c0833b9" - -[[package]] -name = "rawpointer" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "60a357793950651c4ed0f3f52338f53b2f809f32d83a07f72909fa13e4c6c1e3" - -[[package]] -name = "rayon" -version = "1.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e4963ed1bc86e4f3ee217022bd855b297cef07fb9eac5dfa1f788b220b49b3bd" -dependencies = [ - "either", - "rayon-core", -] - -[[package]] -name = "rayon-core" -version = "1.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1465873a3dfdaa8ae7cb14b4383657caab0b3e8a0aa9ae8e04b044854c8dfce2" -dependencies = [ - "crossbeam-deque", - "crossbeam-utils", -] - -[[package]] -name = "rdkafka" -version = "0.36.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1beea247b9a7600a81d4cc33f659ce1a77e1988323d7d2809c7ed1c21f4c316d" -dependencies = [ - "futures-channel", - "futures-util", - "libc", - "log", - "rdkafka-sys", - "serde", - "serde_derive", - "serde_json", - "slab", - "tokio", -] - -[[package]] -name = "rdkafka-sys" -version = "4.7.0+2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "55e0d2f9ba6253f6ec72385e453294f8618e9e15c2c6aba2a5c01ccf9622d615" -dependencies = [ - "libc", - "libz-sys", - "num_enum", - "pkg-config", -] - -[[package]] -name = "redox_syscall" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4722d768eff46b75989dd134e5c353f0d6296e5aaa3132e776cbdb56be7731aa" -dependencies = [ - "bitflags 1.3.2", -] - -[[package]] -name = "redox_users" -version = "0.4.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a18479200779601e498ada4e8c1e1f50e3ee19deb0259c25825a98b5603b2cb4" -dependencies = [ - "getrandom", - "libredox", - "thiserror", -] - -[[package]] -name = "regex" -version = "1.10.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b62dbe01f0b06f9d8dc7d49e05a0785f153b00b2c227856282f671e0318c9b15" -dependencies = [ - "aho-corasick", - "memchr", - "regex-automata 0.4.6", - "regex-syntax 0.8.2", -] - -[[package]] -name = "regex-automata" -version = "0.1.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6c230d73fb8d8c1b9c0b3135c5142a8acee3a0558fb8db5cf1cb65f8d7862132" -dependencies = [ - "regex-syntax 0.6.29", -] - -[[package]] -name = "regex-automata" -version = "0.4.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "86b83b8b9847f9bf95ef68afb0b8e6cdb80f498442f5179a29fad448fcc1eaea" -dependencies = [ - "aho-corasick", - "memchr", - "regex-syntax 0.8.2", -] - -[[package]] -name = "regex-lite" -version = "0.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "30b661b2f27137bdbc16f00eda72866a92bb28af1753ffbd56744fb6e2e9cd8e" - -[[package]] -name = "regex-syntax" -version = "0.6.29" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f162c6dd7b008981e4d40210aca20b4bd0f9b60ca9271061b07f78537722f2e1" - -[[package]] -name = "regex-syntax" -version = "0.8.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c08c74e62047bb2de4ff487b251e4a92e24f48745648451635cec7d591162d9f" - -[[package]] -name = "rend" -version = "0.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "71fe3824f5629716b1589be05dacd749f6aa084c87e00e016714a8cdfccc997c" -dependencies = [ - "bytecheck", -] - -[[package]] -name = "reqwest" -version = "0.11.20" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3e9ad3fe7488d7e34558a2033d45a0c90b72d97b4f80705666fea71472e2e6a1" -dependencies = [ - "async-compression", - "base64 0.21.7", - "bytes", - "cookie", - "cookie_store", - "encoding_rs", - "futures-core", - "futures-util", - "h2 0.3.26", - "http 0.2.12", - "http-body 0.4.6", - "hyper 0.14.28", - "hyper-rustls", - "hyper-tls", - "ipnet", - "js-sys", - "log", - "mime", - "native-tls", - "once_cell", - "percent-encoding", - "pin-project-lite", - "rustls 0.21.10", - "rustls-native-certs 0.6.3", - "rustls-pemfile 1.0.4", - "serde", - "serde_json", - "serde_urlencoded", - "tokio", - "tokio-native-tls", - "tokio-rustls 0.24.1", - "tokio-socks", - "tokio-util", - "tower-service", - "url", - "wasm-bindgen", - "wasm-bindgen-futures", - "wasm-streams", - "web-sys", - "webpki-roots 0.25.4", - "winreg", -] - -[[package]] -name = "resolv-conf" -version = "0.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "52e44394d2086d010551b14b53b1f24e31647570cd1deb0379e2c21b329aba00" -dependencies = [ - "hostname", - "quick-error", -] - -[[package]] -name = "rfc6979" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8dd2a808d456c4a54e300a23e9f5a67e122c3024119acbfd73e3bf664491cb2" -dependencies = [ - "hmac", - "subtle", -] - -[[package]] -name = "ring" -version = "0.17.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c17fa4cb658e3583423e915b9f3acc01cceaee1860e33d59ebae66adc3a2dc0d" -dependencies = [ - "cc", - "cfg-if", - "getrandom", - "libc", - "spin 0.9.8", - "untrusted", - "windows-sys 0.52.0", -] - -[[package]] -name = "rkyv" -version = "0.7.44" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5cba464629b3394fc4dbc6f940ff8f5b4ff5c7aef40f29166fd4ad12acbc99c0" -dependencies = [ - "bitvec", - "bytecheck", - "bytes", - "hashbrown 0.12.3", - "ptr_meta", - "rend", - "rkyv_derive", - "seahash", - "tinyvec", - "uuid", -] - -[[package]] -name = "rkyv_derive" -version = "0.7.44" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a7dddfff8de25e6f62b9d64e6e432bf1c6736c57d20323e15ee10435fbda7c65" -dependencies = [ - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "rle-decode-fast" -version = "1.0.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3582f63211428f83597b51b2ddb88e2a91a9d52d12831f9d08f5e624e8977422" - -[[package]] -name = "rlp" -version = "0.5.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bb919243f34364b6bd2fc10ef797edbfa75f33c252e7998527479c6d6b47e1ec" -dependencies = [ - "bytes", - "rustc-hex", -] - -[[package]] -name = "rmp" -version = "0.8.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7f9860a6cc38ed1da53456442089b4dfa35e7cedaa326df63017af88385e6b20" -dependencies = [ - "byteorder", - "num-traits", - "paste", -] - -[[package]] -name = "rmp-serde" -version = "1.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bffea85eea980d8a74453e5d02a8d93028f3c34725de143085a844ebe953258a" -dependencies = [ - "byteorder", - "rmp", - "serde", -] - -[[package]] -name = "roaring" -version = "0.10.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a1c77081a55300e016cb86f2864415b7518741879db925b8d488a0ee0d2da6bf" -dependencies = [ - "bytemuck", - "byteorder", -] - -[[package]] -name = "robust" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cbf4a6aa5f6d6888f39e980649f3ad6b666acdce1d78e95b8a2cb076e687ae30" - -[[package]] -name = "rsa" -version = "0.9.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5d0e5124fcb30e76a7e79bfee683a2746db83784b86289f6251b54b7950a0dfc" -dependencies = [ - "const-oid", - "digest 0.10.7", - "num-bigint-dig", - "num-integer", - "num-traits", - "pkcs1", - "pkcs8", - "rand_core", - "signature", - "spki", - "subtle", - "zeroize", -] - -[[package]] -name = "rstar" -version = "0.11.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "73111312eb7a2287d229f06c00ff35b51ddee180f017ab6dec1f69d62ac098d6" -dependencies = [ - "heapless", - "num-traits", - "smallvec", -] - -[[package]] -name = "rusqlite" -version = "0.29.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "549b9d036d571d42e6e85d1c1425e2ac83491075078ca9a15be021c56b1641f2" -dependencies = [ - "bitflags 2.5.0", - "fallible-iterator", - "fallible-streaming-iterator", - "hashlink", - "libsqlite3-sys", - "smallvec", -] - -[[package]] -name = "rust_decimal" -version = "1.34.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b39449a79f45e8da28c57c341891b69a183044b29518bb8f86dbac9df60bb7df" -dependencies = [ - "arbitrary", - "arrayvec", - "borsh", - "bytes", - "num-traits", - "postgres", - "rand", - "rkyv", - "serde", - "serde_json", -] - -[[package]] -name = "rustc-demangle" -version = "0.1.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d626bb9dae77e28219937af045c257c28bfd3f69333c512553507f5f9798cb76" - -[[package]] -name = "rustc-hash" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "08d43f7aa6b08d49f382cde6a7982047c3426db949b1424bc4b7ec9ae12c6ce2" - -[[package]] -name = "rustc-hex" -version = "2.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3e75f6a532d0fd9f7f13144f392b6ad56a32696bfcd9c78f797f16bbb6f072d6" - -[[package]] -name = "rustc_version" -version = "0.2.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "138e3e0acb6c9fb258b19b67cb8abd63c00679d2851805ea151465464fe9030a" -dependencies = [ - "semver 0.9.0", -] - -[[package]] -name = "rustc_version" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bfa0f585226d2e68097d4f95d113b15b83a82e819ab25717ec0590d9584ef366" -dependencies = [ - "semver 1.0.22", -] - -[[package]] -name = "rustc_version_runtime" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d31b7153270ebf48bf91c65ae5b0c00e749c4cfad505f66530ac74950249582f" -dependencies = [ - "rustc_version 0.2.3", - "semver 0.9.0", -] - -[[package]] -name = "rustix" -version = "0.38.32" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "65e04861e65f21776e67888bfbea442b3642beaa0138fdb1dd7a84a52dffdb89" -dependencies = [ - "bitflags 2.5.0", - "errno", - "libc", - "linux-raw-sys", - "windows-sys 0.52.0", -] - -[[package]] -name = "rustls" -version = "0.21.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f9d5a6813c0759e4609cd494e8e725babae6a2ca7b62a5536a13daaec6fcb7ba" -dependencies = [ - "log", - "ring", - "rustls-webpki 0.101.7", - "sct", -] - -[[package]] -name = "rustls" -version = "0.22.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e87c9956bd9807afa1f77e0f7594af32566e830e088a5576d27c5b6f30f49d41" -dependencies = [ - "log", - "ring", - "rustls-pki-types", - "rustls-webpki 0.102.2", - "subtle", - "zeroize", -] - -[[package]] -name = "rustls-native-certs" -version = "0.6.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a9aace74cb666635c918e9c12bc0d348266037aa8eb599b5cba565709a8dff00" -dependencies = [ - "openssl-probe", - "rustls-pemfile 1.0.4", - "schannel", - "security-framework", -] - -[[package]] -name = "rustls-native-certs" -version = "0.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8f1fb85efa936c42c6d5fc28d2629bb51e4b2f4b8a5211e297d599cc5a093792" -dependencies = [ - "openssl-probe", - "rustls-pemfile 2.1.1", - "rustls-pki-types", - "schannel", - "security-framework", -] - -[[package]] -name = "rustls-pemfile" -version = "1.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1c74cae0a4cf6ccbbf5f359f08efdf8ee7e1dc532573bf0db71968cb56b1448c" -dependencies = [ - "base64 0.21.7", -] - -[[package]] -name = "rustls-pemfile" -version = "2.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f48172685e6ff52a556baa527774f61fcaa884f59daf3375c62a3f1cd2549dab" -dependencies = [ - "base64 0.21.7", - "rustls-pki-types", -] - -[[package]] -name = "rustls-pki-types" -version = "1.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5ede67b28608b4c60685c7d54122d4400d90f62b40caee7700e700380a390fa8" - -[[package]] -name = "rustls-tokio-stream" -version = "0.2.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ded7a36e8ac05b8ada77a84c5ceec95361942ee9dedb60a82f93f788a791aae8" -dependencies = [ - "futures", - "rustls 0.21.10", - "socket2 0.5.6", - "tokio", -] - -[[package]] -name = "rustls-webpki" -version = "0.101.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b6275d1ee7a1cd780b64aca7726599a1dbc893b1e64144529e55c3c2f745765" -dependencies = [ - "ring", - "untrusted", -] - -[[package]] -name = "rustls-webpki" -version = "0.102.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "faaa0a62740bedb9b2ef5afa303da42764c012f743917351dc9a237ea1663610" -dependencies = [ - "ring", - "rustls-pki-types", - "untrusted", -] - -[[package]] -name = "rustversion" -version = "1.0.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7ffc183a10b4478d04cbbbfc96d0873219d962dd5accaff2ffbd4ceb7df837f4" - -[[package]] -name = "rusty-fork" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cb3dcc6e454c328bb824492db107ab7c0ae8fcffe4ad210136ef014458c1bc4f" -dependencies = [ - "fnv", - "quick-error", - "tempfile", - "wait-timeout", -] - -[[package]] -name = "rustyline" -version = "13.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "02a2d683a4ac90aeef5b1013933f6d977bd37d51ff3f4dad829d4931a7e6be86" -dependencies = [ - "bitflags 2.5.0", - "cfg-if", - "clipboard-win", - "fd-lock", - "home", - "libc", - "log", - "memchr", - "nix", - "radix_trie", - "unicode-segmentation", - "unicode-width", - "utf8parse", - "winapi", -] - -[[package]] -name = "rustyline-derive" -version = "0.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e5af959c8bf6af1aff6d2b463a57f71aae53d1332da58419e30ad8dc7011d951" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "ryu" -version = "1.0.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e86697c916019a8588c99b5fac3cead74ec0b4b819707a682fd4d23fa0ce1ba1" - -[[package]] -name = "ryu-js" -version = "1.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ad97d4ce1560a5e27cec89519dc8300d1aa6035b099821261c651486a19e44d5" - -[[package]] -name = "same-file" -version = "1.0.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "93fc1dc3aaa9bfed95e02e6eadabb4baf7e3078b0bd1b4d7b6b0b68378900502" -dependencies = [ - "winapi-util", -] - -[[package]] -name = "saturating" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ece8e78b2f38ec51c51f5d475df0a7187ba5111b2a28bdc761ee05b075d40a71" - -[[package]] -name = "schannel" -version = "0.1.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fbc91545643bcf3a0bbb6569265615222618bdf33ce4ffbbd13c4bbd4c093534" -dependencies = [ - "windows-sys 0.52.0", -] - -[[package]] -name = "schema_registry_converter" -version = "4.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3e5a442d604e325eddde3c8108212a4c48463e73daaafa2861a2bc8977414b9" -dependencies = [ - "apache-avro", - "byteorder", - "dashmap 5.5.3", - "futures", - "reqwest", - "serde", - "serde_json", -] - -[[package]] -name = "schemars" -version = "0.8.16" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "45a28f4c49489add4ce10783f7911893516f15afe45d015608d41faca6bc4d29" -dependencies = [ - "dyn-clone", - "schemars_derive", - "serde", - "serde_json", -] - -[[package]] -name = "schemars_derive" -version = "0.8.16" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c767fd6fa65d9ccf9cf026122c1b555f2ef9a4f0cea69da4d7dbc3e258d30967" -dependencies = [ - "proc-macro2", - "quote", - "serde_derive_internals", - "syn 1.0.109", -] - -[[package]] -name = "scoped-tls" -version = "1.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e1cf6437eb19a8f4a6cc0f7dca544973b0b78843adbfeb3683d1a94a0024a294" - -[[package]] -name = "scopeguard" -version = "1.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49" - -[[package]] -name = "scroll" -version = "0.11.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "04c565b551bafbef4157586fa379538366e4385d42082f255bfd96e4fe8519da" - -[[package]] -name = "sct" -version = "0.7.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "da046153aa2352493d6cb7da4b6e5c0c057d8a1d0a9aa8560baffdd945acd414" -dependencies = [ - "ring", - "untrusted", -] - -[[package]] -name = "seahash" -version = "4.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1c107b6f4780854c8b126e228ea8869f4d7b71260f962fefb57b996b8959ba6b" - -[[package]] -name = "sec1" -version = "0.7.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3e97a565f76233a6003f9f5c54be1d9c5bdfa3eccfb189469f11ec4901c47dc" -dependencies = [ - "base16ct", - "der", - "generic-array", - "pkcs8", - "subtle", - "zeroize", -] - -[[package]] -name = "secp256k1" -version = "0.27.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "25996b82292a7a57ed3508f052cfff8640d38d32018784acd714758b43da9c8f" -dependencies = [ - "secp256k1-sys", -] - -[[package]] -name = "secp256k1-sys" -version = "0.8.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "70a129b9e9efbfb223753b9163c4ab3b13cff7fd9c7f010fbac25ab4099fa07e" -dependencies = [ - "cc", -] - -[[package]] -name = "security-framework" -version = "2.9.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "05b64fb303737d99b81884b2c63433e9ae28abebe5eb5045dcdd175dc2ecf4de" -dependencies = [ - "bitflags 1.3.2", - "core-foundation", - "core-foundation-sys", - "libc", - "security-framework-sys", -] - -[[package]] -name = "security-framework-sys" -version = "2.9.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e932934257d3b408ed8f30db49d85ea163bfe74961f017f405b025af298f0c7a" -dependencies = [ - "core-foundation-sys", - "libc", -] - -[[package]] -name = "semver" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1d7eb9ef2c18661902cc47e535f9bc51b78acd254da71d375c2f6720d9a40403" -dependencies = [ - "semver-parser", -] - -[[package]] -name = "semver" -version = "1.0.22" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92d43fe69e652f3df9bdc2b85b2854a0825b86e4fb76bc44d945137d053639ca" - -[[package]] -name = "semver-parser" -version = "0.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "388a1df253eca08550bef6c72392cfe7c30914bf41df5269b68cbd6ff8f570a3" - -[[package]] -name = "seq-macro" -version = "0.3.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a3f0bf26fd526d2a95683cd0f87bf103b8539e2ca1ef48ce002d67aad59aa0b4" - -[[package]] -name = "serde" -version = "1.0.197" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3fb1c873e1b9b056a4dc4c0c198b24c3ffa059243875552b2bd0933b1aee4ce2" -dependencies = [ - "serde_derive", -] - -[[package]] -name = "serde-value" -version = "0.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f3a1a3341211875ef120e117ea7fd5228530ae7e7036a779fdc9117be6b3282c" -dependencies = [ - "ordered-float 2.10.1", - "serde", -] - -[[package]] -name = "serde_bytes" -version = "0.11.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b8497c313fd43ab992087548117643f6fcd935cbf36f176ffda0aacf9591734" -dependencies = [ - "serde", -] - -[[package]] -name = "serde_derive" -version = "1.0.197" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7eb0b34b42edc17f6b7cac84a52a1c5f0e1bb2227e997ca9011ea3dd34e8610b" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "serde_derive_internals" -version = "0.26.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "85bf8229e7920a9f636479437026331ce11aa132b4dde37d121944a44d6e5f3c" -dependencies = [ - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "serde_json" -version = "1.0.114" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c5f09b1bd632ef549eaa9f60a1f8de742bdbc698e6cee2095fc84dde5f549ae0" -dependencies = [ - "indexmap 2.2.5", - "itoa", - "ryu", - "serde", -] - -[[package]] -name = "serde_urlencoded" -version = "0.7.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3491c14715ca2294c4d6a88f15e84739788c1d030eed8c110436aafdaa2f3fd" -dependencies = [ - "form_urlencoded", - "itoa", - "ryu", - "serde", -] - -[[package]] -name = "serde_v8" -version = "0.179.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "80ed6b8604315921ba50f2a872b89b93327aa53a1219d11304ee29fb625344bc" -dependencies = [ - "bytes", - "num-bigint", - "serde", - "smallvec", - "thiserror", - "v8", -] - -[[package]] -name = "serde_with" -version = "1.14.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "678b5a069e50bf00ecd22d0cd8ddf7c236f68581b03db652061ed5eb13a312ff" -dependencies = [ - "serde", - "serde_with_macros", -] - -[[package]] -name = "serde_with_macros" -version = "1.5.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e182d6ec6f05393cc0e5ed1bf81ad6db3a8feedf8ee515ecdd369809bcce8082" -dependencies = [ - "darling 0.13.4", - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "serde_yaml" -version = "0.9.33" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a0623d197252096520c6f2a5e1171ee436e5af99a5d7caa2891e55e61950e6d9" -dependencies = [ - "indexmap 2.2.5", - "itoa", - "ryu", - "serde", - "unsafe-libyaml", -] - -[[package]] -name = "serial_test" -version = "1.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "538c30747ae860d6fb88330addbbd3e0ddbe46d662d032855596d8a8ca260611" -dependencies = [ - "dashmap 5.5.3", - "futures", - "lazy_static", - "log", - "parking_lot", - "serial_test_derive 1.0.0", -] - -[[package]] -name = "serial_test" -version = "2.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0e56dd856803e253c8f298af3f4d7eb0ae5e23a737252cd90bb4f3b435033b2d" -dependencies = [ - "dashmap 5.5.3", - "futures", - "lazy_static", - "log", - "parking_lot", - "serial_test_derive 2.0.0", -] - -[[package]] -name = "serial_test_derive" -version = "1.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "079a83df15f85d89a68d64ae1238f142f172b1fa915d0d76b26a7cba1b659a69" -dependencies = [ - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "serial_test_derive" -version = "2.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "91d129178576168c589c9ec973feedf7d3126c01ac2bf08795109aa35b69fb8f" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "sha-1" -version = "0.9.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "99cd6713db3cf16b6c84e06321e049a9b9f699826e16096d23bbcc44d15d51a6" -dependencies = [ - "block-buffer 0.9.0", - "cfg-if", - "cpufeatures", - "digest 0.9.0", - "opaque-debug", -] - -[[package]] -name = "sha-1" -version = "0.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "028f48d513f9678cda28f6e4064755b3fbb2af6acd672f2c209b62323f7aea0f" -dependencies = [ - "cfg-if", - "cpufeatures", - "digest 0.10.7", -] - -[[package]] -name = "sha1" -version = "0.10.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e3bf829a2d51ab4a5ddf1352d8470c140cadc8301b2ae1789db023f01cedd6ba" -dependencies = [ - "cfg-if", - "cpufeatures", - "digest 0.10.7", -] - -[[package]] -name = "sha2" -version = "0.10.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "793db75ad2bcafc3ffa7c68b215fee268f537982cd901d132f89c6343f3a3dc8" -dependencies = [ - "cfg-if", - "cpufeatures", - "digest 0.10.7", -] - -[[package]] -name = "sha3" -version = "0.10.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "75872d278a8f37ef87fa0ddbda7802605cb18344497949862c0d4dcb291eba60" -dependencies = [ - "digest 0.10.7", - "keccak", -] - -[[package]] -name = "sharded-slab" -version = "0.1.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f40ca3c46823713e0d4209592e8d6e826aa57e928f09752619fc696c499637f6" -dependencies = [ - "lazy_static", -] - -[[package]] -name = "shlex" -version = "1.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0fda2ff0d084019ba4d7c6f371c95d8fd75ce3524c3cb8fb653a3023f6323e64" - -[[package]] -name = "signal-hook-registry" -version = "1.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d8229b473baa5980ac72ef434c4415e70c4b5e71b423043adb4ba059f89c99a1" -dependencies = [ - "libc", -] - -[[package]] -name = "signature" -version = "2.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "77549399552de45a898a580c1b41d445bf730df867cc44e6c0233bbc4b8329de" -dependencies = [ - "digest 0.10.7", - "rand_core", -] - -[[package]] -name = "simd-abstraction" -version = "0.7.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9cadb29c57caadc51ff8346233b5cec1d240b68ce55cf1afc764818791876987" -dependencies = [ - "outref 0.1.0", -] - -[[package]] -name = "simdutf8" -version = "0.1.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f27f6278552951f1f2b8cf9da965d10969b2efdea95a6ec47987ab46edfe263a" - -[[package]] -name = "similar" -version = "2.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32fea41aca09ee824cc9724996433064c89f7777e60762749a4170a14abbfa21" - -[[package]] -name = "siphasher" -version = "0.3.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "38b58827f4464d87d377d175e90bf58eb00fd8716ff0a62f80356b5e61555d0d" - -[[package]] -name = "slab" -version = "0.4.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8f92a496fb766b417c996b9c5e57daf2f7ad3b0bebe1ccfca4856390e3d3bb67" -dependencies = [ - "autocfg", -] - -[[package]] -name = "smallvec" -version = "1.13.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3c5e1a9a646d36c3599cd173a41282daf47c44583ad367b8e6837255952e5c67" - -[[package]] -name = "smartstring" -version = "1.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3fb72c633efbaa2dd666986505016c32c3044395ceaf881518399d2f4127ee29" -dependencies = [ - "autocfg", - "static_assertions", - "version_check", -] - -[[package]] -name = "snafu" -version = "0.7.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e4de37ad025c587a29e8f3f5605c00f70b98715ef90b9061a815b9e59e9042d6" -dependencies = [ - "doc-comment", - "snafu-derive", -] - -[[package]] -name = "snafu-derive" -version = "0.7.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "990079665f075b699031e9c08fd3ab99be5029b96f3b78dc0709e8f77e4efebf" -dependencies = [ - "heck 0.4.1", - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "snap" -version = "1.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1b6b67fb9a61334225b5b790716f609cd58395f895b3fe8b328786812a40bc3b" - -[[package]] -name = "socket2" -version = "0.4.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9f7916fc008ca5542385b89a3d3ce689953c143e9304a9bf8beec1de48994c0d" -dependencies = [ - "libc", - "winapi", -] - -[[package]] -name = "socket2" -version = "0.5.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "05ffd9c0a93b7543e062e759284fcf5f5e3b098501104bfbdde4d404db792871" -dependencies = [ - "libc", - "windows-sys 0.52.0", -] - -[[package]] -name = "soketto" -version = "0.7.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "41d1c5305e39e09653383c2c7244f2f78b3bcae37cf50c64cb4789c9f5096ec2" -dependencies = [ - "base64 0.13.1", - "bytes", - "futures", - "httparse", - "log", - "rand", - "sha-1 0.9.8", -] - -[[package]] -name = "sourcemap" -version = "6.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e4cbf65ca7dc576cf50e21f8d0712d96d4fcfd797389744b7b222a85cdf5bd90" -dependencies = [ - "data-encoding", - "debugid", - "if_chain", - "rustc_version 0.2.3", - "serde", - "serde_json", - "unicode-id", - "url", -] - -[[package]] -name = "sourcemap" -version = "7.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e7768edd06c02535e0d50653968f46e1e0d3aa54742190d35dd9466f59de9c71" -dependencies = [ - "base64-simd 0.7.0", - "data-encoding", - "debugid", - "if_chain", - "rustc_version 0.2.3", - "serde", - "serde_json", - "unicode-id-start", - "url", -] - -[[package]] -name = "spin" -version = "0.5.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6e63cff320ae2c57904679ba7cb63280a3dc4613885beafb148ee7bf9aa9042d" - -[[package]] -name = "spin" -version = "0.9.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6980e8d7511241f8acf4aebddbb1ff938df5eebe98691418c4468d0b72a96a67" -dependencies = [ - "lock_api", -] - -[[package]] -name = "spki" -version = "0.7.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d91ed6c858b01f942cd56b37a94b3e0a1798290327d1236e4d9cf4eaca44d29d" -dependencies = [ - "base64ct", - "der", -] - -[[package]] -name = "sqllogictest" -version = "0.17.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "78ea1056caa9180e7e5727eed1a377d96c9f4615303fa82d2f4c202c64736dee" -dependencies = [ - "async-trait", - "educe", - "fs-err", - "futures", - "glob", - "humantime", - "itertools 0.11.0", - "libtest-mimic", - "md-5", - "owo-colors", - "regex", - "similar", - "subst", - "tempfile", - "thiserror", - "tracing", -] - -[[package]] -name = "sqlparser" -version = "0.35.0" -source = "git+https://github.com/getdozer/sqlparser-rs.git#3dd4e9f14a9631c9707c40d7e497ffe0558a88cd" -dependencies = [ - "bigdecimal 0.3.1", - "log", -] - -[[package]] -name = "sqlparser" -version = "0.41.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5cc2c25a6c66789625ef164b4c7d2e548d627902280c13710d33da8222169964" -dependencies = [ - "log", - "sqlparser_derive", -] - -[[package]] -name = "sqlparser_derive" -version = "0.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "01b2e185515564f15375f593fb966b5718bc624ba77fe49fa4616ad619690554" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "stable_deref_trait" -version = "1.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a8f112729512f8e442d81f95a8a7ddf2b7c6b8a1a6f509a95864142b30cab2d3" - -[[package]] -name = "stacker" -version = "0.1.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c886bd4480155fd3ef527d45e9ac8dd7118a898a46530b7b94c3e21866259fce" -dependencies = [ - "cc", - "cfg-if", - "libc", - "psm", - "winapi", -] - -[[package]] -name = "static_assertions" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a2eb9349b6444b326872e140eb1cf5e7c522154d69e7a0ffb0fb81c06b37543f" - -[[package]] -name = "string_enum" -version = "0.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1b650ea2087d32854a0f20b837fc56ec987a1cb4f758c9757e1171ee9812da63" -dependencies = [ - "proc-macro2", - "quote", - "swc_macros_common", - "syn 2.0.53", -] - -[[package]] -name = "stringprep" -version = "0.1.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bb41d74e231a107a1b4ee36bd1214b11285b77768d2e3824aedafa988fd36ee6" -dependencies = [ - "finl_unicode", - "unicode-bidi", - "unicode-normalization", -] - -[[package]] -name = "strsim" -version = "0.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "73473c0e59e6d5812c5dfe2a064a6444949f089e20eec9a2e5506596494e4623" - -[[package]] -name = "strsim" -version = "0.11.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5ee073c9e4cd00e28217186dbe12796d692868f432bf2e97ee73bed0c56dfa01" - -[[package]] -name = "strum" -version = "0.25.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "290d54ea6f91c969195bdbcd7442c8c2a2ba87da8bf60a7ee86a235d4bc1e125" -dependencies = [ - "strum_macros", -] - -[[package]] -name = "strum_macros" -version = "0.25.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "23dc1fa9ac9c169a78ba62f0b841814b7abae11bdd047b9c58f893439e309ea0" -dependencies = [ - "heck 0.4.1", - "proc-macro2", - "quote", - "rustversion", - "syn 2.0.53", -] - -[[package]] -name = "subprocess" -version = "0.2.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c2e86926081dda636c546d8c5e641661049d7562a68f5488be4a1f7f66f6086" -dependencies = [ - "libc", - "winapi", -] - -[[package]] -name = "subst" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ca1318e5d6716d6541696727c88d9b8dfc8cfe6afd6908e186546fd4af7f5b98" -dependencies = [ - "memchr", - "unicode-width", -] - -[[package]] -name = "subtle" -version = "2.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "81cdd64d312baedb58e21336b31bc043b77e01cc99033ce76ef539f78e965ebc" - -[[package]] -name = "swc_atoms" -version = "0.6.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7d538eaaa6f085161d088a04cf0a3a5a52c5a7f2b3bd9b83f73f058b0ed357c0" -dependencies = [ - "hstr", - "once_cell", - "rustc-hash", - "serde", -] - -[[package]] -name = "swc_common" -version = "0.33.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9b3ae36feceded27f0178dc9dabb49399830847ffb7f866af01798844de8f973" -dependencies = [ - "ast_node", - "better_scoped_tls", - "cfg-if", - "either", - "from_variant", - "new_debug_unreachable", - "num-bigint", - "once_cell", - "rustc-hash", - "serde", - "siphasher", - "sourcemap 6.4.1", - "swc_atoms", - "swc_eq_ignore_macros", - "swc_visit", - "tracing", - "unicode-width", - "url", -] - -[[package]] -name = "swc_config" -version = "0.1.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "112884e66b60e614c0f416138b91b8b82b7fea6ed0ecc5e26bad4726c57a6c99" -dependencies = [ - "indexmap 2.2.5", - "serde", - "serde_json", - "swc_config_macro", -] - -[[package]] -name = "swc_config_macro" -version = "0.1.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b2574f75082322a27d990116cd2a24de52945fc94172b24ca0b3e9e2a6ceb6b" -dependencies = [ - "proc-macro2", - "quote", - "swc_macros_common", - "syn 2.0.53", -] - -[[package]] -name = "swc_ecma_ast" -version = "0.110.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "79401a45da704f4fb2552c5bf86ee2198e8636b121cb81f8036848a300edd53b" -dependencies = [ - "bitflags 2.5.0", - "is-macro", - "num-bigint", - "phf", - "scoped-tls", - "serde", - "string_enum", - "swc_atoms", - "swc_common", - "unicode-id", -] - -[[package]] -name = "swc_ecma_codegen" -version = "0.146.54" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "99b61ca275e3663238b71c4b5da8e6fb745bde9989ef37d94984dfc81fc6d009" -dependencies = [ - "memchr", - "num-bigint", - "once_cell", - "rustc-hash", - "serde", - "sourcemap 6.4.1", - "swc_atoms", - "swc_common", - "swc_ecma_ast", - "swc_ecma_codegen_macros", - "tracing", -] - -[[package]] -name = "swc_ecma_codegen_macros" -version = "0.7.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "394b8239424b339a12012ceb18726ed0244fce6bf6345053cb9320b2791dcaa5" -dependencies = [ - "proc-macro2", - "quote", - "swc_macros_common", - "syn 2.0.53", -] - -[[package]] -name = "swc_ecma_loader" -version = "0.45.13" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c5713ab3429530c10bdf167170ebbde75b046c8003558459e4de5aaec62ce0f1" -dependencies = [ - "anyhow", - "pathdiff", - "serde", - "swc_common", - "tracing", -] - -[[package]] -name = "swc_ecma_parser" -version = "0.141.37" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c4d17401dd95048a6a62b777d533c0999dabdd531ef9d667e22f8ae2a2a0d294" -dependencies = [ - "either", - "new_debug_unreachable", - "num-bigint", - "num-traits", - "phf", - "serde", - "smallvec", - "smartstring", - "stacker", - "swc_atoms", - "swc_common", - "swc_ecma_ast", - "tracing", - "typed-arena", -] - -[[package]] -name = "swc_ecma_transforms_base" -version = "0.135.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6d4ab26ec124b03e47f54d4daade8e9a9dcd66d3a4ca3cd47045f138d267a60e" -dependencies = [ - "better_scoped_tls", - "bitflags 2.5.0", - "indexmap 2.2.5", - "once_cell", - "phf", - "rustc-hash", - "serde", - "smallvec", - "swc_atoms", - "swc_common", - "swc_ecma_ast", - "swc_ecma_parser", - "swc_ecma_utils", - "swc_ecma_visit", - "tracing", -] - -[[package]] -name = "swc_ecma_transforms_classes" -version = "0.124.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9fe4376c024fa04394cafb8faecafb4623722b92dbbe46532258cc0a6b569d9c" -dependencies = [ - "swc_atoms", - "swc_common", - "swc_ecma_ast", - "swc_ecma_transforms_base", - "swc_ecma_utils", - "swc_ecma_visit", -] - -[[package]] -name = "swc_ecma_transforms_macros" -version = "0.5.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "17e309b88f337da54ef7fe4c5b99c2c522927071f797ee6c9fb8b6bf2d100481" -dependencies = [ - "proc-macro2", - "quote", - "swc_macros_common", - "syn 2.0.53", -] - -[[package]] -name = "swc_ecma_transforms_proposal" -version = "0.169.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "86de99757fc31d8977f47c02a26e5c9a243cb63b03fe8aa8b36d79924b8fa29c" -dependencies = [ - "either", - "rustc-hash", - "serde", - "smallvec", - "swc_atoms", - "swc_common", - "swc_ecma_ast", - "swc_ecma_transforms_base", - "swc_ecma_transforms_classes", - "swc_ecma_transforms_macros", - "swc_ecma_utils", - "swc_ecma_visit", -] - -[[package]] -name = "swc_ecma_transforms_react" -version = "0.181.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9918e22caf1ea4a71085f5d818d6c0bf5c19d669cfb9d38f9fdc3da0496abdc7" -dependencies = [ - "base64 0.21.7", - "dashmap 5.5.3", - "indexmap 2.2.5", - "once_cell", - "serde", - "sha-1 0.10.0", - "string_enum", - "swc_atoms", - "swc_common", - "swc_config", - "swc_ecma_ast", - "swc_ecma_parser", - "swc_ecma_transforms_base", - "swc_ecma_transforms_macros", - "swc_ecma_utils", - "swc_ecma_visit", -] - -[[package]] -name = "swc_ecma_transforms_typescript" -version = "0.186.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e1d1495c969ffdc224384f1fb73646b9c1b170779f20fdb984518deb054aa522" -dependencies = [ - "ryu-js", - "serde", - "swc_atoms", - "swc_common", - "swc_ecma_ast", - "swc_ecma_transforms_base", - "swc_ecma_transforms_react", - "swc_ecma_utils", - "swc_ecma_visit", -] - -[[package]] -name = "swc_ecma_utils" -version = "0.125.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7cead1083e46b0f072a82938f16d366014468f7510350957765bb4d013496890" -dependencies = [ - "indexmap 2.2.5", - "num_cpus", - "once_cell", - "rustc-hash", - "swc_atoms", - "swc_common", - "swc_ecma_ast", - "swc_ecma_visit", - "tracing", - "unicode-id", -] - -[[package]] -name = "swc_ecma_visit" -version = "0.96.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a1d0100c383fb08b6f34911ab6f925950416a5d14404c1cd520d59fb8dfbb3bf" -dependencies = [ - "num-bigint", - "swc_atoms", - "swc_common", - "swc_ecma_ast", - "swc_visit", - "tracing", -] - -[[package]] -name = "swc_eq_ignore_macros" -version = "0.1.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "695a1d8b461033d32429b5befbf0ad4d7a2c4d6ba9cd5ba4e0645c615839e8e4" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "swc_macros_common" -version = "0.3.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "50176cfc1cbc8bb22f41c6fe9d1ec53fbe057001219b5954961b8ad0f336fce9" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "swc_visit" -version = "0.5.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b27078d8571abe23aa52ef608dd1df89096a37d867cf691cbb4f4c392322b7c9" -dependencies = [ - "either", - "swc_visit_macros", -] - -[[package]] -name = "swc_visit_macros" -version = "0.5.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fa8bb05975506741555ea4d10c3a3bdb0e2357cd58e1a4a4332b8ebb4b44c34d" -dependencies = [ - "Inflector", - "pmutil", - "proc-macro2", - "quote", - "swc_macros_common", - "syn 2.0.53", -] - -[[package]] -name = "syn" -version = "1.0.109" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "72b64191b275b66ffe2469e8af2c1cfe3bafa67b529ead792a6d0160888b4237" -dependencies = [ - "proc-macro2", - "quote", - "unicode-ident", -] - -[[package]] -name = "syn" -version = "2.0.53" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7383cd0e49fff4b6b90ca5670bfd3e9d6a733b3f90c686605aa7eec8c4996032" -dependencies = [ - "proc-macro2", - "quote", - "unicode-ident", -] - -[[package]] -name = "syn-mid" -version = "0.5.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fea305d57546cc8cd04feb14b62ec84bf17f50e3f7b12560d7bfa9265f39d9ed" -dependencies = [ - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "syn_derive" -version = "0.1.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1329189c02ff984e9736652b1631330da25eaa6bc639089ed4915d25446cbe7b" -dependencies = [ - "proc-macro-error 1.0.4", - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "sync_wrapper" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2047c6ded9c721764247e62cd3b03c09ffc529b2ba5b10ec482ae507a4a70160" - -[[package]] -name = "take_mut" -version = "0.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f764005d11ee5f36500a149ace24e00e3da98b0158b3e2d53a7495660d3f4d60" - -[[package]] -name = "tap" -version = "1.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "55937e1799185b12863d447f42597ed69d9928686b8d88a1df17376a097d8369" - -[[package]] -name = "tar" -version = "0.4.40" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b16afcea1f22891c49a00c751c7b63b2233284064f11a200fc624137c51e2ddb" -dependencies = [ - "filetime", - "libc", - "xattr", -] - -[[package]] -name = "target-lexicon" -version = "0.12.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e1fc403891a21bcfb7c37834ba66a547a8f402146eba7265b5a6d88059c9ff2f" - -[[package]] -name = "tempfile" -version = "3.10.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "85b77fafb263dd9d05cbeac119526425676db3784113aa9295c88498cbf8bff1" -dependencies = [ - "cfg-if", - "fastrand", - "rustix", - "windows-sys 0.52.0", -] - -[[package]] -name = "term" -version = "0.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c59df8ac95d96ff9bede18eb7300b0fda5e5d8d90960e76f8e14ae765eedbf1f" -dependencies = [ - "dirs-next", - "rustversion", - "winapi", -] - -[[package]] -name = "termcolor" -version = "1.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "06794f8f6c5c898b3275aebefa6b8a1cb24cd2c6c79397ab15774837a0bc5755" -dependencies = [ - "winapi-util", -] - -[[package]] -name = "text_lines" -version = "0.6.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7fd5828de7deaa782e1dd713006ae96b3bee32d3279b79eb67ecf8072c059bcf" -dependencies = [ - "serde", -] - -[[package]] -name = "thiserror" -version = "1.0.58" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "03468839009160513471e86a034bb2c5c0e4baae3b43f79ffc55c4a5427b3297" -dependencies = [ - "thiserror-impl", -] - -[[package]] -name = "thiserror-impl" -version = "1.0.58" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c61f3ba182994efc43764a46c018c347bc492c79f024e705f46567b418f6d4f7" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "thread_local" -version = "1.1.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b9ef9bad013ada3808854ceac7b46812a6465ba368859a37e2100283d2d719c" -dependencies = [ - "cfg-if", - "once_cell", -] - -[[package]] -name = "threadpool" -version = "1.8.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d050e60b33d41c19108b32cea32164033a9013fe3b46cbd4457559bfbf77afaa" -dependencies = [ - "num_cpus", -] - -[[package]] -name = "thrift" -version = "0.17.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e54bc85fc7faa8bc175c4bab5b92ba8d9a3ce893d0e9f42cc455c8ab16a9e09" -dependencies = [ - "byteorder", - "integer-encoding", - "ordered-float 2.10.1", -] - -[[package]] -name = "time" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c8248b6521bb14bc45b4067159b9b6ad792e2d6d754d6c41fb50e29fefe38749" -dependencies = [ - "deranged", - "itoa", - "num-conv", - "powerfmt", - "serde", - "time-core", - "time-macros", -] - -[[package]] -name = "time-core" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ef927ca75afb808a4d64dd374f00a2adf8d0fcff8e7b184af886c3c87ec4a3f3" - -[[package]] -name = "time-macros" -version = "0.2.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7ba3a3ef41e6672a2f0f001392bb5dcd3ff0a9992d618ca761a11c3121547774" -dependencies = [ - "num-conv", - "time-core", -] - -[[package]] -name = "tiny-keccak" -version = "2.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2c9d3793400a45f954c52e73d068316d76b6f4e36977e3fcebb13a2721e80237" -dependencies = [ - "crunchy", -] - -[[package]] -name = "tinytemplate" -version = "1.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "be4d6b5f19ff7664e8c98d03e2139cb510db9b0a60b55f8e8709b689d939b6bc" -dependencies = [ - "serde", - "serde_json", -] - -[[package]] -name = "tinyvec" -version = "1.6.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "87cc5ceb3875bb20c2890005a4e226a4651264a5c75edb2421b52861a0a0cb50" -dependencies = [ - "tinyvec_macros", -] - -[[package]] -name = "tinyvec_macros" -version = "0.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20" - -[[package]] -name = "tokio" -version = "1.36.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "61285f6515fa018fb2d1e46eb21223fff441ee8db5d0f1435e8ab4f5cdb80931" -dependencies = [ - "backtrace", - "bytes", - "libc", - "mio", - "num_cpus", - "parking_lot", - "pin-project-lite", - "signal-hook-registry", - "socket2 0.5.6", - "tokio-macros", - "tracing", - "windows-sys 0.48.0", -] - -[[package]] -name = "tokio-io-timeout" -version = "1.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "30b74022ada614a1b4834de765f9bb43877f910cc8ce4be40e89042c9223a8bf" -dependencies = [ - "pin-project-lite", - "tokio", -] - -[[package]] -name = "tokio-macros" -version = "2.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5b8a1e28f2deaa14e508979454cb3a223b10b938b45af148bc0986de36f1923b" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "tokio-native-tls" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bbae76ab933c85776efabc971569dd6119c580d8f5d448769dec1764bf796ef2" -dependencies = [ - "native-tls", - "tokio", -] - -[[package]] -name = "tokio-postgres" -version = "0.7.8" -source = "git+https://github.com/MaterializeInc/rust-postgres#b759caa33610403aa74b1cfdd37f45eb3100c9af" -dependencies = [ - "async-trait", - "byteorder", - "bytes", - "fallible-iterator", - "futures-channel", - "futures-util", - "log", - "parking_lot", - "percent-encoding", - "phf", - "pin-project-lite", - "postgres-protocol", - "postgres-types", - "socket2 0.5.6", - "tokio", - "tokio-util", -] - -[[package]] -name = "tokio-postgres-rustls" -version = "0.11.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ea13f22eda7127c827983bdaf0d7fff9df21c8817bab02815ac277a21143677" -dependencies = [ - "futures", - "ring", - "rustls 0.22.2", - "tokio", - "tokio-postgres", - "tokio-rustls 0.25.0", - "x509-certificate", -] - -[[package]] -name = "tokio-rustls" -version = "0.24.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c28327cf380ac148141087fbfb9de9d7bd4e84ab5d2c28fbc911d753de8a7081" -dependencies = [ - "rustls 0.21.10", - "tokio", -] - -[[package]] -name = "tokio-rustls" -version = "0.25.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "775e0c0f0adb3a2f22a00c4745d728b479985fc15ee7ca6a2608388c5569860f" -dependencies = [ - "rustls 0.22.2", - "rustls-pki-types", - "tokio", -] - -[[package]] -name = "tokio-socks" -version = "0.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "51165dfa029d2a65969413a6cc96f354b86b464498702f174a4efa13608fd8c0" -dependencies = [ - "either", - "futures-util", - "thiserror", - "tokio", -] - -[[package]] -name = "tokio-stream" -version = "0.1.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "267ac89e0bec6e691e5813911606935d77c476ff49024f98abcea3e7b15e37af" -dependencies = [ - "futures-core", - "pin-project-lite", - "tokio", -] - -[[package]] -name = "tokio-util" -version = "0.7.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5419f34732d9eb6ee4c3578b7989078579b7f039cbbb9ca2c4da015749371e15" -dependencies = [ - "bytes", - "futures-core", - "futures-io", - "futures-sink", - "pin-project-lite", - "tokio", - "tracing", -] - -[[package]] -name = "toml_datetime" -version = "0.6.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3550f4e9685620ac18a50ed434eb3aec30db8ba93b0287467bca5826ea25baf1" - -[[package]] -name = "toml_edit" -version = "0.19.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1b5bb770da30e5cbfde35a2d7b9b8a2c4b8ef89548a7a6aeab5c9a576e3e7421" -dependencies = [ - "indexmap 2.2.5", - "toml_datetime", - "winnow", -] - -[[package]] -name = "toml_edit" -version = "0.20.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "70f427fce4d84c72b5b732388bf4a9f4531b53f74e2887e3ecb2481f68f66d81" -dependencies = [ - "indexmap 2.2.5", - "toml_datetime", - "winnow", -] - -[[package]] -name = "toml_edit" -version = "0.21.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6a8534fd7f78b5405e860340ad6575217ce99f38d4d5c8f2442cb5ecb50090e1" -dependencies = [ - "indexmap 2.2.5", - "toml_datetime", - "winnow", -] - -[[package]] -name = "tonic" -version = "0.10.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d560933a0de61cf715926b9cac824d4c883c2c43142f787595e48280c40a1d0e" -dependencies = [ - "async-stream", - "async-trait", - "axum", - "base64 0.21.7", - "bytes", - "h2 0.3.26", - "http 0.2.12", - "http-body 0.4.6", - "hyper 0.14.28", - "hyper-timeout", - "percent-encoding", - "pin-project", - "prost", - "tokio", - "tokio-stream", - "tower", - "tower-layer", - "tower-service", - "tracing", -] - -[[package]] -name = "tonic" -version = "0.11.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "76c4eb7a4e9ef9d4763600161f12f5070b92a578e1b634db88a6887844c91a13" -dependencies = [ - "async-stream", - "async-trait", - "axum", - "base64 0.21.7", - "bytes", - "h2 0.3.26", - "http 0.2.12", - "http-body 0.4.6", - "hyper 0.14.28", - "hyper-timeout", - "percent-encoding", - "pin-project", - "prost", - "rustls-native-certs 0.7.0", - "rustls-pemfile 2.1.1", - "rustls-pki-types", - "tokio", - "tokio-rustls 0.25.0", - "tokio-stream", - "tower", - "tower-layer", - "tower-service", - "tracing", -] - -[[package]] -name = "tonic-build" -version = "0.10.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9d021fc044c18582b9a2408cd0dd05b1596e3ecdb5c4df822bb0183545683889" -dependencies = [ - "prettyplease", - "proc-macro2", - "prost-build", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "tonic-reflection" -version = "0.11.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "548c227bd5c0fae5925812c4ec6c66ffcfced23ea370cb823f4d18f0fc1cb6a7" -dependencies = [ - "prost", - "prost-types", - "tokio", - "tokio-stream", - "tonic 0.11.0", -] - -[[package]] -name = "tonic-web" -version = "0.11.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc3b0e1cedbf19fdfb78ef3d672cb9928e0a91a9cb4629cc0c916e8cff8aaaa1" -dependencies = [ - "base64 0.21.7", - "bytes", - "http 0.2.12", - "http-body 0.4.6", - "hyper 0.14.28", - "pin-project", - "tokio-stream", - "tonic 0.11.0", - "tower-http", - "tower-layer", - "tower-service", - "tracing", -] - -[[package]] -name = "tower" -version = "0.4.13" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b8fa9be0de6cf49e536ce1851f987bd21a43b771b09473c3549a6c853db37c1c" -dependencies = [ - "futures-core", - "futures-util", - "indexmap 1.9.3", - "pin-project", - "pin-project-lite", - "rand", - "slab", - "tokio", - "tokio-util", - "tower-layer", - "tower-service", - "tracing", -] - -[[package]] -name = "tower-http" -version = "0.4.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "61c5bb1d698276a2443e5ecfabc1008bf15a36c12e6a7176e7bf089ea9131140" -dependencies = [ - "async-compression", - "base64 0.21.7", - "bitflags 2.5.0", - "bytes", - "futures-core", - "futures-util", - "http 0.2.12", - "http-body 0.4.6", - "http-range-header", - "httpdate", - "iri-string", - "mime", - "mime_guess", - "percent-encoding", - "pin-project-lite", - "tokio", - "tokio-util", - "tower", - "tower-layer", - "tower-service", - "tracing", - "uuid", -] - -[[package]] -name = "tower-layer" -version = "0.3.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c20c8dbed6283a09604c3e69b4b7eeb54e298b8a600d4d5ecb5ad39de609f1d0" - -[[package]] -name = "tower-service" -version = "0.3.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b6bc1c9ce2b5135ac7f93c72918fc37feb872bdc6a5533a8b85eb4b86bfdae52" - -[[package]] -name = "tracing" -version = "0.1.40" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c3523ab5a71916ccf420eebdf5521fcef02141234bbc0b8a49f2fdc4544364ef" -dependencies = [ - "log", - "pin-project-lite", - "tracing-attributes", - "tracing-core", -] - -[[package]] -name = "tracing-attributes" -version = "0.1.27" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "34704c8d6ebcbc939824180af020566b01a7c01f80641264eba0999f6c2b6be7" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "tracing-core" -version = "0.1.32" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c06d3da6113f116aaee68e4d601191614c9053067f9ab7f6edbcb161237daa54" -dependencies = [ - "once_cell", - "valuable", -] - -[[package]] -name = "tracing-log" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ee855f1f400bd0e5c02d150ae5de3840039a3f54b025156404e34c23c03f47c3" -dependencies = [ - "log", - "once_cell", - "tracing-core", -] - -[[package]] -name = "tracing-opentelemetry" -version = "0.23.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a9be14ba1bbe4ab79e9229f7f89fab8d120b865859f10527f31c033e599d2284" -dependencies = [ - "js-sys", - "once_cell", - "opentelemetry", - "opentelemetry_sdk", - "smallvec", - "tracing", - "tracing-core", - "tracing-log", - "tracing-subscriber", - "web-time", -] - -[[package]] -name = "tracing-subscriber" -version = "0.3.18" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ad0f048c97dbd9faa9b7df56362b8ebcaa52adb06b498c050d2f4e32f90a7a8b" -dependencies = [ - "matchers", - "nu-ansi-term", - "once_cell", - "regex", - "sharded-slab", - "smallvec", - "thread_local", - "tracing", - "tracing-core", - "tracing-log", -] - -[[package]] -name = "trust-dns-proto" -version = "0.21.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9c31f240f59877c3d4bb3b3ea0ec5a6a0cff07323580ff8c7a605cd7d08b255d" -dependencies = [ - "async-trait", - "cfg-if", - "data-encoding", - "enum-as-inner 0.4.0", - "futures-channel", - "futures-io", - "futures-util", - "idna 0.2.3", - "ipnet", - "lazy_static", - "log", - "rand", - "smallvec", - "thiserror", - "tinyvec", - "tokio", - "url", -] - -[[package]] -name = "trust-dns-proto" -version = "0.22.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4f7f83d1e4a0e4358ac54c5c3681e5d7da5efc5a7a632c90bb6d6669ddd9bc26" -dependencies = [ - "async-trait", - "cfg-if", - "data-encoding", - "enum-as-inner 0.5.1", - "futures-channel", - "futures-io", - "futures-util", - "idna 0.2.3", - "ipnet", - "lazy_static", - "rand", - "serde", - "smallvec", - "thiserror", - "tinyvec", - "tokio", - "tracing", - "url", -] - -[[package]] -name = "trust-dns-resolver" -version = "0.21.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e4ba72c2ea84515690c9fcef4c6c660bb9df3036ed1051686de84605b74fd558" -dependencies = [ - "cfg-if", - "futures-util", - "ipconfig", - "lazy_static", - "log", - "lru-cache", - "parking_lot", - "resolv-conf", - "smallvec", - "thiserror", - "tokio", - "trust-dns-proto 0.21.2", -] - -[[package]] -name = "trust-dns-resolver" -version = "0.22.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "aff21aa4dcefb0a1afbfac26deb0adc93888c7d295fb63ab273ef276ba2b7cfe" -dependencies = [ - "cfg-if", - "futures-util", - "ipconfig", - "lazy_static", - "lru-cache", - "parking_lot", - "resolv-conf", - "serde", - "smallvec", - "thiserror", - "tokio", - "tracing", - "trust-dns-proto 0.22.0", -] - -[[package]] -name = "try-lock" -version = "0.2.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" - -[[package]] -name = "twox-hash" -version = "1.6.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "97fee6b57c6a41524a810daee9286c02d7752c4253064d0b05472833a438f675" -dependencies = [ - "cfg-if", - "rand", - "static_assertions", -] - -[[package]] -name = "typed-arena" -version = "2.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6af6ae20167a9ece4bcb41af5b80f8a1f1df981f6391189ce00fd257af04126a" - -[[package]] -name = "typed-builder" -version = "0.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "89851716b67b937e393b3daa8423e67ddfc4bbbf1654bcf05488e95e0828db0c" -dependencies = [ - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "typed-builder" -version = "0.16.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "34085c17941e36627a879208083e25d357243812c30e7d7387c3b954f30ade16" -dependencies = [ - "typed-builder-macro", -] - -[[package]] -name = "typed-builder-macro" -version = "0.16.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f03ca4cb38206e2bef0700092660bb74d696f808514dae47fa1467cbfe26e96e" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "typenum" -version = "1.17.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "42ff0bf0c66b8238c6f3b578df37d0b7848e55df8577b3f74f92a69acceeb825" - -[[package]] -name = "ucd-trie" -version = "0.1.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ed646292ffc8188ef8ea4d1e0e0150fb15a5c2e12ad9b8fc191ae7a8a7f3c4b9" - -[[package]] -name = "uint" -version = "0.9.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "76f64bba2c53b04fcab63c01a7d7427eadc821e3bc48c34dc9ba29c501164b52" -dependencies = [ - "byteorder", - "crunchy", - "hex", - "static_assertions", -] - -[[package]] -name = "unarray" -version = "0.1.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eaea85b334db583fe3274d12b4cd1880032beab409c0d774be044d4480ab9a94" - -[[package]] -name = "unic-char-property" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a8c57a407d9b6fa02b4795eb81c5b6652060a15a7903ea981f3d723e6c0be221" -dependencies = [ - "unic-char-range", -] - -[[package]] -name = "unic-char-range" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0398022d5f700414f6b899e10b8348231abf9173fa93144cbc1a43b9793c1fbc" - -[[package]] -name = "unic-common" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "80d7ff825a6a654ee85a63e80f92f054f904f21e7d12da4e22f9834a4aaa35bc" - -[[package]] -name = "unic-ucd-ident" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e230a37c0381caa9219d67cf063aa3a375ffed5bf541a452db16e744bdab6987" -dependencies = [ - "unic-char-property", - "unic-char-range", - "unic-ucd-version", -] - -[[package]] -name = "unic-ucd-version" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "96bd2f2237fe450fcd0a1d2f5f4e91711124f7857ba2e964247776ebeeb7b0c4" -dependencies = [ - "unic-common", -] - -[[package]] -name = "unicase" -version = "2.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f7d2d4dafb69621809a81864c9c1b864479e1235c0dd4e199924b9742439ed89" -dependencies = [ - "version_check", -] - -[[package]] -name = "unicode-bidi" -version = "0.3.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "08f95100a766bf4f8f28f90d77e0a5461bbdb219042e7679bebe79004fed8d75" - -[[package]] -name = "unicode-id" -version = "0.3.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b1b6def86329695390197b82c1e244a54a131ceb66c996f2088a3876e2ae083f" - -[[package]] -name = "unicode-id-start" -version = "1.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b8f73150333cb58412db36f2aca8f2875b013049705cc77b94ded70a1ab1f5da" - -[[package]] -name = "unicode-ident" -version = "1.0.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3354b9ac3fae1ff6755cb6db53683adb661634f67557942dea4facebec0fee4b" - -[[package]] -name = "unicode-normalization" -version = "0.1.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a56d1686db2308d901306f92a263857ef59ea39678a5458e7cb17f01415101f5" -dependencies = [ - "tinyvec", -] - -[[package]] -name = "unicode-segmentation" -version = "1.11.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d4c87d22b6e3f4a18d4d40ef354e97c90fcb14dd91d7dc0aa9d8a1172ebf7202" - -[[package]] -name = "unicode-width" -version = "0.1.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e51733f11c9c4f72aa0c160008246859e340b00807569a0da0e7a1079b27ba85" - -[[package]] -name = "unindent" -version = "0.1.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e1766d682d402817b5ac4490b3c3002d91dfa0d22812f341609f97b08757359c" - -[[package]] -name = "universal-hash" -version = "0.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fc1de2c688dc15305988b563c3854064043356019f97a4b46276fe734c4f07ea" -dependencies = [ - "crypto-common", - "subtle", -] - -[[package]] -name = "unsafe-libyaml" -version = "0.2.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "673aac59facbab8a9007c7f6108d11f63b603f7cabff99fabf650fea5c32b861" - -[[package]] -name = "untrusted" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8ecb6da28b8a351d773b68d5825ac39017e680750f980f3a1a85cd8dd28a47c1" - -[[package]] -name = "ureq" -version = "2.9.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "11f214ce18d8b2cbe84ed3aa6486ed3f5b285cf8d8fbdbce9f3f767a724adc35" -dependencies = [ - "base64 0.21.7", - "log", - "once_cell", - "rustls 0.22.2", - "rustls-pki-types", - "rustls-webpki 0.102.2", - "url", - "webpki-roots 0.26.1", -] - -[[package]] -name = "url" -version = "2.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "31e6302e3bb753d46e83516cae55ae196fc0c309407cf11ab35cc51a4c2a4633" -dependencies = [ - "form_urlencoded", - "idna 0.5.0", - "percent-encoding", - "serde", -] - -[[package]] -name = "urlencoding" -version = "2.1.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "daf8dba3b7eb870caf1ddeed7bc9d2a049f3cfdfae7cb521b087cc33ae4c49da" - -[[package]] -name = "urlpattern" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f9bd5ff03aea02fa45b13a7980151fe45009af1980ba69f651ec367121a31609" -dependencies = [ - "derive_more", - "regex", - "serde", - "unic-ucd-ident", - "url", -] - -[[package]] -name = "utf-8" -version = "0.7.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "09cc8ee72d2a9becf2f2febe0205bbed8fc6615b7cb429ad062dc7b7ddd036a9" - -[[package]] -name = "utf8parse" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "711b9620af191e0cdc7468a8d14e709c3dcdb115b36f838e601583af800a370a" - -[[package]] -name = "uuid" -version = "1.8.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a183cf7feeba97b4dd1c0d46788634f6221d87fa961b305bed08c851829efcc0" -dependencies = [ - "atomic", - "getrandom", - "rand", - "serde", -] - -[[package]] -name = "v8" -version = "0.85.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ec8e09551fa5c3500b47f08912b4a39e07ae20a3874051941408fbd52e3e5190" -dependencies = [ - "bitflags 2.5.0", - "fslock", - "once_cell", - "which 5.0.0", -] - -[[package]] -name = "v_htmlescape" -version = "0.15.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4e8257fbc510f0a46eb602c10215901938b5c2a7d5e70fc11483b1d3c9b5b18c" - -[[package]] -name = "valuable" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "830b7e5d4d90034032940e4ace0d9a9a057e7a45cd94e6c007832e39edb82f6d" - -[[package]] -name = "vcpkg" -version = "0.2.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426" - -[[package]] -name = "version_check" -version = "0.9.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "49874b5167b65d7193b8aba1567f5c7d93d001cafc34600cee003eda787e483f" - -[[package]] -name = "virtue" -version = "0.0.13" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9dcc60c0624df774c82a0ef104151231d37da4962957d691c011c852b2473314" - -[[package]] -name = "vsimd" -version = "0.8.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5c3082ca00d5a5ef149bb8b555a72ae84c9c59f7250f013ac822ac2e49b19c64" - -[[package]] -name = "vswhom" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "be979b7f07507105799e854203b470ff7c78a1639e330a58f183b5fea574608b" -dependencies = [ - "libc", - "vswhom-sys", -] - -[[package]] -name = "vswhom-sys" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3b17ae1f6c8a2b28506cd96d412eebf83b4a0ff2cbefeeb952f2f9dfa44ba18" -dependencies = [ - "cc", - "libc", -] - -[[package]] -name = "vte" -version = "0.11.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f5022b5fbf9407086c180e9557be968742d839e68346af7792b8592489732197" -dependencies = [ - "arrayvec", - "utf8parse", - "vte_generate_state_changes", -] - -[[package]] -name = "vte_generate_state_changes" -version = "0.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d257817081c7dffcdbab24b9e62d2def62e2ff7d00b1c20062551e6cccc145ff" -dependencies = [ - "proc-macro2", - "quote", -] - -[[package]] -name = "wait-timeout" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9f200f5b12eb75f8c1ed65abd4b2db8a6e1b138a20de009dacee265a2498f3f6" -dependencies = [ - "libc", -] - -[[package]] -name = "walkdir" -version = "2.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "29790946404f91d9c5d06f9874efddea1dc06c5efe94541a7d6863108e3a5e4b" -dependencies = [ - "same-file", - "winapi-util", -] - -[[package]] -name = "want" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bfa7760aed19e106de2c7c0b581b509f2f25d3dacaf737cb82ac61bc6d760b0e" -dependencies = [ - "try-lock", -] - -[[package]] -name = "wasi" -version = "0.11.0+wasi-snapshot-preview1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9c8d87e72b64a3b4db28d11ce29237c246188f4f51057d65a7eab63b7987e423" - -[[package]] -name = "wasm-bindgen" -version = "0.2.92" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4be2531df63900aeb2bca0daaaddec08491ee64ceecbee5076636a3b026795a8" -dependencies = [ - "cfg-if", - "wasm-bindgen-macro", -] - -[[package]] -name = "wasm-bindgen-backend" -version = "0.2.92" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "614d787b966d3989fa7bb98a654e369c762374fd3213d212cfc0251257e747da" -dependencies = [ - "bumpalo", - "log", - "once_cell", - "proc-macro2", - "quote", - "syn 2.0.53", - "wasm-bindgen-shared", -] - -[[package]] -name = "wasm-bindgen-futures" -version = "0.4.42" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "76bc14366121efc8dbb487ab05bcc9d346b3b5ec0eaa76e46594cabbe51762c0" -dependencies = [ - "cfg-if", - "js-sys", - "wasm-bindgen", - "web-sys", -] - -[[package]] -name = "wasm-bindgen-macro" -version = "0.2.92" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a1f8823de937b71b9460c0c34e25f3da88250760bec0ebac694b49997550d726" -dependencies = [ - "quote", - "wasm-bindgen-macro-support", -] - -[[package]] -name = "wasm-bindgen-macro-support" -version = "0.2.92" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e94f17b526d0a461a191c78ea52bbce64071ed5c04c9ffe424dcb38f74171bb7" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", - "wasm-bindgen-backend", - "wasm-bindgen-shared", -] - -[[package]] -name = "wasm-bindgen-shared" -version = "0.2.92" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "af190c94f2773fdb3729c55b007a722abb5384da03bc0986df4c289bf5567e96" - -[[package]] -name = "wasm-streams" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b4609d447824375f43e1ffbc051b50ad8f4b3ae8219680c94452ea05eb240ac7" -dependencies = [ - "futures-util", - "js-sys", - "wasm-bindgen", - "wasm-bindgen-futures", - "web-sys", -] - -[[package]] -name = "web-sys" -version = "0.3.69" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "77afa9a11836342370f4817622a2f0f418b134426d91a82dfb48f532d2ec13ef" -dependencies = [ - "js-sys", - "wasm-bindgen", -] - -[[package]] -name = "web-time" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5a6580f308b1fad9207618087a65c04e7a10bc77e02c8e84e9b00dd4b12fa0bb" -dependencies = [ - "js-sys", - "wasm-bindgen", -] - -[[package]] -name = "web3" -version = "0.19.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5388522c899d1e1c96a4c307e3797e0f697ba7c77dd8e0e625ecba9dd0342937" -dependencies = [ - "arrayvec", - "base64 0.21.7", - "bytes", - "derive_more", - "ethabi", - "ethereum-types", - "futures", - "futures-timer", - "headers", - "hex", - "idna 0.4.0", - "jsonrpc-core", - "log", - "once_cell", - "parking_lot", - "pin-project", - "reqwest", - "rlp", - "secp256k1", - "serde", - "serde_json", - "soketto", - "tiny-keccak", - "tokio", - "tokio-stream", - "tokio-util", - "url", - "web3-async-native-tls", -] - -[[package]] -name = "web3-async-native-tls" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1f6d8d1636b2627fe63518d5a9b38a569405d9c9bc665c43c9c341de57227ebb" -dependencies = [ - "native-tls", - "thiserror", - "tokio", - "url", -] - -[[package]] -name = "webbrowser" -version = "0.8.13" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d1b04c569c83a9bb971dd47ec6fd48753315f4bf989b9b04a2e7ca4d7f0dc950" -dependencies = [ - "core-foundation", - "home", - "jni", - "log", - "ndk-context", - "objc", - "raw-window-handle", - "url", - "web-sys", -] - -[[package]] -name = "webpki" -version = "0.22.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ed63aea5ce73d0ff405984102c42de94fc55a6b75765d621c65262469b3c9b53" -dependencies = [ - "ring", - "untrusted", -] - -[[package]] -name = "webpki-roots" -version = "0.25.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5f20c57d8d7db6d3b86154206ae5d8fba62dd39573114de97c2cb0578251f8e1" - -[[package]] -name = "webpki-roots" -version = "0.26.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b3de34ae270483955a94f4b21bdaaeb83d508bb84a01435f393818edb0012009" -dependencies = [ - "rustls-pki-types", -] - -[[package]] -name = "which" -version = "4.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "87ba24419a2078cd2b0f2ede2691b6c66d8e47836da3b6db8265ebad47afbfc7" -dependencies = [ - "either", - "home", - "once_cell", - "rustix", -] - -[[package]] -name = "which" -version = "5.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9bf3ea8596f3a0dd5980b46430f2058dfe2c36a27ccfbb1845d6fbfcd9ba6e14" -dependencies = [ - "either", - "home", - "once_cell", - "rustix", - "windows-sys 0.48.0", -] - -[[package]] -name = "widestring" -version = "1.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "653f141f39ec16bba3c5abe400a0c60da7468261cc2cbf36805022876bc721a8" - -[[package]] -name = "winapi" -version = "0.3.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5c839a674fcd7a98952e593242ea400abe93992746761e38641405d28b00f419" -dependencies = [ - "winapi-i686-pc-windows-gnu", - "winapi-x86_64-pc-windows-gnu", -] - -[[package]] -name = "winapi-i686-pc-windows-gnu" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ac3b87c63620426dd9b991e5ce0329eff545bccbbb34f3be09ff6fb6ab51b7b6" - -[[package]] -name = "winapi-util" -version = "0.1.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f29e6f9198ba0d26b4c9f07dbe6f9ed633e1f3d5b8b414090084349e46a52596" -dependencies = [ - "winapi", -] - -[[package]] -name = "winapi-x86_64-pc-windows-gnu" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "712e227841d057c1ee1cd2fb22fa7e5a5461ae8e48fa2ca79ec42cfc1931183f" - -[[package]] -name = "windows-core" -version = "0.52.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "33ab640c8d7e35bf8ba19b884ba838ceb4fba93a4e8c65a9059d08afcfc683d9" -dependencies = [ - "windows-targets 0.52.4", -] - -[[package]] -name = "windows-sys" -version = "0.45.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "75283be5efb2831d37ea142365f009c02ec203cd29a3ebecbc093d52315b66d0" -dependencies = [ - "windows-targets 0.42.2", -] - -[[package]] -name = "windows-sys" -version = "0.48.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "677d2418bec65e3338edb076e806bc1ec15693c5d0104683f2efe857f61056a9" -dependencies = [ - "windows-targets 0.48.5", -] - -[[package]] -name = "windows-sys" -version = "0.52.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d" -dependencies = [ - "windows-targets 0.52.4", -] - -[[package]] -name = "windows-targets" -version = "0.42.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8e5180c00cd44c9b1c88adb3693291f1cd93605ded80c250a75d472756b4d071" -dependencies = [ - "windows_aarch64_gnullvm 0.42.2", - "windows_aarch64_msvc 0.42.2", - "windows_i686_gnu 0.42.2", - "windows_i686_msvc 0.42.2", - "windows_x86_64_gnu 0.42.2", - "windows_x86_64_gnullvm 0.42.2", - "windows_x86_64_msvc 0.42.2", -] - -[[package]] -name = "windows-targets" -version = "0.48.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a2fa6e2155d7247be68c096456083145c183cbbbc2764150dda45a87197940c" -dependencies = [ - "windows_aarch64_gnullvm 0.48.5", - "windows_aarch64_msvc 0.48.5", - "windows_i686_gnu 0.48.5", - "windows_i686_msvc 0.48.5", - "windows_x86_64_gnu 0.48.5", - "windows_x86_64_gnullvm 0.48.5", - "windows_x86_64_msvc 0.48.5", -] - -[[package]] -name = "windows-targets" -version = "0.52.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7dd37b7e5ab9018759f893a1952c9420d060016fc19a472b4bb20d1bdd694d1b" -dependencies = [ - "windows_aarch64_gnullvm 0.52.4", - "windows_aarch64_msvc 0.52.4", - "windows_i686_gnu 0.52.4", - "windows_i686_msvc 0.52.4", - "windows_x86_64_gnu 0.52.4", - "windows_x86_64_gnullvm 0.52.4", - "windows_x86_64_msvc 0.52.4", -] - -[[package]] -name = "windows_aarch64_gnullvm" -version = "0.42.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "597a5118570b68bc08d8d59125332c54f1ba9d9adeedeef5b99b02ba2b0698f8" - -[[package]] -name = "windows_aarch64_gnullvm" -version = "0.48.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2b38e32f0abccf9987a4e3079dfb67dcd799fb61361e53e2882c3cbaf0d905d8" - -[[package]] -name = "windows_aarch64_gnullvm" -version = "0.52.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bcf46cf4c365c6f2d1cc93ce535f2c8b244591df96ceee75d8e83deb70a9cac9" - -[[package]] -name = "windows_aarch64_msvc" -version = "0.42.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e08e8864a60f06ef0d0ff4ba04124db8b0fb3be5776a5cd47641e942e58c4d43" - -[[package]] -name = "windows_aarch64_msvc" -version = "0.48.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc35310971f3b2dbbf3f0690a219f40e2d9afcf64f9ab7cc1be722937c26b4bc" - -[[package]] -name = "windows_aarch64_msvc" -version = "0.52.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "da9f259dd3bcf6990b55bffd094c4f7235817ba4ceebde8e6d11cd0c5633b675" - -[[package]] -name = "windows_i686_gnu" -version = "0.42.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c61d927d8da41da96a81f029489353e68739737d3beca43145c8afec9a31a84f" - -[[package]] -name = "windows_i686_gnu" -version = "0.48.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a75915e7def60c94dcef72200b9a8e58e5091744960da64ec734a6c6e9b3743e" - -[[package]] -name = "windows_i686_gnu" -version = "0.52.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b474d8268f99e0995f25b9f095bc7434632601028cf86590aea5c8a5cb7801d3" - -[[package]] -name = "windows_i686_msvc" -version = "0.42.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "44d840b6ec649f480a41c8d80f9c65108b92d89345dd94027bfe06ac444d1060" - -[[package]] -name = "windows_i686_msvc" -version = "0.48.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8f55c233f70c4b27f66c523580f78f1004e8b5a8b659e05a4eb49d4166cca406" - -[[package]] -name = "windows_i686_msvc" -version = "0.52.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1515e9a29e5bed743cb4415a9ecf5dfca648ce85ee42e15873c3cd8610ff8e02" - -[[package]] -name = "windows_x86_64_gnu" -version = "0.42.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8de912b8b8feb55c064867cf047dda097f92d51efad5b491dfb98f6bbb70cb36" - -[[package]] -name = "windows_x86_64_gnu" -version = "0.48.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53d40abd2583d23e4718fddf1ebec84dbff8381c07cae67ff7768bbf19c6718e" - -[[package]] -name = "windows_x86_64_gnu" -version = "0.52.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5eee091590e89cc02ad514ffe3ead9eb6b660aedca2183455434b93546371a03" - -[[package]] -name = "windows_x86_64_gnullvm" -version = "0.42.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "26d41b46a36d453748aedef1486d5c7a85db22e56aff34643984ea85514e94a3" - -[[package]] -name = "windows_x86_64_gnullvm" -version = "0.48.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b7b52767868a23d5bab768e390dc5f5c55825b6d30b86c844ff2dc7414044cc" - -[[package]] -name = "windows_x86_64_gnullvm" -version = "0.52.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "77ca79f2451b49fa9e2af39f0747fe999fcda4f5e241b2898624dca97a1f2177" - -[[package]] -name = "windows_x86_64_msvc" -version = "0.42.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9aec5da331524158c6d1a4ac0ab1541149c0b9505fde06423b02f5ef0106b9f0" - -[[package]] -name = "windows_x86_64_msvc" -version = "0.48.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ed94fce61571a4006852b7389a063ab983c02eb1bb37b47f8272ce92d06d9538" - -[[package]] -name = "windows_x86_64_msvc" -version = "0.52.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32b752e52a2da0ddfbdbcc6fceadfeede4c939ed16d13e648833a61dfb611ed8" - -[[package]] -name = "winnow" -version = "0.5.40" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f593a95398737aeed53e489c785df13f3618e41dbcd6718c6addbf1395aa6876" -dependencies = [ - "memchr", -] - -[[package]] -name = "winreg" -version = "0.50.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "524e57b2c537c0f9b1e69f1965311ec12182b4122e45035b1508cd24d2adadb1" -dependencies = [ - "cfg-if", - "windows-sys 0.48.0", -] - -[[package]] -name = "wkt" -version = "0.10.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c3c2252781f8927974e8ba6a67c965a759a2b88ea2b1825f6862426bbb1c8f41" -dependencies = [ - "geo-types", - "log", - "num-traits", - "thiserror", -] - -[[package]] -name = "wyz" -version = "0.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "05f360fc0b24296329c78fda852a1e9ae82de9cf7b27dae4b7f62f118f77b9ed" -dependencies = [ - "tap", -] - -[[package]] -name = "x25519-dalek" -version = "2.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c7e468321c81fb07fa7f4c636c3972b9100f0346e5b6a9f2bd0603a52f7ed277" -dependencies = [ - "curve25519-dalek", - "rand_core", - "serde", - "zeroize", -] - -[[package]] -name = "x509-certificate" -version = "0.23.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "66534846dec7a11d7c50a74b7cdb208b9a581cad890b7866430d438455847c85" -dependencies = [ - "bcder", - "bytes", - "chrono", - "der", - "hex", - "pem", - "ring", - "signature", - "spki", - "thiserror", - "zeroize", -] - -[[package]] -name = "xattr" -version = "1.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8da84f1a25939b27f6820d92aed108f83ff920fdf11a7b19366c27c4cda81d4f" -dependencies = [ - "libc", - "linux-raw-sys", - "rustix", -] - -[[package]] -name = "xz2" -version = "0.1.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "388c44dc09d76f1536602ead6d325eb532f5c122f17782bd57fb47baeeb767e2" -dependencies = [ - "lzma-sys", -] - -[[package]] -name = "z85" -version = "3.0.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2a599daf1b507819c1121f0bf87fa37eb19daac6aff3aefefd4e6e2e0f2020fc" - -[[package]] -name = "zerocopy" -version = "0.7.32" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "74d4d3961e53fa4c9a25a8637fc2bfaf2595b3d3ae34875568a5cf64787716be" -dependencies = [ - "zerocopy-derive", -] - -[[package]] -name = "zerocopy-derive" -version = "0.7.32" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ce1b18ccd8e73a9321186f97e46f9f04b778851177567b1975109d26a08d2a6" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "zeroize" -version = "1.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "525b4ec142c6b68a2d10f01f7bbf6755599ca3f81ea53b8431b7dd348f5fdb2d" -dependencies = [ - "zeroize_derive", -] - -[[package]] -name = "zeroize_derive" -version = "1.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ce36e65b0d2999d2aafac989fb249189a141aee1f53c612c1f37d72631959f69" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.53", -] - -[[package]] -name = "zip" -version = "0.6.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "760394e246e4c28189f19d488c058bf16f564016aefac5d32bb1f3b51d5e9261" -dependencies = [ - "byteorder", - "crc32fast", - "crossbeam-utils", - "flate2", -] - -[[package]] -name = "zstd" -version = "0.13.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bffb3309596d527cfcba7dfc6ed6052f1d39dfbd7c867aa2e865e4a449c10110" -dependencies = [ - "zstd-safe", -] - -[[package]] -name = "zstd-safe" -version = "7.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "43747c7422e2924c11144d5229878b98180ef8b06cca4ab5af37afc8a8d8ea3e" -dependencies = [ - "zstd-sys", -] - -[[package]] -name = "zstd-sys" -version = "2.0.9+zstd.1.5.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9e16efa8a874a0481a574084d34cc26fdb3b99627480f785888deb6386506656" -dependencies = [ - "cc", - "pkg-config", -] diff --git a/Cargo.toml b/Cargo.toml deleted file mode 100644 index b47925d2f1..0000000000 --- a/Cargo.toml +++ /dev/null @@ -1,29 +0,0 @@ -[workspace] -members = [ - "dozer-ingestion", - "dozer-types", - "dozer-core", - "dozer-cli", - "dozer-sql", - "dozer-tracing", - "dozer-tests", - "dozer-utils", - "dozer-sink-clickhouse", -] -resolver = "2" - -[workspace.dependencies] -bincode = { version = "2.0.0-rc.3", features = ["derive"] } -deno_core = { version = "0.270", features = ["lazy_eval_snapshot"] } - -[patch.crates-io] -postgres = { git = "https://github.com/MaterializeInc/rust-postgres" } -postgres-protocol = { git = "https://github.com/MaterializeInc/rust-postgres" } -postgres-types = { git = "https://github.com/MaterializeInc/rust-postgres" } -tokio-postgres = { git = "https://github.com/MaterializeInc/rust-postgres" } - -clickhouse-rs = { git = "https://github.com/getdozer/clickhouse-rs" } -clickhouse-rs-cityhash-sys = { git = "https://github.com/getdozer/clickhouse-rs" } - -[profile.dev] -debug = 0 diff --git a/Cross.toml b/Cross.toml deleted file mode 100644 index 9554703875..0000000000 --- a/Cross.toml +++ /dev/null @@ -1,5 +0,0 @@ -[target.aarch64-unknown-linux-gnu] -dockerfile = "./ci/Dockerfile.aarch64-unknown-linux-gnu" - -[target.x86_64-unknown-linux-gnu] -dockerfile = "./ci/Dockerfile.x86_64-unknown-linux-gnu" diff --git a/LICENSE.txt b/LICENSE.txt deleted file mode 100644 index be3f7b28e5..0000000000 --- a/LICENSE.txt +++ /dev/null @@ -1,661 +0,0 @@ - GNU AFFERO GENERAL PUBLIC LICENSE - Version 3, 19 November 2007 - - Copyright (C) 2007 Free Software Foundation, Inc. - Everyone is permitted to copy and distribute verbatim copies - of this license document, but changing it is not allowed. - - Preamble - - The GNU Affero General Public License is a free, copyleft license for -software and other kinds of works, specifically designed to ensure -cooperation with the community in the case of network server software. - - The licenses for most software and other practical works are designed -to take away your freedom to share and change the works. By contrast, -our General Public Licenses are intended to guarantee your freedom to -share and change all versions of a program--to make sure it remains free -software for all its users. - - When we speak of free software, we are referring to freedom, not -price. Our General Public Licenses are designed to make sure that you -have the freedom to distribute copies of free software (and charge for -them if you wish), that you receive source code or can get it if you -want it, that you can change the software or use pieces of it in new -free programs, and that you know you can do these things. - - Developers that use our General Public Licenses protect your rights -with two steps: (1) assert copyright on the software, and (2) offer -you this License which gives you legal permission to copy, distribute -and/or modify the software. - - A secondary benefit of defending all users' freedom is that -improvements made in alternate versions of the program, if they -receive widespread use, become available for other developers to -incorporate. Many developers of free software are heartened and -encouraged by the resulting cooperation. However, in the case of -software used on network servers, this result may fail to come about. -The GNU General Public License permits making a modified version and -letting the public access it on a server without ever releasing its -source code to the public. - - The GNU Affero General Public License is designed specifically to -ensure that, in such cases, the modified source code becomes available -to the community. It requires the operator of a network server to -provide the source code of the modified version running there to the -users of that server. Therefore, public use of a modified version, on -a publicly accessible server, gives the public access to the source -code of the modified version. - - An older license, called the Affero General Public License and -published by Affero, was designed to accomplish similar goals. This is -a different license, not a version of the Affero GPL, but Affero has -released a new version of the Affero GPL which permits relicensing under -this license. - - The precise terms and conditions for copying, distribution and -modification follow. - - TERMS AND CONDITIONS - - 0. Definitions. - - "This License" refers to version 3 of the GNU Affero General Public License. - - "Copyright" also means copyright-like laws that apply to other kinds of -works, such as semiconductor masks. - - "The Program" refers to any copyrightable work licensed under this -License. Each licensee is addressed as "you". "Licensees" and -"recipients" may be individuals or organizations. - - To "modify" a work means to copy from or adapt all or part of the work -in a fashion requiring copyright permission, other than the making of an -exact copy. The resulting work is called a "modified version" of the -earlier work or a work "based on" the earlier work. - - A "covered work" means either the unmodified Program or a work based -on the Program. - - To "propagate" a work means to do anything with it that, without -permission, would make you directly or secondarily liable for -infringement under applicable copyright law, except executing it on a -computer or modifying a private copy. Propagation includes copying, -distribution (with or without modification), making available to the -public, and in some countries other activities as well. - - To "convey" a work means any kind of propagation that enables other -parties to make or receive copies. Mere interaction with a user through -a computer network, with no transfer of a copy, is not conveying. - - An interactive user interface displays "Appropriate Legal Notices" -to the extent that it includes a convenient and prominently visible -feature that (1) displays an appropriate copyright notice, and (2) -tells the user that there is no warranty for the work (except to the -extent that warranties are provided), that licensees may convey the -work under this License, and how to view a copy of this License. If -the interface presents a list of user commands or options, such as a -menu, a prominent item in the list meets this criterion. - - 1. Source Code. - - The "source code" for a work means the preferred form of the work -for making modifications to it. "Object code" means any non-source -form of a work. - - A "Standard Interface" means an interface that either is an official -standard defined by a recognized standards body, or, in the case of -interfaces specified for a particular programming language, one that -is widely used among developers working in that language. - - The "System Libraries" of an executable work include anything, other -than the work as a whole, that (a) is included in the normal form of -packaging a Major Component, but which is not part of that Major -Component, and (b) serves only to enable use of the work with that -Major Component, or to implement a Standard Interface for which an -implementation is available to the public in source code form. A -"Major Component", in this context, means a major essential component -(kernel, window system, and so on) of the specific operating system -(if any) on which the executable work runs, or a compiler used to -produce the work, or an object code interpreter used to run it. - - The "Corresponding Source" for a work in object code form means all -the source code needed to generate, install, and (for an executable -work) run the object code and to modify the work, including scripts to -control those activities. However, it does not include the work's -System Libraries, or general-purpose tools or generally available free -programs which are used unmodified in performing those activities but -which are not part of the work. For example, Corresponding Source -includes interface definition files associated with source files for -the work, and the source code for shared libraries and dynamically -linked subprograms that the work is specifically designed to require, -such as by intimate data communication or control flow between those -subprograms and other parts of the work. - - The Corresponding Source need not include anything that users -can regenerate automatically from other parts of the Corresponding -Source. - - The Corresponding Source for a work in source code form is that -same work. - - 2. Basic Permissions. - - All rights granted under this License are granted for the term of -copyright on the Program, and are irrevocable provided the stated -conditions are met. This License explicitly affirms your unlimited -permission to run the unmodified Program. The output from running a -covered work is covered by this License only if the output, given its -content, constitutes a covered work. This License acknowledges your -rights of fair use or other equivalent, as provided by copyright law. - - You may make, run and propagate covered works that you do not -convey, without conditions so long as your license otherwise remains -in force. You may convey covered works to others for the sole purpose -of having them make modifications exclusively for you, or provide you -with facilities for running those works, provided that you comply with -the terms of this License in conveying all material for which you do -not control copyright. Those thus making or running the covered works -for you must do so exclusively on your behalf, under your direction -and control, on terms that prohibit them from making any copies of -your copyrighted material outside their relationship with you. - - Conveying under any other circumstances is permitted solely under -the conditions stated below. Sublicensing is not allowed; section 10 -makes it unnecessary. - - 3. Protecting Users' Legal Rights From Anti-Circumvention Law. - - No covered work shall be deemed part of an effective technological -measure under any applicable law fulfilling obligations under article -11 of the WIPO copyright treaty adopted on 20 December 1996, or -similar laws prohibiting or restricting circumvention of such -measures. - - When you convey a covered work, you waive any legal power to forbid -circumvention of technological measures to the extent such circumvention -is effected by exercising rights under this License with respect to -the covered work, and you disclaim any intention to limit operation or -modification of the work as a means of enforcing, against the work's -users, your or third parties' legal rights to forbid circumvention of -technological measures. - - 4. Conveying Verbatim Copies. - - You may convey verbatim copies of the Program's source code as you -receive it, in any medium, provided that you conspicuously and -appropriately publish on each copy an appropriate copyright notice; -keep intact all notices stating that this License and any -non-permissive terms added in accord with section 7 apply to the code; -keep intact all notices of the absence of any warranty; and give all -recipients a copy of this License along with the Program. - - You may charge any price or no price for each copy that you convey, -and you may offer support or warranty protection for a fee. - - 5. Conveying Modified Source Versions. - - You may convey a work based on the Program, or the modifications to -produce it from the Program, in the form of source code under the -terms of section 4, provided that you also meet all of these conditions: - - a) The work must carry prominent notices stating that you modified - it, and giving a relevant date. - - b) The work must carry prominent notices stating that it is - released under this License and any conditions added under section - 7. This requirement modifies the requirement in section 4 to - "keep intact all notices". - - c) You must license the entire work, as a whole, under this - License to anyone who comes into possession of a copy. This - License will therefore apply, along with any applicable section 7 - additional terms, to the whole of the work, and all its parts, - regardless of how they are packaged. This License gives no - permission to license the work in any other way, but it does not - invalidate such permission if you have separately received it. - - d) If the work has interactive user interfaces, each must display - Appropriate Legal Notices; however, if the Program has interactive - interfaces that do not display Appropriate Legal Notices, your - work need not make them do so. - - A compilation of a covered work with other separate and independent -works, which are not by their nature extensions of the covered work, -and which are not combined with it such as to form a larger program, -in or on a volume of a storage or distribution medium, is called an -"aggregate" if the compilation and its resulting copyright are not -used to limit the access or legal rights of the compilation's users -beyond what the individual works permit. Inclusion of a covered work -in an aggregate does not cause this License to apply to the other -parts of the aggregate. - - 6. Conveying Non-Source Forms. - - You may convey a covered work in object code form under the terms -of sections 4 and 5, provided that you also convey the -machine-readable Corresponding Source under the terms of this License, -in one of these ways: - - a) Convey the object code in, or embodied in, a physical product - (including a physical distribution medium), accompanied by the - Corresponding Source fixed on a durable physical medium - customarily used for software interchange. - - b) Convey the object code in, or embodied in, a physical product - (including a physical distribution medium), accompanied by a - written offer, valid for at least three years and valid for as - long as you offer spare parts or customer support for that product - model, to give anyone who possesses the object code either (1) a - copy of the Corresponding Source for all the software in the - product that is covered by this License, on a durable physical - medium customarily used for software interchange, for a price no - more than your reasonable cost of physically performing this - conveying of source, or (2) access to copy the - Corresponding Source from a network server at no charge. - - c) Convey individual copies of the object code with a copy of the - written offer to provide the Corresponding Source. This - alternative is allowed only occasionally and noncommercially, and - only if you received the object code with such an offer, in accord - with subsection 6b. - - d) Convey the object code by offering access from a designated - place (gratis or for a charge), and offer equivalent access to the - Corresponding Source in the same way through the same place at no - further charge. You need not require recipients to copy the - Corresponding Source along with the object code. If the place to - copy the object code is a network server, the Corresponding Source - may be on a different server (operated by you or a third party) - that supports equivalent copying facilities, provided you maintain - clear directions next to the object code saying where to find the - Corresponding Source. Regardless of what server hosts the - Corresponding Source, you remain obligated to ensure that it is - available for as long as needed to satisfy these requirements. - - e) Convey the object code using peer-to-peer transmission, provided - you inform other peers where the object code and Corresponding - Source of the work are being offered to the general public at no - charge under subsection 6d. - - A separable portion of the object code, whose source code is excluded -from the Corresponding Source as a System Library, need not be -included in conveying the object code work. - - A "User Product" is either (1) a "consumer product", which means any -tangible personal property which is normally used for personal, family, -or household purposes, or (2) anything designed or sold for incorporation -into a dwelling. In determining whether a product is a consumer product, -doubtful cases shall be resolved in favor of coverage. For a particular -product received by a particular user, "normally used" refers to a -typical or common use of that class of product, regardless of the status -of the particular user or of the way in which the particular user -actually uses, or expects or is expected to use, the product. A product -is a consumer product regardless of whether the product has substantial -commercial, industrial or non-consumer uses, unless such uses represent -the only significant mode of use of the product. - - "Installation Information" for a User Product means any methods, -procedures, authorization keys, or other information required to install -and execute modified versions of a covered work in that User Product from -a modified version of its Corresponding Source. The information must -suffice to ensure that the continued functioning of the modified object -code is in no case prevented or interfered with solely because -modification has been made. - - If you convey an object code work under this section in, or with, or -specifically for use in, a User Product, and the conveying occurs as -part of a transaction in which the right of possession and use of the -User Product is transferred to the recipient in perpetuity or for a -fixed term (regardless of how the transaction is characterized), the -Corresponding Source conveyed under this section must be accompanied -by the Installation Information. But this requirement does not apply -if neither you nor any third party retains the ability to install -modified object code on the User Product (for example, the work has -been installed in ROM). - - The requirement to provide Installation Information does not include a -requirement to continue to provide support service, warranty, or updates -for a work that has been modified or installed by the recipient, or for -the User Product in which it has been modified or installed. Access to a -network may be denied when the modification itself materially and -adversely affects the operation of the network or violates the rules and -protocols for communication across the network. - - Corresponding Source conveyed, and Installation Information provided, -in accord with this section must be in a format that is publicly -documented (and with an implementation available to the public in -source code form), and must require no special password or key for -unpacking, reading or copying. - - 7. Additional Terms. - - "Additional permissions" are terms that supplement the terms of this -License by making exceptions from one or more of its conditions. -Additional permissions that are applicable to the entire Program shall -be treated as though they were included in this License, to the extent -that they are valid under applicable law. If additional permissions -apply only to part of the Program, that part may be used separately -under those permissions, but the entire Program remains governed by -this License without regard to the additional permissions. - - When you convey a copy of a covered work, you may at your option -remove any additional permissions from that copy, or from any part of -it. (Additional permissions may be written to require their own -removal in certain cases when you modify the work.) You may place -additional permissions on material, added by you to a covered work, -for which you have or can give appropriate copyright permission. - - Notwithstanding any other provision of this License, for material you -add to a covered work, you may (if authorized by the copyright holders of -that material) supplement the terms of this License with terms: - - a) Disclaiming warranty or limiting liability differently from the - terms of sections 15 and 16 of this License; or - - b) Requiring preservation of specified reasonable legal notices or - author attributions in that material or in the Appropriate Legal - Notices displayed by works containing it; or - - c) Prohibiting misrepresentation of the origin of that material, or - requiring that modified versions of such material be marked in - reasonable ways as different from the original version; or - - d) Limiting the use for publicity purposes of names of licensors or - authors of the material; or - - e) Declining to grant rights under trademark law for use of some - trade names, trademarks, or service marks; or - - f) Requiring indemnification of licensors and authors of that - material by anyone who conveys the material (or modified versions of - it) with contractual assumptions of liability to the recipient, for - any liability that these contractual assumptions directly impose on - those licensors and authors. - - All other non-permissive additional terms are considered "further -restrictions" within the meaning of section 10. If the Program as you -received it, or any part of it, contains a notice stating that it is -governed by this License along with a term that is a further -restriction, you may remove that term. If a license document contains -a further restriction but permits relicensing or conveying under this -License, you may add to a covered work material governed by the terms -of that license document, provided that the further restriction does -not survive such relicensing or conveying. - - If you add terms to a covered work in accord with this section, you -must place, in the relevant source files, a statement of the -additional terms that apply to those files, or a notice indicating -where to find the applicable terms. - - Additional terms, permissive or non-permissive, may be stated in the -form of a separately written license, or stated as exceptions; -the above requirements apply either way. - - 8. Termination. - - You may not propagate or modify a covered work except as expressly -provided under this License. Any attempt otherwise to propagate or -modify it is void, and will automatically terminate your rights under -this License (including any patent licenses granted under the third -paragraph of section 11). - - However, if you cease all violation of this License, then your -license from a particular copyright holder is reinstated (a) -provisionally, unless and until the copyright holder explicitly and -finally terminates your license, and (b) permanently, if the copyright -holder fails to notify you of the violation by some reasonable means -prior to 60 days after the cessation. - - Moreover, your license from a particular copyright holder is -reinstated permanently if the copyright holder notifies you of the -violation by some reasonable means, this is the first time you have -received notice of violation of this License (for any work) from that -copyright holder, and you cure the violation prior to 30 days after -your receipt of the notice. - - Termination of your rights under this section does not terminate the -licenses of parties who have received copies or rights from you under -this License. If your rights have been terminated and not permanently -reinstated, you do not qualify to receive new licenses for the same -material under section 10. - - 9. Acceptance Not Required for Having Copies. - - You are not required to accept this License in order to receive or -run a copy of the Program. Ancillary propagation of a covered work -occurring solely as a consequence of using peer-to-peer transmission -to receive a copy likewise does not require acceptance. However, -nothing other than this License grants you permission to propagate or -modify any covered work. These actions infringe copyright if you do -not accept this License. Therefore, by modifying or propagating a -covered work, you indicate your acceptance of this License to do so. - - 10. Automatic Licensing of Downstream Recipients. - - Each time you convey a covered work, the recipient automatically -receives a license from the original licensors, to run, modify and -propagate that work, subject to this License. You are not responsible -for enforcing compliance by third parties with this License. - - An "entity transaction" is a transaction transferring control of an -organization, or substantially all assets of one, or subdividing an -organization, or merging organizations. If propagation of a covered -work results from an entity transaction, each party to that -transaction who receives a copy of the work also receives whatever -licenses to the work the party's predecessor in interest had or could -give under the previous paragraph, plus a right to possession of the -Corresponding Source of the work from the predecessor in interest, if -the predecessor has it or can get it with reasonable efforts. - - You may not impose any further restrictions on the exercise of the -rights granted or affirmed under this License. For example, you may -not impose a license fee, royalty, or other charge for exercise of -rights granted under this License, and you may not initiate litigation -(including a cross-claim or counterclaim in a lawsuit) alleging that -any patent claim is infringed by making, using, selling, offering for -sale, or importing the Program or any portion of it. - - 11. Patents. - - A "contributor" is a copyright holder who authorizes use under this -License of the Program or a work on which the Program is based. The -work thus licensed is called the contributor's "contributor version". - - A contributor's "essential patent claims" are all patent claims -owned or controlled by the contributor, whether already acquired or -hereafter acquired, that would be infringed by some manner, permitted -by this License, of making, using, or selling its contributor version, -but do not include claims that would be infringed only as a -consequence of further modification of the contributor version. For -purposes of this definition, "control" includes the right to grant -patent sublicenses in a manner consistent with the requirements of -this License. - - Each contributor grants you a non-exclusive, worldwide, royalty-free -patent license under the contributor's essential patent claims, to -make, use, sell, offer for sale, import and otherwise run, modify and -propagate the contents of its contributor version. - - In the following three paragraphs, a "patent license" is any express -agreement or commitment, however denominated, not to enforce a patent -(such as an express permission to practice a patent or covenant not to -sue for patent infringement). To "grant" such a patent license to a -party means to make such an agreement or commitment not to enforce a -patent against the party. - - If you convey a covered work, knowingly relying on a patent license, -and the Corresponding Source of the work is not available for anyone -to copy, free of charge and under the terms of this License, through a -publicly available network server or other readily accessible means, -then you must either (1) cause the Corresponding Source to be so -available, or (2) arrange to deprive yourself of the benefit of the -patent license for this particular work, or (3) arrange, in a manner -consistent with the requirements of this License, to extend the patent -license to downstream recipients. "Knowingly relying" means you have -actual knowledge that, but for the patent license, your conveying the -covered work in a country, or your recipient's use of the covered work -in a country, would infringe one or more identifiable patents in that -country that you have reason to believe are valid. - - If, pursuant to or in connection with a single transaction or -arrangement, you convey, or propagate by procuring conveyance of, a -covered work, and grant a patent license to some of the parties -receiving the covered work authorizing them to use, propagate, modify -or convey a specific copy of the covered work, then the patent license -you grant is automatically extended to all recipients of the covered -work and works based on it. - - A patent license is "discriminatory" if it does not include within -the scope of its coverage, prohibits the exercise of, or is -conditioned on the non-exercise of one or more of the rights that are -specifically granted under this License. You may not convey a covered -work if you are a party to an arrangement with a third party that is -in the business of distributing software, under which you make payment -to the third party based on the extent of your activity of conveying -the work, and under which the third party grants, to any of the -parties who would receive the covered work from you, a discriminatory -patent license (a) in connection with copies of the covered work -conveyed by you (or copies made from those copies), or (b) primarily -for and in connection with specific products or compilations that -contain the covered work, unless you entered into that arrangement, -or that patent license was granted, prior to 28 March 2007. - - Nothing in this License shall be construed as excluding or limiting -any implied license or other defenses to infringement that may -otherwise be available to you under applicable patent law. - - 12. No Surrender of Others' Freedom. - - If conditions are imposed on you (whether by court order, agreement or -otherwise) that contradict the conditions of this License, they do not -excuse you from the conditions of this License. If you cannot convey a -covered work so as to satisfy simultaneously your obligations under this -License and any other pertinent obligations, then as a consequence you may -not convey it at all. For example, if you agree to terms that obligate you -to collect a royalty for further conveying from those to whom you convey -the Program, the only way you could satisfy both those terms and this -License would be to refrain entirely from conveying the Program. - - 13. Remote Network Interaction; Use with the GNU General Public License. - - Notwithstanding any other provision of this License, if you modify the -Program, your modified version must prominently offer all users -interacting with it remotely through a computer network (if your version -supports such interaction) an opportunity to receive the Corresponding -Source of your version by providing access to the Corresponding Source -from a network server at no charge, through some standard or customary -means of facilitating copying of software. This Corresponding Source -shall include the Corresponding Source for any work covered by version 3 -of the GNU General Public License that is incorporated pursuant to the -following paragraph. - - Notwithstanding any other provision of this License, you have -permission to link or combine any covered work with a work licensed -under version 3 of the GNU General Public License into a single -combined work, and to convey the resulting work. The terms of this -License will continue to apply to the part which is the covered work, -but the work with which it is combined will remain governed by version -3 of the GNU General Public License. - - 14. Revised Versions of this License. - - The Free Software Foundation may publish revised and/or new versions of -the GNU Affero General Public License from time to time. Such new versions -will be similar in spirit to the present version, but may differ in detail to -address new problems or concerns. - - Each version is given a distinguishing version number. If the -Program specifies that a certain numbered version of the GNU Affero General -Public License "or any later version" applies to it, you have the -option of following the terms and conditions either of that numbered -version or of any later version published by the Free Software -Foundation. If the Program does not specify a version number of the -GNU Affero General Public License, you may choose any version ever published -by the Free Software Foundation. - - If the Program specifies that a proxy can decide which future -versions of the GNU Affero General Public License can be used, that proxy's -public statement of acceptance of a version permanently authorizes you -to choose that version for the Program. - - Later license versions may give you additional or different -permissions. However, no additional obligations are imposed on any -author or copyright holder as a result of your choosing to follow a -later version. - - 15. Disclaimer of Warranty. - - THERE IS NO WARRANTY FOR THE PROGRAM, TO THE EXTENT PERMITTED BY -APPLICABLE LAW. EXCEPT WHEN OTHERWISE STATED IN WRITING THE COPYRIGHT -HOLDERS AND/OR OTHER PARTIES PROVIDE THE PROGRAM "AS IS" WITHOUT WARRANTY -OF ANY KIND, EITHER EXPRESSED OR IMPLIED, INCLUDING, BUT NOT LIMITED TO, -THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR -PURPOSE. THE ENTIRE RISK AS TO THE QUALITY AND PERFORMANCE OF THE PROGRAM -IS WITH YOU. SHOULD THE PROGRAM PROVE DEFECTIVE, YOU ASSUME THE COST OF -ALL NECESSARY SERVICING, REPAIR OR CORRECTION. - - 16. Limitation of Liability. - - IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING -WILL ANY COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MODIFIES AND/OR CONVEYS -THE PROGRAM AS PERMITTED ABOVE, BE LIABLE TO YOU FOR DAMAGES, INCLUDING ANY -GENERAL, SPECIAL, INCIDENTAL OR CONSEQUENTIAL DAMAGES ARISING OUT OF THE -USE OR INABILITY TO USE THE PROGRAM (INCLUDING BUT NOT LIMITED TO LOSS OF -DATA OR DATA BEING RENDERED INACCURATE OR LOSSES SUSTAINED BY YOU OR THIRD -PARTIES OR A FAILURE OF THE PROGRAM TO OPERATE WITH ANY OTHER PROGRAMS), -EVEN IF SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE POSSIBILITY OF -SUCH DAMAGES. - - 17. Interpretation of Sections 15 and 16. - - If the disclaimer of warranty and limitation of liability provided -above cannot be given local legal effect according to their terms, -reviewing courts shall apply local law that most closely approximates -an absolute waiver of all civil liability in connection with the -Program, unless a warranty or assumption of liability accompanies a -copy of the Program in return for a fee. - - END OF TERMS AND CONDITIONS - - How to Apply These Terms to Your New Programs - - If you develop a new program, and you want it to be of the greatest -possible use to the public, the best way to achieve this is to make it -free software which everyone can redistribute and change under these terms. - - To do so, attach the following notices to the program. It is safest -to attach them to the start of each source file to most effectively -state the exclusion of warranty; and each file should have at least -the "copyright" line and a pointer to where the full notice is found. - - - Copyright (C) - - This program is free software: you can redistribute it and/or modify - it under the terms of the GNU Affero General Public License as published by - the Free Software Foundation, either version 3 of the License, or - (at your option) any later version. - - This program is distributed in the hope that it will be useful, - but WITHOUT ANY WARRANTY; without even the implied warranty of - MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the - GNU Affero General Public License for more details. - - You should have received a copy of the GNU Affero General Public License - along with this program. If not, see . - -Also add information on how to contact you by electronic and paper mail. - - If your software can interact with users remotely through a computer -network, you should also make sure that it provides a way for users to -get its source. For example, if your program is a web application, its -interface could display a "Source" link that leads users to an archive -of the code. There are many ways you could offer source, and different -solutions will be better for different programs; see section 13 for the -specific requirements. - - You should also get your employer (if you work as a programmer) or school, -if any, to sign a "copyright disclaimer" for the program, if necessary. -For more information on this, and how to apply and follow the GNU AGPL, see -. diff --git a/README.md b/README.md deleted file mode 100644 index c971899975..0000000000 --- a/README.md +++ /dev/null @@ -1,53 +0,0 @@ -## Overview - -Dozer is a **real time data movement tool leveraging CDC from various sources to multiple sinks.** - -Dozer is magnitudes of times faster than Debezium+Kafka and natively supports stateless transformations. -Primarily used for moving data into warehouses. In our own application, we move data to **Clickhouse** and build data APIs and integration with LLMs. - -## How to use it -Dozer runs with a single configuration file like the following: -```yaml -app_name: dozer-bench -version: 1 -connections: - - name: pg_1 - config: !Postgres - user: user - password: postgres - host: localhost - port: 5432 - database: customers -sinks: - - name: customers - config: !Dummy - table_name: customers -``` - -Full documentation can be found [here](https://github.com/getdozer/dozer/blob/main/dozer-types/src/models/config.rs#L15) - - -## Supported Sources - -| Connector | Extraction | Resuming | Enterprise | -| -------------------- | ---------- | -------- | ------------------- | -| Postgres | ✅ | ✅ | ✅ | -| MySQL | ✅ | ✅ | ✅ | -| Snowflake | ✅ | ✅ | ✅ | -| Kafka | ✅ | 🚧 | ✅ | -| MongoDB | ✅ | 🎯 | ✅ | -| Amazon S3 | ✅ | 🎯 | ✅ | -| Google Cloud Storage | ✅ | 🎯 | ✅ | -| **Oracle | ✅ | ✅ | **Enterprise Only** | -| **Aerospike | ✅ | ✅ | **Enterprise Only** | - - -## Supported Sinks -| Database | Connectivity | Enterprise | -| ---------- | ------------ | ------------------- | -| Clickhouse | ✅ | | -| Postgres | ✅ | | -| MySQL | ✅ | | -| Big Query | ✅ | | -| Oracle | ✅ | **Enterprise Only** | -| Aerospike | ✅ | **Enterprise Only** | \ No newline at end of file diff --git a/SECURITY.md b/SECURITY.md deleted file mode 100644 index b71205cadd..0000000000 --- a/SECURITY.md +++ /dev/null @@ -1,21 +0,0 @@ -# Security Policy - -Dozer takes security issues very seriously. If you have uncovered a vulnerability, please get in touch via the e-mail address security@getdozer.io. - - -⚠️ Please do not file GitHub issues or post on our public forum for security vulnerabilities. ⚠️ - - Please describe the issue and preferably a way to reproduce it. If you can share the following details, it will help us triage the issue more quickly. - - Type of issue - - Affected versions and impact - - Source file path - - Steps to reproduce - - Exploit code - -Note that this security address should be used only for undisclosed vulnerabilities. - -## Supported Versions - -Currently security updates will only be merged to latest release. - -We will confirm if the issue exists within two days, and if it is accepted and fixed, the update will be included in the next release, which usually happens every Friday. diff --git a/ci/Dockerfile b/ci/Dockerfile deleted file mode 100644 index e5d97d456e..0000000000 --- a/ci/Dockerfile +++ /dev/null @@ -1,31 +0,0 @@ -FROM ubuntu:20.04 - -RUN apt-get update \ - && apt-get install -y \ - libssl-dev \ - odbcinst \ - unixodbc \ - curl \ - unzip \ - libaio1 - -# INSTALL PROTOBUF -RUN curl -LO https://github.com/protocolbuffers/protobuf/releases/download/v3.18.1/protoc-3.18.1-linux-x86_64.zip -RUN unzip protoc-3.18.1-linux-x86_64.zip -d /usr/local/protoc -RUN rm protoc-3.18.1-linux-x86_64.zip - -# INSTALL SNOWFLAKE DRIVERS -RUN curl -LO https://sfc-repo.snowflakecomputing.com/odbc/linux/2.25.7/snowflake-odbc-2.25.7.x86_64.deb -RUN dpkg -i snowflake-odbc-2.25.7.x86_64.deb -RUN rm snowflake-odbc-2.25.7.x86_64.deb - -# INSTALL DOZER -RUN echo "Installing dozer binary" - -COPY target/release/dozer /usr/local/bin/ - -ENV PATH="$PATH:/usr/local/protoc/bin" - -WORKDIR /usr/dozer -ENTRYPOINT ["dozer"] -CMD ["run"] diff --git a/ci/Dockerfile.aarch64-unknown-linux-gnu b/ci/Dockerfile.aarch64-unknown-linux-gnu deleted file mode 100644 index 3d51cfc25b..0000000000 --- a/ci/Dockerfile.aarch64-unknown-linux-gnu +++ /dev/null @@ -1,16 +0,0 @@ -FROM ghcr.io/cross-rs/aarch64-unknown-linux-gnu:main@sha256:b4f5bf74812f9bb6516140d4b83d1f173c2d5ce0523f3e1c2253d99d851c734f - -ENV PKG_CONFIG_ALLOW_CROSS="true" - -RUN dpkg --add-architecture arm64 && \ - apt-get update && \ - apt-get install --assume-yes clang-8 libclang-8-dev binutils-aarch64-linux-gnu zlib1g-dev:arm64 unzip - -# INSTALL PROTOBUF -ENV PROTOBUF_FILE_NAME=protoc-3.18.2-linux-x86_64.zip -RUN curl -LO https://github.com/protocolbuffers/protobuf/releases/download/v3.18.2/${PROTOBUF_FILE_NAME} -ENV PROTOC_DIR=/usr/local/protoc -RUN unzip ${PROTOBUF_FILE_NAME} -d ${PROTOC_DIR} -RUN chmod -R a+xr ${PROTOC_DIR} -ENV PROTOC=${PROTOC_DIR}/bin/protoc -RUN rm ${PROTOBUF_FILE_NAME} diff --git a/ci/Dockerfile.x86_64-unknown-linux-gnu b/ci/Dockerfile.x86_64-unknown-linux-gnu deleted file mode 100644 index 8bc56f4fb0..0000000000 --- a/ci/Dockerfile.x86_64-unknown-linux-gnu +++ /dev/null @@ -1,14 +0,0 @@ -FROM ghcr.io/cross-rs/x86_64-unknown-linux-gnu:main@sha256:bf0cd3027befe882feb5a2b4040dc6dbdcb799b25c5338342a03163cea43da1b - -RUN apt-get update && \ - apt-get install --assume-yes clang libclang-dev binutils-aarch64-linux-gnu unzip - -# INSTALL PROTOBUF -ENV PROTOBUF_FILE_NAME=protoc-3.18.2-linux-x86_64.zip -RUN curl -LO https://github.com/protocolbuffers/protobuf/releases/download/v3.18.2/${PROTOBUF_FILE_NAME} -RUN unzip ${PROTOBUF_FILE_NAME} -d /usr/local/protoc -ENV PROTOC_DIR=/usr/local/protoc -RUN unzip ${PROTOBUF_FILE_NAME} -d ${PROTOC_DIR} -RUN chmod -R a+xr ${PROTOC_DIR} -ENV PROTOC=${PROTOC_DIR}/bin/protoc -RUN rm ${PROTOBUF_FILE_NAME} diff --git a/ci/README.md b/ci/README.md deleted file mode 100644 index 7d00e3f1eb..0000000000 --- a/ci/README.md +++ /dev/null @@ -1,10 +0,0 @@ -# Cross compilation - -We use [cross](https://github.com/cross-rs/cross) to work around [bug](https://github.com/rust-lang/rust-bindgen/issues/1229) in `bind-gen`. - -To test cross compilation locally: - -```bash -cargo install cross -cross build --target ${target} --bin dozer -``` diff --git a/ci/download-latest.sh b/ci/download-latest.sh deleted file mode 100755 index 9c6f06fe36..0000000000 --- a/ci/download-latest.sh +++ /dev/null @@ -1,174 +0,0 @@ -#!/bin/sh - -# GLOBALS - -# Colors -RED='\033[31m' -GREEN='\033[32m' -DEFAULT='\033[0m' - -# Project name -PNAME='dozer' - -# GitHub API address -GITHUB_API='https://api.github.com/repos/getdozer/dozer/releases' -# GitHub Release address -GITHUB_REL='https://github.com/getdozer/dozer/releases/latest/download/' - -# FUNCTIONS - -# Gets the version of the latest stable version of Dozer by setting the $latest variable. -# Returns 0 in case of success, 1 otherwise. -get_latest() { - # temp_file is needed because the grep would start before the download is over - temp_file=$(mktemp -q /tmp/$PNAME.XXXXXXXXX) - latest_release="$GITHUB_API/latest" - - if [ $? -ne 0 ]; then - echo "$0: Can't create temp file." - fetch_release_failure_usage - exit 1 - fi - - if [ -z "$GITHUB_PAT" ]; then - curl -s "$latest_release" > "$temp_file" || return 1 - else - curl -H "Authorization: token $GITHUB_PAT" -s "$latest_release" > "$temp_file" || return 1 - fi - - latest="$(cat "$temp_file" | grep '"tag_name":' | cut -d ':' -f2 | tr -d '"' | tr -d ',' | tr -d ' ')" - - rm -f "$temp_file" - return 0 -} - -# Gets the OS by setting the $os variable -# Returns 0 in case of success, 1 otherwise -get_os() { - os_name=$(uname -s) - case "$os_name" in - 'Darwin') - os='macos' - ;; - 'Linux') - os='linux' - ;; - *) - return 1 - esac - return 0 -} - -# Gets the architecture by setting the $archi variable. -# Returns 0 in case of success, 1 otherwise. -get_archi() { - architecture=$(uname -m) - case "$architecture" in - 'x86_64' | 'amd64' ) - archi='amd64' - ;; - 'arm64') - # macOS M1/M2 - if [ $os = 'macos' ]; then - archi='arm64' - else - archi='amd64' - fi - ;; - *) - return 1 - esac - return 0 -} - -success_download() { - printf "$GREEN%s\n$DEFAULT" "Dozer $latest binary successfully downloaded as '$release_file' file." -} - -success_unzip_remove() { - printf "$GREEN%s\n$DEFAULT" "Dozer $latest binary successfully extracted as 'dozer'." - echo '' - echo 'Run it:' - echo " $ $PNAME" - echo 'Usage:' - echo " $ $PNAME --help" -} - -not_available_failure_usage() { - printf "$RED%s\n$DEFAULT" 'ERROR: Dozer binary is not available for your OS distribution or your architecture yet.' - echo '' - echo 'However, you can easily compile the binary from the source files.' - echo 'Follow the steps at the page ("Source" tab): https://docs.dozer.com/learn/getting_started/installation.html' -} - -fetch_release_failure_usage() { - echo '' - printf "$RED%s\n$DEFAULT" 'ERROR: Impossible to get the latest stable version of Dozer.' - echo 'Please let us know about this issue: https://github.com/getdozer/dozer/issues/new/choose' - echo '' - echo 'In the meantime, you can manually download the appropriate binary from the GitHub release assets here: https://github.com/getdozer/dozer/releases/latest' -} - -fill_release_variables() { - # Fill $latest variable. - if ! get_latest; then - fetch_release_failure_usage - exit 1 - fi - if [ "$latest" = '' ]; then - fetch_release_failure_usage - exit 1 - fi - # Fill $os variable. - if ! get_os; then - not_available_failure_usage - exit 1 - fi - # Fill $archi variable. - if ! get_archi; then - not_available_failure_usage - exit 1 - fi -} - -unzip_file() { - chmod 744 "$release_file" - export LANG=en_US.UTF-8 - export LC_ALL=$LANG - tar -xvzf "$release_file" -C "./" - rm -f "$release_file" -} - -download_binary() { - fill_release_variables - binary_name="$PNAME" - case "$os" in - 'macos') - release_file="$PNAME-$os-$archi-$latest.tar.gz" - ;; - 'linux') - release_file="$PNAME-$os-$archi-$latest.tar.gz" - ;; - *) - return 1 - esac - - echo "Downloading Dozer binary $latest for $os, architecture $archi -- $release_file..." - - # Fetch the Dozer binary. - curl --fail -OL "$GITHUB_REL/$release_file" - if [ $? -ne 0 ]; then - fetch_release_failure_usage - exit 1 - fi - success_download - - unzip_file - success_unzip_remove -} - -# MAIN -main() { - download_binary -} -main diff --git a/config/README.md b/config/README.md deleted file mode 100644 index 7058bea513..0000000000 --- a/config/README.md +++ /dev/null @@ -1,49 +0,0 @@ -## E2E tests configuration - -Running e2e connector tests requires user to have running particular service servers and proper configuration. Configuration files are stored in /config/tests/local folder. -To create folder and copy files user can run commands from snippet below. - -```shell -mkdir ./config/tests/local -cp ./config/tests/*.{yaml,json} ./config/tests/local -``` -### Postgres -To run postgres tests, you must have installed postgres locally or on server. Requirements can be found here - [Requirments][5] - -[Postgres config file][1] - -```shell -cargo test connector_e2e_connect_postgres -- --ignore -``` - -### Snowflake - -To run snowflake, you must have created database in snowflake with running warehouse. - -[Snowflake config file][2] - -```shell -cargo test connector_e2e_connect_snowflake --features=snowflake -- --ignore -``` - -### Debezium (kafka) - -This requires installation of kafka, postgres and debezium connector. Instruction can be found in [debezium tutorials][6] - -[Debezium config file][3] - -[Postgres source config file][4] - -[Connector config file][7] - -```shell -cargo test connector_e2e_connect_debezium -- --ignore -``` - -[1]: https://github.com/getdozer/dozer/config/tests/test.postgres.yaml -[2]: https://github.com/getdozer/dozer/config/tests/test.snowflake.yaml -[3]: https://github.com/getdozer/dozer/config/tests/test.debezium.yaml -[4]: https://github.com/getdozer/dozer/config/tests/test.postgres.auth.yaml -[5]: https://github.com/getdozer/dozer/dozer-ingestion/src/connectors/postgres/readme.md -[6]: https://debezium.io/documentation/reference/stable/tutorial.html -[7]: https://github.com/getdozer/dozer/config/tests/test.register-postgres.json diff --git a/config/tests/local/.gitkeep b/config/tests/local/.gitkeep deleted file mode 100644 index e69de29bb2..0000000000 diff --git a/config/tests/test.debezium-with-schema-registry.yaml b/config/tests/test.debezium-with-schema-registry.yaml deleted file mode 100644 index cc2dc30e3e..0000000000 --- a/config/tests/test.debezium-with-schema-registry.yaml +++ /dev/null @@ -1,40 +0,0 @@ -app_name: dozer-kafka-with-schema-registry-test -version: 1 - -api: - rest: - port: 8080 - url: "[::0]" - cors: true - grpc: - port: 50051 - url: "[::0]" - cors: true - web: true - auth: false - internal: - port: 50052 - host: "[::1]" - -connections: - - name: products - db_type: Kafka - authentication: !Kafka - broker: ${DEBEZIUM_KAFKA_WITH_REGISTRY_BROKER} - topic: ${DEBEZIUM_KAFKA_TOPIC} - schema_registry_url: ${DEBEZIUM_KAFKA_SCHEMA_REGISTRY_URL} - -source: - - name: products - table_name: ${DEBEZIUM_TABLE_NAME} - connection: products - columns: - - id - -endpoints: - - name: products - path: /products - sql: select id from products; - index: - primary_key: - - id diff --git a/config/tests/test.debezium.pg.yaml b/config/tests/test.debezium.pg.yaml deleted file mode 100644 index ba0137c8b0..0000000000 --- a/config/tests/test.debezium.pg.yaml +++ /dev/null @@ -1,8 +0,0 @@ -postgres_source_authentication: !Postgres - user: ${POSTGRES_USER} - password: ${POSTGRES_PASSWORD} - host: ${POSTGRES_HOST} - port: ${POSTGRES_PORT} - database: ${POSTGRES_DATABASE} - -connector_url: ${KAFKA_CONNECTOR_URL} \ No newline at end of file diff --git a/config/tests/test.debezium.yaml b/config/tests/test.debezium.yaml deleted file mode 100644 index 68dcc7d158..0000000000 --- a/config/tests/test.debezium.yaml +++ /dev/null @@ -1,39 +0,0 @@ -app_name: dozer-kafka-test -version: 1 - -api: - rest: - port: 8080 - url: "[::0]" - cors: true - grpc: - port: 50051 - url: "[::0]" - cors: true - web: true - auth: false - internal: - port: 50052 - host: "[::1]" - -connections: - - name: products - db_type: Kafka - authentication: !Kafka - broker: ${DEBEZIUM_KAFKA_BROKER} - topic: ${DEBEZIUM_KAFKA_TOPIC} - -sources: - - name: products - table_name: ${DEBEZIUM_TABLE_NAME} - connection: products - columns: - - id - -endpoints: - - name: products - path: /products - sql: select id from products; - index: - primary_key: - - id diff --git a/config/tests/test.postgres.auth.yaml b/config/tests/test.postgres.auth.yaml deleted file mode 100644 index 3e4c3588c0..0000000000 --- a/config/tests/test.postgres.auth.yaml +++ /dev/null @@ -1,6 +0,0 @@ -!Postgres - user: ${POSTGRES_USER} - password: ${POSTGRES_PASSWORD} - host: ${POSTGRES_HOST} - port: ${POSTGRES_PORT} - database: ${POSTGRES_DATABASE} \ No newline at end of file diff --git a/config/tests/test.postgres.yaml b/config/tests/test.postgres.yaml deleted file mode 100644 index 9ec8f20eb1..0000000000 --- a/config/tests/test.postgres.yaml +++ /dev/null @@ -1,42 +0,0 @@ -app_name: postgres-test -version: 1 - -api: - rest: - port: 8080 - url: "[::0]" - cors: true - grpc: - port: 50051 - url: "[::0]" - cors: true - web: true - auth: false - internal: - port: 50052 - host: "[::1]" - -connections: - - name: products_test - db_type: Postgres - authentication: !Postgres - user: ${POSTGRES_USER} - password: ${POSTGRES_PASSWORD} - host: ${POSTGRES_HOST} - port: ${POSTGRES_PORT} - database: ${POSTGRES_DATABASE} - -sources: - - name: users - table_name: users - connection: products_test - columns: - - id - -endpoints: - - name: products_test - path: /products_test - sql: select id from products_test; - index: - primary_key: - - id diff --git a/config/tests/test.register-postgres-with-schema-registry.json b/config/tests/test.register-postgres-with-schema-registry.json deleted file mode 100644 index 7242122d6c..0000000000 --- a/config/tests/test.register-postgres-with-schema-registry.json +++ /dev/null @@ -1,24 +0,0 @@ -{ - "name": "dozer_with_schema_registry", - "config": { - "name": "dozer_with_schema_registry", - "connector.class": "io.debezium.connector.postgresql.PostgresConnector", - "tasks.max": "1", - "database.hostname": "${POSTGRES_HOST}", - "database.port": "${POSTGRES_PORT}", - "database.user": "${POSTGRES_USER}", - "database.password": "${POSTGRES_PASSWORD}", - "database.dbname": "${POSTGRES_DATABASE}", - "database.server.name": "dbserver1", - "topic.prefix": "dbserver1", - "schema.include.list": "${POSTGRES_SCHEMA}", - "plugin.name": "pgoutput", - "schema.history.internal.kafka.bootstrap.servers": "localhost:9092", - "schema.history.internal.kafka.topic": "schema-changes.inventory", - "key.converter": "io.confluent.connect.avro.AvroConverter", - "value.converter": "io.confluent.connect.avro.AvroConverter", - "key.converter.schema.registry.url": "http://localhost:8081", - "value.converter.schema.registry.url": "http://localhost:8081", - "slot.name": "dozer_with_registry" - } -} \ No newline at end of file diff --git a/config/tests/test.register-postgres.json b/config/tests/test.register-postgres.json deleted file mode 100644 index 677764a324..0000000000 --- a/config/tests/test.register-postgres.json +++ /dev/null @@ -1,17 +0,0 @@ -{ - "name": "dozer-postgres-connector", - "config": { - "name": "dozer-postgres-connector", - "connector.class": "io.debezium.connector.postgresql.PostgresConnector", - "tasks.max": "1", - "database.hostname": "${POSTGRES_HOST}", - "database.port": "${POSTGRES_PORT}", - "database.user": "${POSTGRES_USER}", - "database.password": "${POSTGRES_PASSWORD}", - "database.dbname": "${POSTGRES_DATABASE}", - "database.server.name": "dbserver1", - "topic.prefix": "dbserver1", - "schema.include.list": "${POSTGRES_SCHEMA}", - "plugin.name": "pgoutput" - } -} \ No newline at end of file diff --git a/config/tests/test.snowflake.yaml b/config/tests/test.snowflake.yaml deleted file mode 100644 index 80860ebad1..0000000000 --- a/config/tests/test.snowflake.yaml +++ /dev/null @@ -1,45 +0,0 @@ -app_name: dozer-snowflake-test -version: 1 - -api: - rest: - port: 8080 - url: "[::0]" - cors: true - grpc: - port: 50051 - url: "[::0]" - cors: true - web: true - auth: false - internal: - port: 50052 - host: "[::1]" - -connections: - - name: customers - db_type: Snowflake - authentication: !Snowflake - server: "${SERVER}" - port: 443 - user: "${USERNAME}" - password: "${PASSWORD}" - database: "${DATABASE}" - schema: "${SCHEMA}" - warehouse: "${WAREHOUSE}" - driver: SnowflakeDSIIDriver - -sources: - - name: customer - table_name: customer_test_1000 - connection: customers - columns: - - id - -endpoints: - - name: customer - path: /customer - sql: select id from customer; - index: - primary_key: - - id diff --git a/deny.toml b/deny.toml deleted file mode 100644 index 8d3df3337d..0000000000 --- a/deny.toml +++ /dev/null @@ -1,119 +0,0 @@ -[graph] -targets = [ - "x86_64-unknown-linux-gnu", -] -all-features = true -no-default-features = false - -[output] -feature-depth = 1 - -[advisories] -ignore = ["RUSTSEC-2023-0071"] -[licenses] -allow = [ - "MIT", - "Apache-2.0", - "Apache-2.0 WITH LLVM-exception", - "MPL-2.0", - "BSD-2-Clause", - "BSD-3-Clause", - "ISC", - "Unicode-DFS-2016", - "CC0-1.0", - "OpenSSL", - "Zlib" -] -confidence-threshold = 0.8 - -[licenses.private] -ignore = false -registries = [] - -[[licenses.clarify]] -name = "ring" -expression = "MIT AND ISC AND OpenSSL" -license-files = [ - { path = "LICENSE", hash = 0xbd0eed23 } -] - -[bans] -# Lint level for when multiple versions of the same crate are detected -multiple-versions = "deny" -# Lint level for when a crate version requirement is `*` -wildcards = "allow" -# The graph highlighting used when creating dotgraphs for crates -# with multiple versions -# * lowest-version - The path to the lowest versioned duplicate is highlighted -# * simplest-path - The path to the version with the fewest edges is highlighted -# * all - Both lowest-version and simplest-path are used -highlight = "all" -# The default lint level for `default` features for crates that are members of -# the workspace that is being checked. This can be overridden by allowing/denying -# `default` on a crate-by-crate basis if desired. -workspace-default-features = "allow" -# The default lint level for `default` features for external crates that are not -# members of the workspace. This can be overridden by allowing/denying `default` -# on a crate-by-crate basis if desired. -external-default-features = "allow" -# List of crates that are allowed. Use with care! -allow = [] -deny = [] - -# Certain crates/versions that will be skipped when doing duplicate detection. -skip = [ - "typed-builder@0.10.0", - "webpki-roots@0.25.4", - "syn@1.0", - # Required by rdkafka-sys - "toml_edit@0.19", - # Needs updating. See if we can get rid of our fork - "sqlparser@0.35.0", - # dozer-sql-expression. Can't update because of sqlparser dep - "bigdecimal", - "sourcemap@6", - # deno - "libloading@0.7", - # deno pins reqwest 0.11.20, which should be fixed once reqwest 0.12 is released - "idna@0.3" -] -# Similarly to `skip` allows you to skip certain crates during duplicate -# detection. Unlike skip, it also includes the entire tree of transitive -# dependencies starting at the specified crate, up to a certain depth, which is -# by default infinite. -skip-tree = [ - # mongodb uses a couple old versions of crates - "mongodb@2.8.2", - # dozer-tracing is updated in a separate PR - "dozer-tracing", - # snowflake source. Let's get rid of this dependency - "genawaiter@0.99", - # Uses old version of proc macro crates through num_enum. Let's send a PR to update" - "rdkafka", - # Used by sourcemap. Let's send a PR to update - "base64-simd@0.7", -] - -# This section is considered when running `cargo deny check sources`. -# More documentation about the 'sources' section can be found here: -# https://embarkstudios.github.io/cargo-deny/checks/sources/cfg.html -[sources] -# Lint level for what to happen when a crate from a crate registry that is not -# in the allow list is encountered -unknown-registry = "deny" -# Lint level for what to happen when a crate from a git repository that is not -# in the allow list is encountered -unknown-git = "deny" -# List of URLs for allowed crate registries. Defaults to the crates.io index -# if not specified. If it is specified but empty, no registries are allowed. -allow-registry = ["https://github.com/rust-lang/crates.io-index"] -# List of URLs for allowed Git repositories -allow-git = ["https://github.com/MaterializeInc/rust-postgres"] - -[sources.allow-org] -# 1 or more github.com organizations to allow git sources for -github = ["getdozer"] -# 1 or more gitlab.com organizations to allow git sources for -gitlab = [] -# 1 or more bitbucket.org organizations to allow git sources for -bitbucket = [] diff --git a/docker/snowflake.Dockerfile b/docker/snowflake.Dockerfile deleted file mode 100644 index 1b545f2494..0000000000 --- a/docker/snowflake.Dockerfile +++ /dev/null @@ -1,36 +0,0 @@ -FROM rust:latest as builder -WORKDIR "/usr/dozer" -RUN curl -LO https://github.com/protocolbuffers/protobuf/releases/download/v21.9/protoc-21.9-linux-aarch_64.zip -RUN unzip protoc-21.9-linux-aarch_64.zip -d $HOME/.local -ENV PATH="$PATH:$HOME/.local/bin" -RUN apt-get update && apt-get install -y \ - build-essential \ - make \ - g++ \ - libclang-dev \ - devscripts \ - debhelper \ - build-essential \ - libssl-dev \ - pkg-config \ - unixodbc-dev - -ENV PATH="$PATH:/root/.local/bin" -RUN protoc --version -COPY . . -RUN cargo build --release --bin dozer --features snowflake - - - -FROM rust:latest as runtime -WORKDIR "/usr/dozer" -RUN apt-get update && apt-get install -y unixodbc-dev unixodbc -RUN curl -LO https://sfc-repo.snowflakecomputing.com/odbc/linuxaarch64/2.25.6/snowflake-odbc-2.25.6.aarch64.deb -RUN dpkg -i snowflake-odbc-2.25.6.aarch64.deb -COPY --from=builder /usr/dozer/target/release/dozer /usr/local/bin -COPY --from=builder /usr/dozer/config/log4rs.release.yaml /usr/dozer -RUN cp /usr/dozer/log4rs.release.yaml /usr/local/bin/log4rs.yaml -RUN cp /usr/dozer/log4rs.release.yaml /usr/dozer/log4rs.yaml -COPY --from=builder /usr/dozer/tests/connectors/snowflake/dozer-config.yaml /usr/dozer -ENTRYPOINT ["/usr/local/bin/dozer"] -EXPOSE 8080 \ No newline at end of file diff --git a/dozer-cli/Cargo.toml b/dozer-cli/Cargo.toml deleted file mode 100644 index cb0519af7e..0000000000 --- a/dozer-cli/Cargo.toml +++ /dev/null @@ -1,67 +0,0 @@ -[package] -name = "dozer-cli" -version = "0.4.0" -edition = "2021" -default-run = "dozer" -authors = ["getdozer/dozer-dev"] -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html -[package.metadata.deb] -name = "dozer" - -[dependencies] -dozer-ingestion = { path = "../dozer-ingestion" } -dozer-core = { path = "../dozer-core" } -dozer-sql = { path = "../dozer-sql" } -dozer-types = { path = "../dozer-types" } -dozer-tracing = { path = "../dozer-tracing" } -dozer-sink-clickhouse = { path = "../dozer-sink-clickhouse" } -actix-web = "4.4.0" -async-trait = "0.1.74" -uuid = { version = "1.6.1", features = ["v4", "serde"] } -tokio = { version = "1", features = ["full"] } -tempfile = "3.10.1" -clap = { version = "4.4.1", features = ["derive"] } -prost-reflect = { version = "0.12.0", features = ["serde", "text-format"] } -tonic = { version = "0.11.0", features = ["tls", "tls-roots"] } -tonic-reflection = "0.11.0" -tonic-web = "0.11.0" -tokio-stream = "0.1.12" -include_dir = "0.7.3" -handlebars = "4.4.0" -rustyline = "13.0.0" -rustyline-derive = "0.10.0" -futures = "0.3.28" -page_size = "0.6.0" -reqwest = { version = "0.11.20", features = [ - "rustls-tls", - "cookies", - "json", -], default-features = false } -glob = "0.3.1" -tower = "0.4.13" -tower-http = { version = "0.4", features = ["full"] } -zip = { version = "0.6.6", default-features = false, features = ["deflate"] } -notify = "6.0.1" -notify-debouncer-full = "0.2.0" -webbrowser = "0.8.12" -actix-files = "0.6.2" -prometheus-parse = "0.2.4" -camino = "1.1.6" - -[build-dependencies] -dozer-types = { path = "../dozer-types" } - -[[bin]] -edition = "2021" -name = "dozer" -path = "src/main.rs" - -[features] -snowflake = ["dozer-ingestion/snowflake"] -mongodb = ["dozer-ingestion/mongodb"] -onnx = ["dozer-sql/onnx"] -tokio-console = ["dozer-tracing/tokio-console"] -javascript = ["dozer-ingestion/javascript", "dozer-sql/javascript"] -datafusion = ["dozer-ingestion/datafusion"] diff --git a/dozer-cli/build.rs b/dozer-cli/build.rs deleted file mode 100644 index a0e4b4202c..0000000000 --- a/dozer-cli/build.rs +++ /dev/null @@ -1,22 +0,0 @@ -use std::fs::File; -use std::io::Write; -use std::path::Path; - -fn main() { - let schema_path = Path::new("../json_schemas"); - // Define the path to the file we want to create or overwrite - let connection_path = schema_path.join("connections.json"); - let dozer_path = schema_path.join("dozer.json"); - - let mut file = File::create(connection_path).expect("Failed to create connections.json"); - let schemas = dozer_types::models::get_connection_schemas().unwrap(); - write!(file, "{}", schemas).expect("Unable to write file"); - - let mut dozer_schema_file = File::create(dozer_path).expect("Failed to create dozer.json"); - let schema = dozer_types::models::get_dozer_schema().unwrap(); - write!(dozer_schema_file, "{}", schema).expect("Unable to write file"); - - // Print a message to indicate the file has been written - println!("cargo:rerun-if-changed=build.rs"); - println!("Written to {:?}", schema_path.display()); -} diff --git a/dozer-cli/src/cli/helper.rs b/dozer-cli/src/cli/helper.rs deleted file mode 100644 index d2a3590813..0000000000 --- a/dozer-cli/src/cli/helper.rs +++ /dev/null @@ -1,270 +0,0 @@ -use crate::config_helper::combine_config; -use crate::errors::CliError; -use crate::errors::CliError::{ConfigurationFilePathNotProvided, FailedToFindConfigurationFiles}; -use crate::errors::ConfigCombineError::CannotReadConfig; -use crate::errors::OrchestrationError; -use crate::simple::SimpleOrchestrator as Dozer; - -use camino::Utf8PathBuf; -use dozer_tracing::DozerMonitorContext; -use dozer_types::prettytable::{row, Table}; -use dozer_types::serde_json; -use dozer_types::tracing::info; -use dozer_types::{models::config::Config, serde_yaml}; -use handlebars::Handlebars; -use std::collections::{BTreeMap, HashSet}; -use std::env; -use std::io::{self, stdin, IsTerminal, Read}; -use std::sync::Arc; -use tokio::runtime::Runtime; - -pub async fn init_config( - config_paths: Vec, - config_token: Option, - config_overrides: Vec<(String, serde_json::Value)>, - ignore_pipe: bool, -) -> Result<(Config, Vec), CliError> { - let (mut config, loaded_files) = load_config(config_paths, config_token, ignore_pipe).await?; - - config = apply_overrides(&config, config_overrides)?; - - Ok((config, loaded_files)) -} - -pub fn get_base_dir() -> Result { - let base_directory = std::env::current_dir().map_err(CliError::Io)?; - - Utf8PathBuf::try_from(base_directory).map_err(|e| CliError::Io(e.into_io_error())) -} - -pub fn init_dozer( - runtime: Arc, - config: Config, - labels: DozerMonitorContext, -) -> Result { - let base_directory = get_base_dir()?; - Ok(Dozer::new(base_directory, config, runtime, labels)) -} - -pub async fn list_sources( - runtime: Arc, - config_paths: Vec, - config_token: Option, - config_overrides: Vec<(String, serde_json::Value)>, - ignore_pipe: bool, - filter: Option, -) -> Result<(), OrchestrationError> { - let (config, loaded_files) = - init_config(config_paths, config_token, config_overrides, ignore_pipe).await?; - info!("Loaded config from: {}", loaded_files.join(", ")); - let source_connections: HashSet = config - .sources - .iter() - .map(|source| source.connection.clone()) - .collect(); - let dozer = init_dozer(runtime, config, Default::default())?; - let connection_map = dozer.list_connectors(source_connections).await?; - let mut table_parent = Table::new(); - for (connection_name, (tables, schemas)) in connection_map { - let mut first_table_found = false; - - for (table, schema) in tables.into_iter().zip(schemas) { - let name = table.schema.map_or(table.name.clone(), |schema_name| { - format!("{schema_name}.{}", table.name) - }); - - if filter - .as_ref() - .map_or(true, |name_part| name.contains(name_part)) - { - if !first_table_found { - table_parent.add_row(row!["Connection", "Table", "Columns"]); - first_table_found = true; - } - let schema_table = schema.schema.print(); - - table_parent.add_row(row![connection_name, name, schema_table]); - } - } - - if first_table_found { - table_parent.add_empty_row(); - } - } - table_parent.printstd(); - Ok(()) -} - -async fn load_config( - config_url_or_paths: Vec, - config_token: Option, - ignore_pipe: bool, -) -> Result<(Config, Vec), CliError> { - let read_stdin = !stdin().is_terminal() && !ignore_pipe; - let first_config_path = config_url_or_paths.first(); - match first_config_path { - None => Err(ConfigurationFilePathNotProvided), - Some(path) => { - if path.starts_with("https://") || path.starts_with("http://") { - Ok(( - load_config_from_http_url(path, config_token).await?, - vec![path.to_owned()], - )) - } else { - load_config_from_file(config_url_or_paths, read_stdin) - } - } - } -} - -async fn load_config_from_http_url( - config_url: &str, - config_token: Option, -) -> Result { - let client = reqwest::Client::new(); - let mut get_request = client.get(config_url); - if let Some(token) = config_token { - get_request = get_request.bearer_auth(token); - } - let response: reqwest::Response = get_request.send().await?.error_for_status()?; - let contents = response.text().await?; - parse_config(&contents) -} - -pub fn load_config_from_file( - config_path: Vec, - read_stdin: bool, -) -> Result<(Config, Vec), CliError> { - let stdin_path = ""; - let input = if read_stdin { - let mut input = String::new(); - io::stdin() - .read_to_string(&mut input) - .map_err(|e| CannotReadConfig(stdin_path.into(), e))?; - Some(input) - } else { - None - }; - - let mut loaded_files = Vec::new(); - if input.is_some() { - loaded_files.push(stdin_path.to_owned()); - } - - let (config_template, files) = combine_config(config_path.clone(), input)?; - loaded_files.extend_from_slice(&files); - let current_directory = env::current_dir().unwrap(); - let config_files_with_path: Vec<_> = loaded_files - .iter() - .map(|file| current_directory.join(file).to_string_lossy().to_string()) - .collect(); - - match config_template { - Some(template) => Ok((parse_config(&template)?, config_files_with_path)), - None => Err(FailedToFindConfigurationFiles(config_path.join(", "))), - } -} - -fn parse_config(config_template: &str) -> Result { - let mut handlebars = Handlebars::new(); - handlebars - .register_template_string("config", config_template) - .map_err(|e| CliError::FailedToParseYaml(Box::new(e)))?; - - let mut data = BTreeMap::new(); - - for (key, value) in std::env::vars() { - data.insert(key, value); - } - - let config_str = handlebars - .render("config", &data) - .map_err(|e| CliError::FailedToParseYaml(Box::new(e)))?; - - let config: Config = serde_yaml::from_str(&config_str) - .map_err(|e: serde_yaml::Error| CliError::FailedToParseYaml(Box::new(e)))?; - - Ok(config) -} - -/// Convert `config` to JSON, apply JSON pointer overrides, then convert back to `Config`. -fn apply_overrides( - config: &Config, - config_overrides: Vec<(String, serde_json::Value)>, -) -> Result { - let mut config_json = serde_json::to_value(config).map_err(CliError::SerializeConfigToJson)?; - - for (pointer, value) in config_overrides { - if let Some(pointee) = config_json.pointer_mut(&pointer) { - *pointee = value; - } else { - return Err(CliError::MissingConfigOverride(pointer)); - } - } - - // Directly convert `config_json` to `Config` fails, not sure why. - let config_json_string = - serde_json::to_string(&config_json).map_err(CliError::SerializeConfigToJson)?; - let config: Config = - serde_json::from_str(&config_json_string).map_err(CliError::DeserializeConfigFromJson)?; - - Ok(config) -} - -pub const LOGO: &str = r" -.____ ___ __________ ____ -| _ \ / _ \__ / ____| _ \ -| | | | | | |/ /| _| | |_) | -| |_| | |_| / /_| |___| _ < -|____/ \___/____|_____|_| \_\ -"; - -pub const DESCRIPTION: &str = r#"Open-source platform to build, publish and manage blazing-fast real-time data APIs in minutes. - - If no sub commands are passed, dozer will bring up both app and api services. -"#; - -#[cfg(test)] -mod tests { - use dozer_types::models::{api_config::ApiConfig, api_security::ApiSecurity}; - - use super::*; - - #[test] - fn test_override_top_level() { - let mut config = Config { - app_name: "test_override_top_level".to_string(), - ..Default::default() - }; - config.sql = Some("sql1".to_string()); - let sql = "sql2".to_string(); - let config = apply_overrides( - &config, - vec![("/sql".to_string(), serde_json::to_value(&sql).unwrap())], - ) - .unwrap(); - assert_eq!(config.sql.unwrap(), sql); - } - - #[test] - fn test_override_nested() { - let mut config = Config { - app_name: "test_override_nested".to_string(), - ..Default::default() - }; - config.api = ApiConfig { - api_security: Some(ApiSecurity::Jwt("secret1".to_string())), - ..Default::default() - }; - let api_security = ApiSecurity::Jwt("secret2".to_string()); - let config = apply_overrides( - &config, - vec![( - "/api/api_security".to_string(), - serde_json::to_value(&api_security).unwrap(), - )], - ) - .unwrap(); - assert_eq!(config.api.api_security.unwrap(), api_security); - } -} diff --git a/dozer-cli/src/cli/init.rs b/dozer-cli/src/cli/init.rs deleted file mode 100644 index 85cda91385..0000000000 --- a/dozer-cli/src/cli/init.rs +++ /dev/null @@ -1,287 +0,0 @@ -use crate::errors::{CliError, OrchestrationError}; -use dozer_types::constants::{DEFAULT_LAMBDAS_DIRECTORY, DEFAULT_QUERIES_DIRECTORY}; -use dozer_types::log::warn; -use dozer_types::models::config::default_home_dir; -use dozer_types::{ - constants::DEFAULT_CONFIG_PATH, - log::info, - models::ingestion_types::{ - EthConfig, EthFilter, EthLogConfig, EthProviderConfig, MongodbConfig, MySQLConfig, - S3Details, S3Storage, SnowflakeConfig, - }, - models::{ - config::Config, - connection::{Connection, ConnectionConfig, PostgresConfig}, - }, - serde_yaml, -}; -use rustyline::history::DefaultHistory; -use rustyline::{ - completion::{Completer, Pair}, - Context, -}; -use rustyline::{error::ReadlineError, Editor}; -use rustyline_derive::{Helper, Highlighter, Hinter, Validator}; -use std::path::{Path, PathBuf}; - -#[derive(Helper, Highlighter, Hinter, Validator)] -pub struct InitHelper {} - -impl Completer for InitHelper { - type Candidate = Pair; - fn complete( - &self, - line: &str, - _pos: usize, - _ctx: &Context, - ) -> rustyline::Result<(usize, Vec)> { - let line = format!("{line}_"); - let mut tokens = line.split_whitespace(); - let mut last_token = String::from(tokens.next_back().unwrap()); - last_token.pop(); - let candidates: Vec = vec![ - "Postgres".to_owned(), - "Ethereum".to_owned(), - "Snowflake".to_owned(), - "MySQL".to_owned(), - "S3".to_owned(), - "MongoDB".to_owned(), - ]; - let mut match_pair: Vec = candidates - .iter() - .filter_map(|f| { - if f.to_lowercase().starts_with(&last_token.to_lowercase()) { - Some(Pair { - display: f.to_owned(), - replacement: f.to_owned(), - }) - } else { - None - } - }) - .collect(); - if match_pair.is_empty() { - match_pair = vec![Pair { - display: "Postgres".to_owned(), - replacement: "Postgres".to_owned(), - }] - } - Ok((line.len() - last_token.len() - 1, match_pair)) - } -} - -pub fn generate_connection(connection_name: &str) -> Connection { - match connection_name { - "Snowflake" | "snowflake" | "S" | "s" => { - let snowflake_config = SnowflakeConfig { - server: "..snowflakecomputing.com".to_owned(), - port: "443".to_owned(), - user: "bob".to_owned(), - password: "password".to_owned(), - database: "database".to_owned(), - schema: "schema".to_owned(), - warehouse: "warehouse".to_owned(), - driver: Some("SnowflakeDSIIDriver".to_owned()), - role: "role".to_owned(), - poll_interval_seconds: None, - }; - let connection: Connection = Connection { - name: "snowflake".to_owned(), - config: ConnectionConfig::Snowflake(snowflake_config), - }; - connection - } - "Ethereum" | "ethereum" | "E" | "e" => { - let eth_filter = EthFilter { - from_block: Some(0), - to_block: None, - addresses: vec![], - topics: vec![], - }; - let ethereum_config = EthConfig { - provider: EthProviderConfig::Log(EthLogConfig { - wss_url: "wss://link".to_owned(), - filter: Some(eth_filter), - contracts: vec![], - }), - }; - let connection: Connection = Connection { - name: "ethereum".to_owned(), - config: ConnectionConfig::Ethereum(ethereum_config), - }; - connection - } - "MySQL" | "MYSQL" | "mysql" | "Mysql" | "My" => { - let mysql_config = MySQLConfig { - url: "mysql://:@localhost:3306/".to_owned(), - server_id: Some((1).to_owned()), - }; - let connection: Connection = Connection { - name: "mysql".to_owned(), - config: ConnectionConfig::MySQL(mysql_config), - }; - connection - } - "S3" | "s3" => { - let s3_details = S3Details { - access_key_id: "".to_owned(), - secret_access_key: "".to_owned(), - region: "".to_owned(), - bucket_name: "".to_owned(), - }; - let s3_config = S3Storage { - details: s3_details, - tables: vec![], - }; - let connection: Connection = Connection { - name: "s3".to_owned(), - config: ConnectionConfig::S3Storage(s3_config), - }; - connection - } - "MongoDB" | "mongodb" | "MONGODB" | "Mongodb" | "Mo" | "MO" => { - let mongo_config = MongodbConfig { - connection_string: - "mongodb://:@localhost:27017/".to_owned(), - }; - let connection: Connection = Connection { - name: "mongodb".to_owned(), - config: ConnectionConfig::MongoDB(mongo_config), - }; - connection - } - _ => { - let postgres_config = PostgresConfig { - user: Some("postgres".to_owned()), - password: Some("postgres".to_owned()), - host: Some("localhost".to_owned()), - port: Some(5432), - database: Some("users".to_owned()), - sslmode: None, - connection_url: None, - schema: None, - batch_size: None, - }; - let connection: Connection = Connection { - name: "postgres".to_owned(), - config: ConnectionConfig::Postgres(postgres_config), - }; - connection - } - } -} -type Question = ( - String, - Box Result<(), OrchestrationError>>, -); -pub fn generate_config_repl() -> Result<(), OrchestrationError> { - let mut rl = Editor::::new() - .map_err(|e| OrchestrationError::CliError(CliError::ReadlineError(e)))?; - rl.set_helper(Some(InitHelper {})); - let mut default_config = Config { - version: 1, - ..Default::default() - }; - let default_app_name = "quick-start-app"; - let questions: Vec = vec![ - ( - format!("question: App name ({:}): ", default_app_name), - Box::new(move |(app_name, config)| { - let app_name = app_name.trim(); - if app_name.is_empty() { - config.app_name = default_app_name.to_string(); - } else { - config.app_name = app_name.to_string(); - } - Ok(()) - }), - ), - ( - format!("question: Data directory ({:}): ", default_home_dir()), - Box::new(move |(home_dir, config)| { - if home_dir.is_empty() { - config.home_dir = Some(default_home_dir()); - } else { - config.home_dir = Some(home_dir); - } - Ok(()) - }), - ), - ( - "question: Connection Type - one of: [P]ostgres, [E]thereum, [S]nowflake, [My]SQL, [S3]Storage, [Mo]ngoDB: " - .to_string(), - Box::new(move |(connection, config)| { - let sample_connection = generate_connection(&connection); - config.connections.push(sample_connection); - - Ok(()) - }), - ), - ( - format!("question: Config path ({:}): ", DEFAULT_CONFIG_PATH), - Box::new(move |(yaml_path, config)| { - let mut yaml_path = yaml_path.trim(); - if yaml_path.is_empty() { - yaml_path = DEFAULT_CONFIG_PATH; - } - let f = std::fs::OpenOptions::new() - .create(true) - .write(true) - .truncate(false) - .open(yaml_path) - .map_err(|e| { - OrchestrationError::CliError(CliError::FileSystem( - yaml_path.to_string().into(), - e, - )) - })?; - serde_yaml::to_writer(f, &config) - .map_err(OrchestrationError::FailedToWriteConfigYaml)?; - - info!("Generating workspace: \n\ - \n- {} (main configuration)\n- ./queries (folder for sql queries)\n- ./lambdas (folder for lambda functions) - \n• More details about our config: https://getdozer.io/docs/reference/configuration/introduction\ - \n• Connector & Sources: https://getdozer.io/docs/reference/configuration/connectors\ - \n• Endpoints: https://getdozer.io/docs/reference/configuration/endpoints/", - yaml_path.to_owned()); - - let path = PathBuf::from(yaml_path); - if let Some(dir) = path.parent() { - let queries_path = Path::new(dir).join(DEFAULT_QUERIES_DIRECTORY); - if let Err(_e) = std::fs::create_dir(queries_path) { - warn!("Cannot create queries directory"); - } - - let lambdas_path = Path::new(dir).join(DEFAULT_LAMBDAS_DIRECTORY); - if let Err(_e) = std::fs::create_dir(lambdas_path) { - warn!("Cannot create lambdas directory"); - } - } - - Ok(()) - }), - ), - ]; - let result = questions.iter().try_for_each(|(question, func)| { - let readline = rl.readline(question); - match readline { - Ok(input) => func((input, &mut default_config)), - Err(err) => Err(OrchestrationError::CliError(CliError::ReadlineError(err))), - } - }); - - match result { - Ok(_) => Ok(()), - Err(e) => match e { - OrchestrationError::CliError(CliError::ReadlineError(ReadlineError::Interrupted)) => { - info!("Exiting.."); - Ok(()) - } - OrchestrationError::CliError(CliError::ReadlineError(ReadlineError::Eof)) => { - info!("CTRL-D - exiting..."); - Ok(()) - } - _ => Err(e), - }, - } -} diff --git a/dozer-cli/src/cli/mod.rs b/dozer-cli/src/cli/mod.rs deleted file mode 100644 index 2af1734577..0000000000 --- a/dozer-cli/src/cli/mod.rs +++ /dev/null @@ -1,7 +0,0 @@ -mod helper; -mod init; -pub mod types; -pub use helper::{ - get_base_dir, init_config, init_dozer, list_sources, load_config_from_file, LOGO, -}; -pub use init::{generate_config_repl, generate_connection}; diff --git a/dozer-cli/src/cli/tests.rs b/dozer-cli/src/cli/tests.rs deleted file mode 100644 index c05cdf4834..0000000000 --- a/dozer-cli/src/cli/tests.rs +++ /dev/null @@ -1,307 +0,0 @@ -use super::Config; -use dozer_types::{ - constants::DEFAULT_HOME_DIR, - models::{ - api_config::{ - default_api_config, ApiConfig, ApiGrpc, ApiInternal, ApiPipelineInternal, ApiRest, - }, - api_endpoint::ApiEndpoint, - app_config::Flags, - connection::{Authentication, Connection, PostgresAuthentication}, - source::{RefreshConfig, Source}, - }, - serde_yaml, -}; -fn test_yml_content_full() -> (&'static str, Config) { - let test_connection = test_connection(); - let test_source = test_source(test_connection.to_owned()); - let api_endpoint = test_api_endpoint(); - let api_config = test_api_config(); - let flags = Some(Flags { - dynamic: true, - grpc_web: true, - push_events: false, - }); - let config = Config { - app_name: "dozer-config-sample".to_owned(), - version: 1, - home_dir: DEFAULT_HOME_DIR.to_owned(), - api: Some(api_config), - connections: vec![test_connection], - sources: vec![test_source], - endpoints: vec![api_endpoint], - flags: flags, - ..Default::default() - }; - ( - r#" - app_name: dozer-config-sample - version: 1, - home_dir: './.dozer' - api: - rest: - port: 8080 - host: "[::0]" - cors: true - grpc: - port: 50051 - host: "[::0]" - cors: true - web: true - auth: false - api_internal: - port: 50052 - host: '[::1]' - home_dir: './.dozer/api' - pipeline_internal: - port: 50053 - host: '[::1]' - home_dir: './.dozer/pipeline' - flags: - grpc_web: true - dynamic: true - push_events: false - connections: - - db_type: Postgres - authentication: !Postgres - user: postgres - password: postgres - host: localhost - port: 5432 - database: users - name: users - sources: - - name: users - table_name: users - columns: - - id - - email - - phone - connection: users - endpoints: - - id: null - name: users - path: /users - sql: select id, email, phone from users where 1=1; - index: - primary_key: - - id - "#, - config, - ) -} -fn test_yml_content_missing_api_config() -> &'static str { - r#" - app_name: dozer-config-sample - version: 1, - connections: - - db_type: Postgres - authentication: !Postgres - user: postgres - password: postgres - host: localhost - port: 5432 - database: users - name: users - sources: - - name: users - table_name: users - columns: - - id - - email - - phone - connection: users - endpoints: - - id: null - name: users - path: /users - sql: select id, email, phone from users where 1=1; - index: - primary_key: - - id - "# -} -fn test_yml_content_missing_internal_config() -> &'static str { - r#" - app_name: dozer-config-sample - version: 1, - api: - rest: - port: 8080 - host: "[::0]" - cors: true - grpc: - port: 50051 - host: "[::0]" - cors: true - web: true - auth: false - connections: - - db_type: Postgres - authentication: !Postgres - user: postgres - password: postgres - host: localhost - port: 5432 - database: users - name: users - sources: - - name: users - table_name: users - columns: - - id - - email - - phone - connection: users - endpoints: - - id: null - name: users - path: /users - sql: select id, email, phone from users where 1=1; - index: - primary_key: - - id - "# -} - -fn test_connection() -> Connection { - Connection { - authentication: Some(Authentication::Postgres(PostgresAuthentication { - user: "postgres".to_owned(), - password: "postgres".to_owned(), - host: "localhost".to_owned(), - port: 5432, - database: "users".to_owned(), - })), - db_type: dozer_types::models::connection::DBType::Postgres as i32, - name: "users".to_owned(), - ..Default::default() - } -} -fn test_api_endpoint() -> ApiEndpoint { - ApiEndpoint { - name: "users".to_owned(), - path: "/users".to_owned(), - sql: "select id, email, phone from users where 1=1;".to_owned(), - index: Some(dozer_types::models::api_endpoint::ApiIndex { - primary_key: vec!["id".to_owned()], - }), - ..Default::default() - } -} -fn test_source(connection: Connection) -> Source { - Source { - id: None, - name: "users".to_owned(), - table_name: "users".to_owned(), - columns: vec!["id".to_owned(), "email".to_owned(), "phone".to_owned()], - connection: Some(connection), - refresh_config: Some(RefreshConfig::default()), - ..Default::default() - } -} -fn test_api_config() -> ApiConfig { - ApiConfig { - rest: Some(ApiRest { - port: 8080, - host: "[::0]".to_owned(), - cors: true, - }), - grpc: Some(ApiGrpc { - port: 50051, - host: "[::0]".to_owned(), - cors: true, - web: true, - }), - auth: false, - api_internal: Some(ApiInternal { - home_dir: format!("{:}/api", DEFAULT_HOME_DIR.to_owned()), - }), - pipeline_internal: Some(ApiPipelineInternal { - port: 50053, - host: "[::1]".to_owned(), - home_dir: format!("{:}/pipeline", DEFAULT_HOME_DIR.to_owned()), - }), - ..Default::default() - } -} -fn test_config() -> Config { - let test_connection = test_connection(); - let test_source = test_source(test_connection.to_owned()); - let api_endpoint = test_api_endpoint(); - let api_config = test_api_config(); - Config { - app_name: "dozer-config-sample".to_owned(), - version: 1, - home_dir: DEFAULT_HOME_DIR.to_owned(), - api: Some(api_config), - connections: vec![test_connection], - sources: vec![test_source], - endpoints: vec![api_endpoint], - ..Default::default() - } -} -#[test] -fn test_deserialize_config() { - let test_full = test_yml_content_full(); - let test_full_str = test_full.0; - let deserializer_result = serde_yaml::from_str::(test_full_str).unwrap(); - let expected = test_full.1; - assert_eq!(deserializer_result.api, expected.api); - assert_eq!(deserializer_result.app_name, expected.app_name); - assert_eq!(deserializer_result.connections, expected.connections); - assert_eq!(deserializer_result.endpoints, expected.endpoints); - assert_eq!(deserializer_result.sources, expected.sources); - assert_eq!(deserializer_result, expected); -} -#[test] -fn test_deserialize_default_api_config() { - let test_str = test_yml_content_missing_api_config(); - let deserializer_result = serde_yaml::from_str::(test_str).unwrap(); - let expected = test_config(); - let default_api_config = default_api_config(); - assert_eq!(deserializer_result.api, Some(default_api_config)); - assert_eq!(deserializer_result.app_name, expected.app_name); - assert_eq!(deserializer_result.connections, expected.connections); - assert_eq!(deserializer_result.endpoints, expected.endpoints); - assert_eq!(deserializer_result.sources, expected.sources); - assert_eq!(deserializer_result, expected); -} -#[test] -fn test_deserialize_yaml_missing_internal_config() { - let test_str = test_yml_content_missing_internal_config(); - let deserializer_result = serde_yaml::from_str::(test_str).unwrap(); - let expected = test_config(); - let default_api_config = default_api_config(); - assert_eq!(deserializer_result.api, Some(default_api_config.to_owned())); - assert_eq!( - deserializer_result - .api - .to_owned() - .unwrap() - .api_internal - .unwrap(), - default_api_config.api_internal.unwrap() - ); - assert_eq!( - deserializer_result - .api - .to_owned() - .unwrap() - .pipeline_internal - .unwrap(), - default_api_config.pipeline_internal.unwrap() - ); - assert_eq!(deserializer_result.app_name, expected.app_name); - assert_eq!(deserializer_result.connections, expected.connections); - assert_eq!(deserializer_result.endpoints, expected.endpoints); - assert_eq!(deserializer_result.sources, expected.sources); - assert_eq!(deserializer_result, expected); -} -#[test] -fn test_serialize_config() { - let config = test_config(); - let serialize_yml = serde_yaml::to_string(&config).unwrap(); - let deserializer_result = serde_yaml::from_str::(&serialize_yml).unwrap(); - assert_eq!(deserializer_result, config); -} diff --git a/dozer-cli/src/cli/types.rs b/dozer-cli/src/cli/types.rs deleted file mode 100644 index ce979cbd06..0000000000 --- a/dozer-cli/src/cli/types.rs +++ /dev/null @@ -1,135 +0,0 @@ -use clap::{Args, Parser, Subcommand}; - -use super::helper::{DESCRIPTION, LOGO}; - -use dozer_types::{ - constants::{DEFAULT_CONFIG_PATH_PATTERNS, LOCK_FILE}, - serde_json, -}; - -#[derive(Parser, Debug)] -#[command(author, version, name = "dozer")] -#[command( - about = format!("{} \n {}", LOGO, DESCRIPTION), - long_about = None, -)] -pub struct Cli { - #[arg( - global = true, - short = 'c', - long = "config-path", - default_values = DEFAULT_CONFIG_PATH_PATTERNS - )] - pub config_paths: Vec, - #[arg(global = true, long, hide = true)] - pub config_token: Option, - #[arg(global = true, long = "enable-progress")] - pub enable_progress: bool, - #[arg(global = true, long, value_parser(parse_config_override))] - pub config_overrides: Vec<(String, serde_json::Value)>, - - #[arg(global = true, long = "ignore-pipe")] - pub ignore_pipe: bool, - #[clap(subcommand)] - pub cmd: Commands, -} - -fn parse_config_override( - arg: &str, -) -> Result<(String, serde_json::Value), Box> { - let mut split = arg.split('='); - let pointer = split.next().ok_or("missing json pointer")?; - let value = split.next().ok_or("missing json value")?; - Ok((pointer.to_string(), serde_json::from_str(value)?)) -} - -#[derive(Debug, Subcommand)] -pub enum Commands { - #[command( - about = "Clean home directory", - long_about = "Clean home directory. It removes all data, schemas and other files in app \ - directory" - )] - Clean, - #[command(about = "Build YAML definitions as a dozer pipeline")] - Build(Build), - #[command(about = "Run a replication instance with the provided configuration")] - Run, - #[command(about = "Run UI server")] - UI(UI), -} - -#[derive(Debug, Args)] -pub struct UI { - #[command(subcommand)] - pub command: Option, -} - -#[derive(Debug, Subcommand)] -pub enum UICommands { - #[command( - about = "Updates the latest UI code", - long_about = "Updates the latest UI code" - )] - Update, -} - -#[derive(Debug, Args)] -#[command(args_conflicts_with_subcommands = true)] -pub struct Live { - #[arg(long, hide = true)] - pub disable_live_ui: bool, -} - -#[derive(Debug, Args)] -#[command(args_conflicts_with_subcommands = true)] -pub struct Build { - #[arg(help = format!("Require that {LOCK_FILE} is up-to-date"), long = "locked")] - pub locked: bool, - #[arg(short = 'f')] - pub force: Option>, -} - -#[derive(Debug, Args)] -#[command(args_conflicts_with_subcommands = true)] -pub struct Deploy { - pub target_url: String, - #[arg(short = 'u')] - pub username: Option, - #[arg(short = 'p')] - pub password: Option, -} - -#[cfg(test)] -mod tests { - use dozer_types::serde_json; - - #[test] - fn test_parse_config_override_string() { - let arg = "/app=\"abc\""; - let result = super::parse_config_override(arg).unwrap(); - assert_eq!(result.0, "/app"); - assert_eq!(result.1, serde_json::Value::String("abc".to_string())); - } - - #[test] - fn test_parse_config_override_number() { - let arg = "/app=123"; - let result = super::parse_config_override(arg).unwrap(); - assert_eq!(result.0, "/app"); - assert_eq!(result.1, serde_json::Value::Number(123.into())); - } - - #[test] - fn test_parse_config_override_object() { - let arg = "/app={\"a\": 1}"; - let result = super::parse_config_override(arg).unwrap(); - assert_eq!(result.0, "/app"); - assert_eq!( - result.1, - serde_json::json!({ - "a": 1 - }) - ); - } -} diff --git a/dozer-cli/src/config_helper.rs b/dozer-cli/src/config_helper.rs deleted file mode 100644 index 832dcfab14..0000000000 --- a/dozer-cli/src/config_helper.rs +++ /dev/null @@ -1,139 +0,0 @@ -use crate::errors::ConfigCombineError; -use crate::errors::ConfigCombineError::{ - CannotReadConfig, CannotReadFile, CannotSerializeToString, SqlIsNotStringType, - WrongPatternOfConfigFilesGlob, -}; -use dozer_types::log::warn; -use dozer_types::serde_yaml; -use dozer_types::serde_yaml::mapping::Entry; -use dozer_types::serde_yaml::{Mapping, Value}; -use glob::glob; - -pub fn combine_config( - config_paths: Vec, - stdin_yaml: Option, -) -> Result<(Option, Vec), ConfigCombineError> { - let mut combined_yaml = serde_yaml::Value::Mapping(Mapping::new()); - - let mut loaded_files = Vec::new(); - let mut config_found = false; - for pattern in config_paths { - let files_glob = glob(&pattern).map_err(WrongPatternOfConfigFilesGlob)?; - - for entry in files_glob { - let path = entry.map_err(CannotReadFile)?; - match path.clone().to_str() { - None => { - warn!("[Config] Path {:?} is not valid", path) - } - Some(name) => { - let content = std::fs::read(path.clone()) - .map_err(|e| CannotReadConfig(path.clone(), e))?; - - if name.contains(".yml") || name.contains(".yaml") { - config_found = true; - } - add_file_content_to_config(&mut combined_yaml, name, content)?; - loaded_files.push(name.to_owned()); - } - } - } - } - let stdin_name = "Stdin"; - // Merge stdin_yaml content into combined_yaml if provided - if let Some(stdin_content) = stdin_yaml { - let stdin_yaml: serde_yaml::Value = serde_yaml::from_str(&stdin_content) - .map_err(|e| ConfigCombineError::ParseYaml(stdin_name.to_string(), e))?; //deserialise yaml content from stdin - merge_yaml(stdin_yaml, &mut combined_yaml)?; //merge with yaml from config-paths - } - - if config_found { - // `serde_yaml::from_value` will return deserialization error, not sure why. - let string = serde_yaml::to_string(&combined_yaml).map_err(CannotSerializeToString)?; - Ok((Some(string), loaded_files)) - } else { - Ok((None, vec![])) - } -} - -pub fn add_file_content_to_config( - combined_yaml: &mut serde_yaml::Value, - name: &str, - content: Vec, -) -> Result<(), ConfigCombineError> { - if name.contains(".yml") || name.contains(".yaml") { - let content_string = String::from_utf8(content)?; - let yaml: serde_yaml::Value = serde_yaml::from_str(&content_string) - .map_err(|e| ConfigCombineError::ParseYaml(name.to_string(), e))?; - merge_yaml(yaml, combined_yaml)?; - } else if name.contains(".sql") { - let mapping = combined_yaml.as_mapping_mut().expect("Should be mapping"); - let sql = mapping.get_mut(serde_yaml::Value::String("sql".into())); - - let content_string = String::from_utf8(content)?; - - match sql { - None => { - mapping.insert( - serde_yaml::Value::String("sql".into()), - serde_yaml::Value::String(content_string), - ); - } - Some(s) => { - let query = s.as_str(); - *s = match query { - None => { - return Err(SqlIsNotStringType); - } - Some(current_query) => { - Value::String(format!("{};{}", current_query, content_string.as_str())) - } - } - } - } - } else { - warn!("Config file \"{name}\" extension not supported"); - } - - Ok(()) -} - -pub fn merge_yaml( - from: serde_yaml::Value, - to: &mut serde_yaml::Value, -) -> Result<(), ConfigCombineError> { - match (from, to) { - (serde_yaml::Value::Mapping(from), serde_yaml::Value::Mapping(to)) => { - for (key, value) in from { - match to.entry(key) { - Entry::Occupied(mut entry) => { - merge_yaml(value, entry.get_mut())?; - } - Entry::Vacant(entry) => { - entry.insert(value); - } - } - } - Ok(()) - } - (serde_yaml::Value::Sequence(from), serde_yaml::Value::Sequence(to)) => { - for value in from { - to.push(value); - } - Ok(()) - } - (serde_yaml::Value::Tagged(from), serde_yaml::Value::Tagged(to)) => { - if from.tag != to.tag { - return Err(ConfigCombineError::CannotMerge { - from: serde_yaml::Value::Tagged(from), - to: serde_yaml::Value::Tagged(to.clone()), - }); - } - merge_yaml(from.value, &mut to.value) - } - (from, to) => Err(ConfigCombineError::CannotMerge { - from, - to: to.clone(), - }), - } -} diff --git a/dozer-cli/src/console_helper.rs b/dozer-cli/src/console_helper.rs deleted file mode 100644 index 9f39f05a4f..0000000000 --- a/dozer-cli/src/console_helper.rs +++ /dev/null @@ -1,14 +0,0 @@ -pub fn get_colored_text(text: &str, color_code: &str) -> String { - let mut result = String::from("\u{1b}["); - result.push_str(color_code); - result.push_str(";1m"); // End token. - result.push_str(text); - result.push_str("\u{1b}[0m"); - result -} - -pub const RED: &str = "31"; -pub const GREEN: &str = "32"; -pub const PURPLE: &str = "35"; -pub const YELLOW: &str = "136"; -pub const GREY: &str = "250"; diff --git a/dozer-cli/src/errors.rs b/dozer-cli/src/errors.rs deleted file mode 100644 index ecca284d63..0000000000 --- a/dozer-cli/src/errors.rs +++ /dev/null @@ -1,282 +0,0 @@ -#![allow(clippy::enum_variant_names)] - -use glob::{GlobError, PatternError}; -use std::io; -use std::path::PathBuf; -use std::string::FromUtf8Error; -use tonic::Code; -use tonic::Code::NotFound; - -use crate::{ - errors::CloudError::{ApplicationNotFound, CloudServiceError}, - ui::app::AppUIError, -}; - -use dozer_core::errors::ExecutionError; -use dozer_sql::errors::PipelineError; -use dozer_types::{constants::LOCK_FILE, thiserror::Error}; -use dozer_types::{errors::internal::BoxedError, serde_json}; -use dozer_types::{serde_yaml, thiserror}; - -use crate::pipeline::connector_source::ConnectorSourceFactoryError; - -pub fn map_tonic_error(e: tonic::Status) -> CloudError { - if e.code() == NotFound && e.message() == "Failed to find app" { - ApplicationNotFound - } else { - CloudServiceError(e) - } -} - -#[derive(Error, Debug)] -pub enum OrchestrationError { - #[error("Failed to write config yaml: {0:?}")] - FailedToWriteConfigYaml(#[source] serde_yaml::Error), - #[error("File system error {0:?}: {1}")] - FileSystem(PathBuf, std::io::Error), - #[error("Failed to find any build")] - NoBuildFound, - #[error("Failed to login: {0}")] - CloudLoginFailed(#[from] CloudLoginError), - #[error("Credential Error: {0}")] - CredentialError(#[from] CloudCredentialError), - #[error("Failed to build: {0}")] - BuildFailed(#[from] BuildError), - #[error("Missing api config or security input")] - MissingSecurityConfig, - #[error(transparent)] - CloudError(#[from] CloudError), - #[error("Failed to server REST API: {0}")] - RestServeFailed(#[source] std::io::Error), - #[error("Failed to server gRPC API: {0:?}")] - GrpcServeFailed(#[source] tonic::transport::Error), - #[error("Failed to server pgwire: {0}")] - PGWireServerFailed(#[source] std::io::Error), - #[error("Cache {0} has reached its maximum size. Try to increase `cache_max_map_size` in the config.")] - CacheFull(String), - #[error("Internal thread panic: {0}")] - JoinError(#[source] tokio::task::JoinError), - #[error("Connector source factory error: {0}")] - ConnectorSourceFactory(#[from] ConnectorSourceFactoryError), - #[error(transparent)] - ExecutionError(#[from] ExecutionError), - #[error(transparent)] - PipelineError(#[from] PipelineError), - #[error(transparent)] - CliError(#[from] CliError), - #[error("table_name: {0:?} not found in any of the connections")] - SourceValidationError(String), - #[error("connection: {0:?} not found")] - ConnectionNotFound(String), - #[error("Pipeline validation failed")] - PipelineValidationError, - #[error("Output table {0} not used in any sink")] - OutputTableNotUsed(String), - #[error("Table name specified in sink not found: {0:?}")] - SinkTableNotFound(String), - #[error("No sinks initialized in the config provided")] - EmptySinks, - #[error(transparent)] - CloudContextError(#[from] CloudContextError), - #[error("Failed to read organisation name. Error: {0}")] - FailedToReadOrganisationName(#[source] io::Error), - #[error(transparent)] - AppUIError(#[from] AppUIError), - #[error("{LOCK_FILE} is out of date")] - LockedOutdatedLockfile, - #[error("{LOCK_FILE} does not exist. `--locked` requires a lock file.")] - LockedNoLockFile, - #[error("Command was aborted")] - Aborted, - #[error("This feature is only supported in enterprise: {0}")] - UnsupportedFeature(String), -} - -#[derive(Error, Debug)] -pub enum CliError { - #[error("Configuration file path not provided")] - ConfigurationFilePathNotProvided, - #[error("Can't find the configuration file(s) at: {0:?}")] - FailedToFindConfigurationFiles(String), - #[error("Unknown Command: {0:?}")] - UnknownCommand(String), - #[error("Failed to parse dozer config: {0:?}")] - FailedToParseYaml(#[source] BoxedError), - #[error("Failed to validate dozer config: {0:?}")] - FailedToParseValidateYaml(#[source] BoxedError), - #[error("Failed to read line: {0}")] - ReadlineError(#[from] rustyline::error::ReadlineError), - #[error("File system error {0:?}: {1}")] - FileSystem(PathBuf, #[source] std::io::Error), - #[error("Failed to create tokio runtime: {0}")] - FailedToCreateTokioRuntime(#[source] std::io::Error), - #[error("Reqwest error: {0}")] - Reqwest(#[from] reqwest::Error), - #[error(transparent)] - ConfigCombineError(#[from] ConfigCombineError), - #[error("Failed to serialize config to json: {0}")] - SerializeConfigToJson(#[source] serde_json::Error), - #[error("Missing config options to be overridden: {0}")] - MissingConfigOverride(String), - #[error("Failed to deserialize config from json: {0}")] - DeserializeConfigFromJson(#[source] serde_json::Error), - // Generic IO error - #[error(transparent)] - Io(#[from] std::io::Error), -} - -#[derive(Error, Debug)] -pub enum CloudError { - #[error("Connection failed. Error: {0:?}")] - ConnectionToCloudServiceError(#[from] tonic::transport::Error), - - #[error("Cloud service returned error: {}", map_status_error_message(.0.clone()))] - CloudServiceError(#[from] tonic::Status), - - #[error("GRPC request failed, error: {} (GRPC status {})", .0.message(), .0.code())] - GRPCCallError(#[source] tonic::Status), - - #[error(transparent)] - CloudCredentialError(#[from] CloudCredentialError), - - #[error("Reqwest error: {0}")] - Reqwest(#[from] reqwest::Error), - - #[error(transparent)] - CloudContextError(#[from] CloudContextError), - - #[error(transparent)] - ConfigCombineError(#[from] ConfigCombineError), - - #[error("Application not found")] - ApplicationNotFound, - - #[error("{LOCK_FILE} not found. Run `dozer build` before deploying, or pass '--no-lock'.")] - LockfileNotFound, -} - -#[derive(Debug, Error)] -pub enum ConfigCombineError { - #[error("Failed to parse yaml file {0}: {1}")] - ParseYaml(String, #[source] serde_yaml::Error), - - #[error("Cannot merge yaml value {from:?} to {to:?}")] - CannotMerge { - from: serde_yaml::Value, - to: serde_yaml::Value, - }, - - #[error("Failed to parse config: {0}")] - ParseConfig(#[source] serde_yaml::Error), - - #[error("Cannot read configuration: {0:?}. Error: {1:?}")] - CannotReadConfig(PathBuf, #[source] std::io::Error), - - #[error("Wrong pattern of config files read glob: {0}")] - WrongPatternOfConfigFilesGlob(#[from] PatternError), - - #[error("Cannot read file: {0}")] - CannotReadFile(#[from] GlobError), - - #[error("Cannot serialize config to string: {0}")] - CannotSerializeToString(#[source] serde_yaml::Error), - - #[error("SQL is not a string type")] - SqlIsNotStringType, - - #[error("Failed to read config to string")] - CannotReadUtf8String(#[from] FromUtf8Error), -} - -#[derive(Debug, Error)] -pub enum BuildError { - #[error("Endpoint {0} not found in DAG")] - MissingEndpoint(String), - #[error("Connection {0} found in DAG but not in config")] - MissingConnection(String), - #[error("Got mismatching primary key for `{table_name}`. Expected: `{expected:?}`, got: `{actual:?}`")] - MismatchPrimaryKey { - table_name: String, - expected: Vec, - actual: Vec, - }, - #[error("Field not found at position {0}")] - FieldNotFound(String), - #[error("File system error {0:?}: {1}")] - FileSystem(PathBuf, std::io::Error), - #[error( - "Failed to load existing contract: {0}. You have to run a force build: `dozer build --force`." - )] - FailedToLoadExistingContract(#[source] serde_json::Error), - #[error("Serde json error: {0}")] - SerdeJson(#[source] serde_json::Error), -} - -#[derive(Debug, Error)] -pub enum CloudLoginError { - #[error("Tonic error: {0}")] - TonicError(#[from] tonic::Status), - - #[error("Transport error: {0}")] - Transport(#[from] tonic::transport::Error), - - #[error("HttpRequest error: {0}")] - HttpRequestError(#[from] reqwest::Error), - - #[error(transparent)] - SerializationError(#[from] dozer_types::serde_json::Error), - - #[error("Failed to read input: {0}")] - InputError(#[from] std::io::Error), - - #[error(transparent)] - CloudCredentialError(#[from] CloudCredentialError), - - #[error("Organisation not found")] - OrganisationNotFound, -} - -#[derive(Debug, Error)] -pub enum CloudCredentialError { - #[error(transparent)] - SerializationError(#[from] dozer_types::serde_yaml::Error), - - #[error(transparent)] - JsonSerializationError(#[from] dozer_types::serde_json::Error), - #[error("Failed to create home directory: {0}")] - FailedToCreateDirectory(#[from] std::io::Error), - - #[error("HttpRequest error: {0}")] - HttpRequestError(#[from] reqwest::Error), - - #[error("Missing credentials.yaml file - Please try to login again")] - MissingCredentialFile, - #[error("There's no profile with given name - Please try to login again")] - MissingProfile, - #[error("{0}")] - LoginError(String), -} - -#[derive(Debug, Error)] -pub enum CloudContextError { - #[error("Failed to create access directory: {0}")] - FailedToAccessDirectory(#[from] std::io::Error), - - #[error("Failed to get current directory path")] - FailedToGetDirectoryPath, - - #[error("App id not found in configuration. You need to run \"deploy\" or \"set-app\" first")] - AppIdNotFound, - - #[error("App id already exists. If you want to create a new app, please remove your cloud configuration")] - AppIdAlreadyExists(String), -} - -fn map_status_error_message(status: tonic::Status) -> String { - match status.code() { - Code::PermissionDenied => { - "Permission denied. Please check your credentials and try again.".to_string() - } - _ => status.message().to_string(), - } -} diff --git a/dozer-cli/src/home_dir.rs b/dozer-cli/src/home_dir.rs deleted file mode 100644 index 4d73e3b8f0..0000000000 --- a/dozer-cli/src/home_dir.rs +++ /dev/null @@ -1,66 +0,0 @@ -use camino::Utf8PathBuf; - -#[derive(Debug, Clone)] -pub struct HomeDir { - home_dir: Utf8PathBuf, -} - -pub type Error = (Utf8PathBuf, std::io::Error); - -impl HomeDir { - pub fn new(home_dir: Utf8PathBuf) -> Self { - Self { home_dir } - } - - pub fn create_build_dir_all(&self, build_id: BuildId) -> Result { - let build_path = self.get_build_path(build_id); - - std::fs::create_dir_all(&build_path.contracts_dir) - .map_err(|e| (build_path.contracts_dir.clone(), e))?; - std::fs::create_dir_all(&build_path.data_dir) - .map_err(|e: std::io::Error| (build_path.data_dir.clone(), e))?; - - Ok(build_path) - } - - fn get_build_path(&self, build_id: BuildId) -> BuildPath { - let build_dir = self.home_dir.join(&build_id.name); - - let contracts_dir = build_dir.join("contracts"); - let descriptor_path = contracts_dir.join("file_descriptor_set.bin"); - - let data_dir = build_dir.join("data"); - - BuildPath { - id: build_id, - contracts_dir, - descriptor_path, - data_dir, - } - } -} - -#[derive(Debug, Clone)] -pub struct BuildId { - name: String, -} - -impl BuildId { - fn from_id(id: u32) -> Self { - Self { - name: format!("v{id:04}"), - } - } - - pub fn first() -> Self { - Self::from_id(1) - } -} - -#[derive(Debug, Clone)] -pub struct BuildPath { - pub id: BuildId, - pub contracts_dir: Utf8PathBuf, - pub descriptor_path: Utf8PathBuf, - pub data_dir: Utf8PathBuf, -} diff --git a/dozer-cli/src/lib.rs b/dozer-cli/src/lib.rs deleted file mode 100644 index 7eeace9f44..0000000000 --- a/dozer-cli/src/lib.rs +++ /dev/null @@ -1,96 +0,0 @@ -pub mod cli; -pub mod errors; -mod home_dir; -pub mod pipeline; -pub mod simple; -pub mod ui; -use dozer_core::errors::ExecutionError; -use dozer_core::shutdown::ShutdownSender; -use dozer_types::log::debug; -use errors::OrchestrationError; - -pub use actix_web; -pub use async_trait; -use std::{ - backtrace::{Backtrace, BacktraceStatus}, - panic, process, - thread::current, -}; -use tokio::task::JoinHandle; -pub mod config_helper; -pub mod console_helper; -pub use dozer_core::shutdown; -pub use tonic_reflection; -pub use tonic_web; -pub use tower_http; -#[cfg(test)] -mod tests; -mod utils; -// Re-exports -pub use dozer_ingestion::{ - errors::ConnectorError, - {get_connector, TableInfo}, -}; - -pub use dozer_types::models::connection::Connection; -use dozer_types::tracing::error; - -async fn flatten_join_handle( - handle: JoinHandle>, -) -> Result<(), OrchestrationError> { - match handle.await { - Ok(Ok(_)) => Ok(()), - Ok(Err(err)) => Err(err), - Err(err) => Err(OrchestrationError::JoinError(err)), - } -} - -pub fn set_panic_hook() { - panic::set_hook(Box::new(move |panic_info| { - // All the orchestrator errors are captured here - if let Some(e) = panic_info.payload().downcast_ref::() { - error!("{}", e); - debug!("{:?}", e); - // All the connector errors are captured here - } else if let Some(e) = panic_info.payload().downcast_ref::() { - error!("{}", e); - debug!("{:?}", e); - // All the pipeline errors are captured here - } else if let Some(e) = panic_info.payload().downcast_ref::() { - error!("{}", e); - debug!("{:?}", e); - // If any errors are sent as strings. - } else if let Some(s) = panic_info.payload().downcast_ref::<&str>() { - error!("{s:?}"); - } else { - error!("{}", panic_info); - } - - let backtrace = Backtrace::capture(); - if backtrace.status() == BacktraceStatus::Captured { - error!( - "thread '{}' panicked at '{}'\n stack backtrace:\n{}", - current() - .name() - .map(ToString::to_string) - .unwrap_or_default(), - panic_info - .location() - .map(ToString::to_string) - .unwrap_or_default(), - backtrace - ); - } - - process::exit(1); - })); -} - -pub async fn set_ctrl_handler(shutdown_sender: ShutdownSender) { - tokio::spawn(async { - tokio::signal::ctrl_c() - .await - .expect("Error setting Ctrl-C handler"); - shutdown_sender.shutdown(); - }); -} diff --git a/dozer-cli/src/main.rs b/dozer-cli/src/main.rs deleted file mode 100644 index 241e751574..0000000000 --- a/dozer-cli/src/main.rs +++ /dev/null @@ -1,145 +0,0 @@ -use clap::Parser; -use dozer_cli::cli::init_config; -use dozer_cli::cli::init_dozer; -use dozer_cli::cli::types::{Cli, Commands, UICommands}; -use dozer_cli::errors::{CliError, CloudError, OrchestrationError}; -use dozer_cli::ui; -use dozer_cli::ui::app::AppUIError; -use dozer_cli::{set_ctrl_handler, set_panic_hook}; -use dozer_core::shutdown; -use dozer_tracing::DozerMonitorContext; -use dozer_types::models::config::Config; -use dozer_types::tracing::{error, error_span, info}; -use futures::TryFutureExt; -use std::process; -use std::sync::Arc; -use tokio::runtime::Runtime; - -fn main() { - if let Err(e) = run() { - display_error(&e); - process::exit(1); - } -} - -fn run() -> Result<(), OrchestrationError> { - // Reloading trace layer seems impossible, so we are running Cli::parse in a closure - // and then initializing it after reading the configuration. This is a hacky workaround, but it works. - - let cli = parse_and_generate()?; - - let runtime = Arc::new(Runtime::new().map_err(CliError::FailedToCreateTokioRuntime)?); - - let (shutdown_sender, shutdown_receiver) = shutdown::new(&runtime); - - runtime.block_on(set_ctrl_handler(shutdown_sender)); - - set_panic_hook(); - - let config_res = init_configuration(&cli, runtime.clone()); - - // Now we have access to telemetry configuration. Telemetry must be initialized in tokio runtime. - let app_id = config_res.as_ref().map(|(c, _)| c.app_name.as_str()).ok(); - - let telemetry_config = config_res - .as_ref() - .map(|(c, _)| c.telemetry.clone()) - .unwrap_or_default(); - - runtime.block_on(async { - let t = dozer_tracing::Telemetry::new(app_id, &telemetry_config); - let _r = tokio::spawn(async move { t.serve().await }); - }); - - // running UI does not require config to be loaded - if let Commands::UI(run) = &cli.cmd { - if let Some(UICommands::Update) = run.command { - runtime.block_on( - ui::downloader::fetch_latest_dozer_app_ui_code() - .map_err(AppUIError::DownloaderError), - )?; - info!("Run `dozer ui` to see the changes."); - } else { - runtime.block_on(ui::app::start_app_ui_server( - &runtime, - shutdown_receiver, - false, - ))?; - } - return Ok(()); - } - - let (config, config_files) = config_res?; - info!("Loaded config from: {}", config_files.join(", ")); - - let dozer = init_dozer( - runtime.clone(), - config.clone(), - DozerMonitorContext::new( - config.id.clone(), - config.company_id.clone(), - cli.enable_progress, - ), - ) - .map_err(OrchestrationError::CliError)?; - - // run individual servers - (match cli.cmd { - Commands::Run => dozer - .runtime - .block_on(dozer.run_apps(shutdown_receiver, None)), - Commands::Build(build) => { - let force = build.force.is_some(); - - dozer - .runtime - .block_on(dozer.build(force, shutdown_receiver, build.locked)) - } - Commands::Clean => dozer.clean(), - Commands::UI(_) => { - panic!("This should not happen as it is handled earlier"); - } - }) - .map_err(|e| { - let _span = error_span!("OrchestrationError", error = %e); - - e - }) -} - -// Some commands dont need to initialize the orchestrator -// This function is used to run those commands -fn parse_and_generate() -> Result { - dozer_tracing::init_telemetry_closure( - None, - &Default::default(), - || -> Result { - let cli = Cli::parse(); - - Ok(cli) - }, - ) -} - -fn init_configuration(cli: &Cli, runtime: Arc) -> Result<(Config, Vec), CliError> { - dozer_tracing::init_telemetry_closure(None, &Default::default(), || -> Result<_, CliError> { - runtime.block_on(init_config( - cli.config_paths.clone(), - cli.config_token.clone(), - cli.config_overrides.clone(), - cli.ignore_pipe, - )) - }) -} - -fn display_error(e: &OrchestrationError) { - if let OrchestrationError::CloudError(CloudError::ApplicationNotFound) = &e { - let description = "Dozer cloud service was not able to find application. \n\n\ - Please check your application id in `dozer-config.cloud.yaml` file.\n\ - To change it, you can manually update file or use \"dozer cloud set-app {app_id}\"."; - - error!("{}", description); - } else { - error!("{}", e); - } -} diff --git a/dozer-cli/src/pipeline/builder.rs b/dozer-cli/src/pipeline/builder.rs deleted file mode 100644 index 45e755a58e..0000000000 --- a/dozer-cli/src/pipeline/builder.rs +++ /dev/null @@ -1,335 +0,0 @@ -use std::collections::HashMap; -use std::collections::HashSet; -use std::sync::Arc; - -use dozer_core::app::App; -use dozer_core::app::AppPipeline; -use dozer_core::app::PipelineEntryPoint; -use dozer_core::node::SinkFactory; -use dozer_core::shutdown::ShutdownReceiver; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_sql::builder::statement_to_pipeline; -use dozer_sql::builder::{OutputNodeInfo, QueryContext}; -use dozer_tracing::DozerMonitorContext; -use dozer_types::log::debug; -use dozer_types::models::connection::Connection; -use dozer_types::models::connection::ConnectionConfig; -use dozer_types::models::flags::Flags; -use dozer_types::models::sink::Sink; -use dozer_types::models::sink::SinkConfig; -use dozer_types::models::source::Source; -use dozer_types::models::udf_config::UdfConfig; -use dozer_types::types::PortHandle; -use std::hash::Hash; -use tokio::runtime::Runtime; - -use crate::pipeline::dummy_sink::DummySinkFactory; -use dozer_sink_clickhouse::ClickhouseSinkFactory; - -use super::source_builder::SourceBuilder; -use crate::errors::OrchestrationError; - -use OrchestrationError::ExecutionError; - -pub enum OutputTableInfo { - Transformed(OutputNodeInfo), - Original(OriginalTableInfo), -} - -pub struct OriginalTableInfo { - pub table_name: String, - pub connection_name: String, -} - -pub struct CalculatedSources { - pub original_sources: Vec, - pub transformed_sources: Vec, - pub query_context: Option, -} - -pub struct PipelineBuilder<'a> { - connections: &'a [Connection], - sources: &'a [Source], - sql: Option<&'a str>, - sinks: &'a [Sink], - labels: DozerMonitorContext, - flags: Flags, - udfs: &'a [UdfConfig], -} - -impl<'a> PipelineBuilder<'a> { - pub fn new( - connections: &'a [Connection], - sources: &'a [Source], - sql: Option<&'a str>, - sinks: &'a [Sink], - labels: DozerMonitorContext, - flags: Flags, - udfs: &'a [UdfConfig], - ) -> Self { - Self { - connections, - sources, - sql, - sinks, - labels, - flags, - udfs, - } - } - - // Based on used_sources, map it to the connection name and create sources - // For not breaking current functionality, current format is to be still supported. - pub async fn get_grouped_tables( - &self, - _runtime: &Arc, - original_sources: &[String], - ) -> Result>, OrchestrationError> { - let mut grouped_connections: HashMap> = HashMap::new(); - - let mut connector_map: HashMap> = HashMap::new(); - for source in self.sources { - let connection = self - .connections - .iter() - .find(|conn| conn.name == source.connection) - .ok_or_else(|| OrchestrationError::ConnectionNotFound(source.connection.clone()))?; - connector_map - .entry(connection.clone()) - .or_default() - .push(source.clone()); - } - - for table_name in original_sources { - let mut table_found = false; - for (connection, tables) in connector_map.iter() { - if let Some(source) = tables - .iter() - .find(|table| table.name == table_name.as_str()) - { - table_found = true; - grouped_connections - .entry(connection.clone()) - .or_default() - .push(source.clone()); - } - } - - if !table_found { - return Err(OrchestrationError::SourceValidationError( - table_name.to_string(), - )); - } - } - - Ok(grouped_connections) - } - - // This function is used to figure out the sources that are used in the pipeline - // based on the SQL and API Endpoints - pub fn calculate_sources( - &self, - runtime: Arc, - ) -> Result { - let mut original_sources = vec![]; - - let mut query_ctx = None; - let mut pipeline = AppPipeline::new((&self.flags).into()); - - let mut transformed_sources = vec![]; - - if let Some(sql) = &self.sql { - let query_context = - statement_to_pipeline(sql, &mut pipeline, None, self.udfs.to_vec(), runtime) - .map_err(OrchestrationError::PipelineError)?; - - query_ctx = Some(query_context.clone()); - - transformed_sources = query_context.output_tables_map.keys().cloned().collect(); - - for name in query_context.used_sources { - // Add all source tables to input tables - original_sources.push(name); - } - } - - // Add Used Souces if direct from source - for sink in self.sinks { - for table_name in table_names(sink) { - // Don't add if the table is a result of SQL - if !transformed_sources.contains(table_name) { - original_sources.push(table_name.to_string()); - } - } - } - dedup(&mut original_sources); - dedup(&mut transformed_sources); - - Ok(CalculatedSources { - original_sources, - transformed_sources, - query_context: query_ctx, - }) - } - - // This function is used by both building and actual execution - pub async fn build( - self, - runtime: &Arc, - shutdown: ShutdownReceiver, - ) -> Result { - let calculated_sources = self.calculate_sources(runtime.clone())?; - - debug!("Used Sources: {:?}", calculated_sources.original_sources); - let grouped_connections = self - .get_grouped_tables(runtime, &calculated_sources.original_sources) - .await?; - - let mut pipelines: Vec = vec![]; - - let mut pipeline = AppPipeline::new(self.flags.into()); - - let mut available_output_tables: HashMap = HashMap::new(); - - // Add all source tables to available output tables - for (connection, sources) in &grouped_connections { - for source in sources { - available_output_tables.insert( - source.name.clone(), - OutputTableInfo::Original(OriginalTableInfo { - connection_name: connection.name.to_string(), - table_name: source.name.clone(), - }), - ); - } - } - - if let Some(sql) = &self.sql { - let query_context = statement_to_pipeline( - sql, - &mut pipeline, - None, - self.udfs.to_vec(), - runtime.clone(), - ) - .map_err(OrchestrationError::PipelineError)?; - - for (name, table_info) in query_context.output_tables_map { - available_output_tables - .insert(name.clone(), OutputTableInfo::Transformed(table_info)); - } - } - - // Check if all output tables are used. - for (table_name, table_info) in &available_output_tables { - if matches!(table_info, OutputTableInfo::Transformed(_)) - && !is_table_used(table_name, self.sinks) - { - return Err(OrchestrationError::OutputTableNotUsed(table_name.clone())); - } - } - - let get_table_info = |table_name: &String| { - available_output_tables - .get(table_name) - .ok_or_else(|| OrchestrationError::SinkTableNotFound(table_name.clone())) - }; - - for sink in self.sinks { - let id = &sink.name; - match &sink.config { - SinkConfig::Dummy(config) => add_sink_to_pipeline( - &mut pipeline, - Box::new(DummySinkFactory), - id, - vec![(get_table_info(&config.table_name)?, DEFAULT_PORT_HANDLE)], - ), - - SinkConfig::Clickhouse(config) => { - let sink = - Box::new(ClickhouseSinkFactory::new(config.clone(), runtime.clone())); - let table_info = get_table_info(&config.source_table_name)?; - add_sink_to_pipeline( - &mut pipeline, - sink, - id, - vec![(table_info, DEFAULT_PORT_HANDLE)], - ); - } - x => { - return Err(OrchestrationError::UnsupportedFeature(x.name())); - } - } - } - - pipelines.push(pipeline); - - let source_builder = SourceBuilder::new(grouped_connections, self.labels); - let asm = source_builder - .build_source_manager(runtime, shutdown) - .await?; - let mut app = App::new(asm); - - Vec::into_iter(pipelines).for_each(|p| { - app.add_pipeline(p); - }); - - let dag = app.into_dag().map_err(ExecutionError)?; - - Ok(dag) - } -} - -fn dedup(v: &mut Vec) { - let mut uniques = HashSet::new(); - v.retain(|e| uniques.insert(e.clone())); -} - -fn table_names(sink: &Sink) -> Vec<&String> { - match &sink.config { - SinkConfig::Dummy(sink) => vec![&sink.table_name], - SinkConfig::Aerospike(sink) => sink - .tables - .iter() - .map(|table| &table.source_table_name) - .collect(), - SinkConfig::Clickhouse(sink) => vec![&sink.source_table_name], - SinkConfig::Oracle(sink) => vec![&sink.table_name], - } -} - -fn is_table_used(table_name: &str, sinks: &[Sink]) -> bool { - sinks.iter().any(|sink| { - table_names(sink) - .iter() - .any(|sink_table_name| sink_table_name == &table_name) - }) -} - -fn add_sink_to_pipeline( - pipeline: &mut AppPipeline, - sink: Box, - id: &str, - table_infos: Vec<(&OutputTableInfo, PortHandle)>, -) { - pipeline.add_sink(sink, id.to_string()); - - for (table_info, port) in table_infos { - match table_info { - OutputTableInfo::Original(table_info) => { - pipeline.add_entry_point( - id.to_string(), - PipelineEntryPoint::new(table_info.table_name.clone(), port), - ); - } - OutputTableInfo::Transformed(table_info) => { - pipeline.connect_nodes( - table_info.node.clone(), - table_info.port, - id.to_string(), - port, - ); - } - } - } -} diff --git a/dozer-cli/src/pipeline/connector_source.rs b/dozer-cli/src/pipeline/connector_source.rs deleted file mode 100644 index 6b4d550fc8..0000000000 --- a/dozer-cli/src/pipeline/connector_source.rs +++ /dev/null @@ -1,389 +0,0 @@ -use dozer_core::event::EventHub; -use dozer_core::node::{OutputPortDef, OutputPortType, PortHandle, Source, SourceFactory}; -use dozer_core::shutdown::ShutdownReceiver; -use dozer_ingestion::{ - get_connector, CdcType, Connector, IngestionIterator, TableIdentifier, TableInfo, -}; -use dozer_ingestion::{IngestionConfig, Ingestor}; -use dozer_tracing::constants::{ - ConnectorEntityType, CONNECTION_LABEL, DOZER_METER_NAME, OPERATION_TYPE_LABEL, - SOURCE_OPERATION_COUNTER_NAME, TABLE_LABEL, -}; -use dozer_tracing::{emit_event, DozerMonitorContext}; -use dozer_types::errors::internal::BoxedError; -use dozer_types::models::connection::Connection; -use dozer_types::models::ingestion_types::IngestionMessage; -use dozer_types::node::OpIdentifier; -use dozer_types::thiserror::{self, Error}; -use dozer_types::tracing::info; -use dozer_types::types::{Operation, Schema, SourceDefinition}; -use futures::stream::{AbortHandle, Abortable, Aborted}; -use std::collections::HashMap; -use std::sync::Arc; -use tokio::runtime::Runtime; -use tokio::sync::mpsc::Sender; -use tonic::async_trait; - -#[derive(Debug)] -struct Table { - schema_name: Option, - name: String, - columns: Vec, - schema: Schema, - cdc_type: CdcType, - port: PortHandle, -} - -#[derive(Debug, Error)] -pub enum ConnectorSourceFactoryError { - #[error("Connector error: {0}")] - Connector(#[source] BoxedError), - #[error("Port not found for source: {0}")] - PortNotFoundInSource(PortHandle), - #[error("Schema not initialized")] - SchemaNotInitialized, -} - -#[derive(Debug)] -pub struct ConnectorSourceFactory { - connection: Connection, - runtime: Arc, - tables: Vec, - labels: DozerMonitorContext, - shutdown: ShutdownReceiver, -} - -fn map_replication_type_to_output_port_type(_typ: &CdcType) -> OutputPortType { - OutputPortType::Stateless -} - -impl ConnectorSourceFactory { - pub async fn new( - mut table_and_ports: Vec<(TableInfo, PortHandle)>, - connection: Connection, - runtime: Arc, - labels: DozerMonitorContext, - shutdown: ShutdownReceiver, - ) -> Result { - let mut connector = - get_connector(runtime.clone(), EventHub::new(1), connection.clone(), None) - .map_err(|e| ConnectorSourceFactoryError::Connector(e.into()))?; - - // Fill column names if not provided. - let table_identifiers = table_and_ports - .iter() - .map(|(table, _)| TableIdentifier::new(table.schema.clone(), table.name.clone())) - .collect(); - let all_columns = connector - .list_columns(table_identifiers) - .await - .map_err(ConnectorSourceFactoryError::Connector)?; - for ((table, _), columns) in table_and_ports.iter_mut().zip(all_columns) { - if table.column_names.is_empty() { - table.column_names = columns.column_names; - } - } - - let tables: Vec = table_and_ports - .iter() - .map(|(table, _)| table.clone()) - .collect(); - let source_schemas = connector - .get_schemas(&tables) - .await - .map_err(ConnectorSourceFactoryError::Connector)?; - - let mut tables = vec![]; - for ((table, port), source_schema) in table_and_ports.into_iter().zip(source_schemas) { - let name = table.name; - let columns = table.column_names; - let source_schema = source_schema.map_err(ConnectorSourceFactoryError::Connector)?; - let schema = source_schema.schema; - let cdc_type = source_schema.cdc_type; - - let table = Table { - name, - schema_name: table.schema.clone(), - columns, - schema, - cdc_type, - port, - }; - - tables.push(table); - } - - Ok(Self { - connection, - runtime, - tables, - labels, - shutdown, - }) - } -} - -impl SourceFactory for ConnectorSourceFactory { - fn get_output_schema(&self, port: &PortHandle) -> Result { - let table = self - .tables - .iter() - .find(|table| table.port == *port) - .ok_or(ConnectorSourceFactoryError::PortNotFoundInSource(*port))?; - let mut schema = table.schema.clone(); - let table_name = &table.name; - - // Add source information to the schema. - for field in &mut schema.fields { - field.source = SourceDefinition::Table { - connection: self.connection.name.clone(), - name: table_name.clone(), - }; - } - - info!( - "Source: Initializing input schema: {}\n{}", - table_name, - schema.print() - ); - - Ok(schema) - } - - fn get_output_port_name(&self, port: &PortHandle) -> String { - let table = self - .tables - .iter() - .find(|table| table.port == *port) - .unwrap_or_else(|| panic!("Port {} not found", port)); - table.name.clone() - } - - fn get_output_ports(&self) -> Vec { - self.tables - .iter() - .map(|table| { - let typ = map_replication_type_to_output_port_type(&table.cdc_type); - OutputPortDef::new(table.port, typ) - }) - .collect() - } - - fn build( - &self, - _output_schemas: HashMap, - event_hub: EventHub, - state: Option>, - ) -> Result, BoxedError> { - // Construct table info. - let tables = self - .tables - .iter() - .map(|table| TableInfo { - schema: table.schema_name.clone(), - name: table.name.clone(), - column_names: table.columns.clone(), - }) - .collect(); - let ports = self.tables.iter().map(|table| table.port).collect(); - - let connector = get_connector( - self.runtime.clone(), - event_hub, - self.connection.clone(), - state, - )?; - - Ok(Box::new(ConnectorSource { - tables, - ports, - connector, - connection_name: self.connection.name.clone(), - labels: self.labels.clone(), - shutdown: self.shutdown.clone(), - ingestion_config: IngestionConfig::default(), - })) - } -} - -#[derive(Debug)] -pub struct ConnectorSource { - tables: Vec, - ports: Vec, - connector: Box, - connection_name: String, - labels: DozerMonitorContext, - shutdown: ShutdownReceiver, - ingestion_config: IngestionConfig, -} - -#[async_trait] -impl Source for ConnectorSource { - async fn serialize_state(&self) -> Result, BoxedError> { - self.connector.serialize_state().await - } - - async fn start( - &mut self, - sender: Sender<(PortHandle, IngestionMessage)>, - last_checkpoint: Option, - ) -> Result<(), BoxedError> { - let (ingestor, iterator) = Ingestor::initialize_channel(self.ingestion_config.clone()); - let connection_name = self.connection_name.clone(); - let tables = self.tables.clone(); - let ports = self.ports.clone(); - let labels = self.labels.clone(); - let handle = tokio::spawn(forward_message_to_pipeline( - iterator, - sender, - connection_name.clone(), - tables, - ports, - labels, - )); - - let shutdown_future = self.shutdown.create_shutdown_future(); - let (abort_handle, abort_registration) = AbortHandle::new_pair(); - let labels = self.labels.clone(); - - // Abort the connector when we shut down - // TODO: pass a `CancellationToken` to the connector to allow - // it to gracefully shut down. - let name = self.connection_name.clone(); - tokio::spawn(async move { - shutdown_future.await; - abort_handle.abort(); - eprintln!("Aborted connector {}", name); - }); - - emit_event( - &connection_name, - &ConnectorEntityType::Connector, - &labels, - "source_started", - ); - - let result = Abortable::new( - self.connector - .start(&ingestor, self.tables.clone(), last_checkpoint), - abort_registration, - ) - .await; - - match result { - Ok(Ok(())) => {} - Ok(Err(e)) => return Err(e), - // Aborted means we are shutting down - Err(Aborted) => { - emit_event( - &connection_name, - &ConnectorEntityType::Connector, - &labels, - "source_aborted", - ); - - return Ok(()); - } - } - drop(ingestor); - - // If we reach here, it means the connector has finished ingesting, so we wait for the forwarding task to finish. - if let Err(e) = handle.await { - emit_event( - &connection_name, - &ConnectorEntityType::Connector, - &labels, - "source_error", - ); - - std::panic::panic_any(e); - } - - Ok(()) - } -} - -async fn forward_message_to_pipeline( - mut iterator: IngestionIterator, - sender: Sender<(PortHandle, IngestionMessage)>, - connection_name: String, - tables: Vec, - ports: Vec, - labels: DozerMonitorContext, -) { - let mut bars = vec![]; - for table in &tables { - let pb = labels.create_progress_bar(table.name.clone()); - bars.push(pb); - } - - let meter = dozer_tracing::global::meter(DOZER_METER_NAME); - - let source_counter = meter - .u64_counter(SOURCE_OPERATION_COUNTER_NAME) - .with_description("Number of operation processed by source") - .init(); - - let mut counter = vec![(0u64, 0u64); tables.len()]; - while let Some(message) = iterator.receiver.recv().await { - match &message { - IngestionMessage::OperationEvent { - table_index, op, .. - } => { - let port = ports[*table_index]; - let table_name = &tables[*table_index].name; - - // Update metrics - - let mut labels = labels.attrs(); - labels.push(dozer_tracing::KeyValue::new( - CONNECTION_LABEL, - connection_name.clone(), - )); - - labels.push(dozer_tracing::KeyValue::new( - TABLE_LABEL, - table_name.clone(), - )); - - let op_str = match op { - Operation::Insert { .. } => "insert", - Operation::Delete { .. } => "delete", - Operation::Update { .. } => "update", - Operation::BatchInsert { .. } => "insert", - }; - labels.push(dozer_tracing::KeyValue::new(OPERATION_TYPE_LABEL, op_str)); - - let counter_number: u64 = match op { - Operation::BatchInsert { new } => new.to_owned().len().try_into().unwrap_or(1), - _ => 1, - }; - - source_counter.add(counter_number, &labels); - - // Update counter - let counter = &mut counter[*table_index]; - if let Operation::BatchInsert { new } = &op { - counter.0 += new.len() as u64; - } else { - counter.0 += 1; - } - if counter.0 >> 10 > counter.1 { - counter.1 = counter.0 >> 10; - bars[*table_index].set_position(counter.0); - } - - // Send message to the pipeline - if sender.send((port, message)).await.is_err() { - break; - } - } - IngestionMessage::TransactionInfo(_) => { - // For transaction level messages, we can send to any port. - if sender.send((ports[0], message)).await.is_err() { - break; - } - } - } - } -} diff --git a/dozer-cli/src/pipeline/dummy_sink.rs b/dozer-cli/src/pipeline/dummy_sink.rs deleted file mode 100644 index 9607278bad..0000000000 --- a/dozer-cli/src/pipeline/dummy_sink.rs +++ /dev/null @@ -1,193 +0,0 @@ -use std::{collections::HashMap, time::Instant}; - -use dozer_core::{ - epoch::Epoch, - event::EventHub, - node::{PortHandle, Sink, SinkFactory}, - DEFAULT_PORT_HANDLE, -}; -use dozer_types::log::debug; -use dozer_types::{ - chrono::Local, - errors::internal::BoxedError, - log::{info, warn}, - node::OpIdentifier, - types::{FieldType, Operation, Schema, TableOperation}, -}; - -use crate::async_trait::async_trait; - -#[derive(Debug)] -pub struct DummySinkFactory; - -#[async_trait] -impl SinkFactory for DummySinkFactory { - fn get_input_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - fn get_input_port_name(&self, _port: &PortHandle) -> String { - "dummy".to_string() - } - - fn prepare(&self, _input_schemas: HashMap) -> Result<(), BoxedError> { - Ok(()) - } - - async fn build( - &self, - input_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - let inserted_at_index = input_schemas - .into_values() - .next() - .and_then(|schema| { - schema.fields.into_iter().enumerate().find(|(_, field)| { - field.name.to_lowercase() == "inserted_at" && field.typ == FieldType::Timestamp - }) - }) - .map(|(index, _)| index); - Ok(Box::new(DummySink { - inserted_at_index, - previous_started: Instant::now(), - count: 0, - snapshotting_started_instant: HashMap::new(), - stop_after: std::env::var("STOP_AFTER").map_or(None, |s| s.parse().ok()), - first_received: None, - total_latency: 0, - previous_op_count: 0, - })) - } - - fn type_name(&self) -> String { - "dummy".to_string() - } -} - -#[derive(Debug)] -struct DummySink { - snapshotting_started_instant: HashMap, - inserted_at_index: Option, - count: usize, - previous_started: Instant, - previous_op_count: usize, - first_received: Option, - stop_after: Option, - total_latency: u64, -} - -impl Sink for DummySink { - fn process(&mut self, op: TableOperation) -> Result<(), BoxedError> { - if self.count == 0 { - self.first_received = Some(Instant::now()); - } - let diff = self.count - self.previous_op_count; - if diff > 1000 { - if self.count > 0 { - info!( - "Rate: {:.0} op/s, Processed {} records. Elapsed {:?}", - diff as f64 / self.previous_started.elapsed().as_secs_f64(), - self.count, - self.previous_started.elapsed(), - ); - } - self.previous_started = Instant::now(); - self.previous_op_count = self.count; - } - - self.count += match op.op { - Operation::BatchInsert { ref new } => new.len(), - _ => 1, - }; - - if let Some(stop_after) = self.stop_after { - if self.count >= stop_after as usize { - if let Some(first_received) = self.first_received { - info!("Stopping after {} records", stop_after); - - info!( - "Rate: {:.0} op/s, Processed {} records. Elapsed {:?}", - stop_after as f64 / first_received.elapsed().as_secs_f64(), - self.count, - first_received.elapsed(), - ); - - if self.total_latency > 0 { - info!( - "Average latency: {}ms", - self.total_latency / stop_after as u64 - ); - } - std::process::exit(0); - } - } - } - - if let Some(inserted_at_index) = self.inserted_at_index { - let records = match op.op { - Operation::BatchInsert { ref new } => new, - Operation::Insert { ref new } => std::slice::from_ref(new), - _ => &[], - }; - - for new in records { - debug!("Received record: {:?}", new); - let value = &new.values[inserted_at_index]; - if let Some(inserted_at) = value.to_timestamp() { - let latency = Local::now().naive_utc() - inserted_at.naive_utc(); - self.total_latency += latency.num_milliseconds() as u64; - info!("Latency: {}ms", latency.num_milliseconds()); - } else { - warn!("expecting timestamp, got {:?}", value); - } - } - } - Ok(()) - } - - fn commit(&mut self, _epoch_details: &Epoch) -> Result<(), BoxedError> { - Ok(()) - } - - fn on_source_snapshotting_started( - &mut self, - connection_name: String, - ) -> Result<(), BoxedError> { - self.snapshotting_started_instant - .insert(connection_name, Instant::now()); - Ok(()) - } - - fn on_source_snapshotting_done( - &mut self, - connection_name: String, - _id: Option, - ) -> Result<(), BoxedError> { - if let Some(started_instant) = self.snapshotting_started_instant.remove(&connection_name) { - info!( - "Snapshotting for connection {} took {:?}", - connection_name, - started_instant.elapsed() - ); - } else { - warn!( - "Snapshotting for connection {} took unknown time", - connection_name - ); - } - Ok(()) - } - - fn set_source_state(&mut self, _source_state: &[u8]) -> Result<(), BoxedError> { - Ok(()) - } - - fn get_source_state(&mut self) -> Result>, BoxedError> { - Ok(None) - } - - fn get_latest_op_id(&mut self) -> Result, BoxedError> { - Ok(None) - } -} diff --git a/dozer-cli/src/pipeline/mod.rs b/dozer-cli/src/pipeline/mod.rs deleted file mode 100644 index ad3ba89f2a..0000000000 --- a/dozer-cli/src/pipeline/mod.rs +++ /dev/null @@ -1,9 +0,0 @@ -mod builder; -pub mod connector_source; -mod dummy_sink; -pub mod source_builder; - -pub use builder::PipelineBuilder; - -#[cfg(test)] -mod tests; diff --git a/dozer-cli/src/pipeline/source_builder.rs b/dozer-cli/src/pipeline/source_builder.rs deleted file mode 100644 index 8f646ae80a..0000000000 --- a/dozer-cli/src/pipeline/source_builder.rs +++ /dev/null @@ -1,89 +0,0 @@ -use crate::pipeline::connector_source::ConnectorSourceFactory; -use crate::OrchestrationError; -use dozer_core::appsource::{AppSourceManager, AppSourceMappings}; -use dozer_core::shutdown::ShutdownReceiver; -use dozer_ingestion::TableInfo; - -use dozer_tracing::DozerMonitorContext; -use dozer_types::models::connection::Connection; -use dozer_types::models::source::Source; -use std::collections::HashMap; -use std::sync::Arc; -use tokio::runtime::Runtime; - -pub struct SourceBuilder { - grouped_connections: HashMap>, - labels: DozerMonitorContext, -} - -const SOURCE_PORTS_RANGE_START: u16 = 1000; - -impl SourceBuilder { - pub fn new( - grouped_connections: HashMap>, - labels: DozerMonitorContext, - ) -> Self { - Self { - grouped_connections, - labels, - } - } - - pub fn get_ports(&self) -> HashMap<(&str, &str), u16> { - let mut port: u16 = SOURCE_PORTS_RANGE_START; - - let mut ports = HashMap::new(); - for (conn, sources_group) in &self.grouped_connections { - for source in sources_group { - ports.insert((conn.name.as_str(), source.name.as_str()), port); - port += 1; - } - } - ports - } - - pub async fn build_source_manager( - &self, - runtime: &Arc, - shutdown: ShutdownReceiver, - ) -> Result { - let mut asm = AppSourceManager::new(); - - let mut port: u16 = SOURCE_PORTS_RANGE_START; - - for (connection, sources_group) in &self.grouped_connections { - let mut ports = HashMap::new(); - let mut table_and_ports = vec![]; - for source in sources_group { - ports.insert(source.name.clone(), port); - - table_and_ports.push(( - TableInfo { - schema: source.schema.clone(), - name: source.table_name.clone(), - column_names: source.columns.clone(), - }, - port, - )); - - port += 1; - } - - let source_factory = ConnectorSourceFactory::new( - table_and_ports, - connection.clone(), - runtime.clone(), - self.labels.clone(), - shutdown.clone(), - ) - .await?; - - asm.add( - Box::new(source_factory), - AppSourceMappings::new(connection.name.to_string(), ports), - )?; - } - - Ok(asm) - } -} diff --git a/dozer-cli/src/pipeline/tests/builder.rs b/dozer-cli/src/pipeline/tests/builder.rs deleted file mode 100644 index a250f79893..0000000000 --- a/dozer-cli/src/pipeline/tests/builder.rs +++ /dev/null @@ -1,90 +0,0 @@ -use std::sync::Arc; - -use crate::pipeline::source_builder::SourceBuilder; -use crate::pipeline::PipelineBuilder; -use dozer_core::shutdown; -use dozer_types::models::config::Config; -use dozer_types::models::ingestion_types::{ConfigSchemas, GrpcConfig}; - -use dozer_types::models::connection::{Connection, ConnectionConfig}; -use dozer_types::models::flags::Flags; -use dozer_types::models::source::Source; - -fn get_default_config() -> Config { - let schema_str = include_str!("./schemas.json"); - let grpc_conn = Connection { - config: ConnectionConfig::Grpc(GrpcConfig { - host: None, - port: None, - adapter: None, - schemas: ConfigSchemas::Inline(schema_str.to_string()), - }), - name: "grpc_conn".to_string(), - }; - - Config { - app_name: "multi".to_string(), - version: 1, - api: Default::default(), - flags: Default::default(), - connections: vec![grpc_conn.clone()], - sources: vec![ - Source { - name: "grpc_conn_users".to_string(), - table_name: "users".to_string(), - columns: vec!["id".to_string(), "name".to_string()], - connection: grpc_conn.name.clone(), - schema: None, - refresh_config: Default::default(), - }, - Source { - name: "grpc_conn_customers".to_string(), - table_name: "customers".to_string(), - columns: vec!["id".to_string(), "name".to_string()], - connection: grpc_conn.name, - schema: None, - refresh_config: Default::default(), - }, - ], - ..Default::default() - } -} - -#[test] -fn load_multi_sources() { - let config = get_default_config(); - - let used_sources = config - .sources - .iter() - .map(|s| s.name.clone()) - .collect::>(); - - let builder = PipelineBuilder::new( - &config.connections, - &config.sources, - config.sql.as_deref(), - &config.sinks, - Default::default(), - Flags::default(), - &config.udfs, - ); - - let runtime = tokio::runtime::Builder::new_current_thread() - .enable_all() - .build() - .unwrap(); - let runtime = Arc::new(runtime); - let grouped_connections = runtime - .block_on(builder.get_grouped_tables(&runtime, &used_sources)) - .unwrap(); - - let source_builder = SourceBuilder::new(grouped_connections, Default::default()); - let (_sender, shutdown_receiver) = shutdown::new(&runtime); - let asm = runtime - .block_on(source_builder.build_source_manager(&runtime, shutdown_receiver)) - .unwrap(); - - asm.get_endpoint(&config.sources[0].name).unwrap(); - asm.get_endpoint(&config.sources[1].name).unwrap(); -} diff --git a/dozer-cli/src/pipeline/tests/mod.rs b/dozer-cli/src/pipeline/tests/mod.rs deleted file mode 100644 index 0d1e2f594c..0000000000 --- a/dozer-cli/src/pipeline/tests/mod.rs +++ /dev/null @@ -1 +0,0 @@ -mod builder; diff --git a/dozer-cli/src/pipeline/tests/schemas.json b/dozer-cli/src/pipeline/tests/schemas.json deleted file mode 100644 index 114a1dba2d..0000000000 --- a/dozer-cli/src/pipeline/tests/schemas.json +++ /dev/null @@ -1,34 +0,0 @@ -{ - "users": { - "schema": { - "fields": [ - { - "name": "id", - "typ": "Int", - "nullable": false - }, - { - "name": "name", - "typ": "String", - "nullable": true - } - ] - } - }, - "customers": { - "schema": { - "fields": [ - { - "name": "id", - "typ": "Int", - "nullable": false - }, - { - "name": "name", - "typ": "String", - "nullable": true - } - ] - } - } -} \ No newline at end of file diff --git a/dozer-cli/src/simple/build/contract/mod.rs b/dozer-cli/src/simple/build/contract/mod.rs deleted file mode 100644 index 99fe8c4e77..0000000000 --- a/dozer-cli/src/simple/build/contract/mod.rs +++ /dev/null @@ -1,209 +0,0 @@ -use std::{collections::HashMap, fs::OpenOptions, path::Path}; - -use dozer_core::{ - dag_schemas::DagSchemas, - daggy, - node::PortHandle, - petgraph::{algo::is_isomorphic_matching, visit::IntoNodeReferences}, -}; -use dozer_types::{models::connection::Connection, node::NodeHandle, types::Schema}; -use dozer_types::{ - serde::{de::DeserializeOwned, Deserialize, Serialize}, - serde_json, -}; - -use crate::errors::BuildError; - -#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)] -#[serde(crate = "dozer_types::serde")] -pub struct NodeType { - pub handle: NodeHandle, - pub kind: NodeKind, -} - -#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)] -#[serde(crate = "dozer_types::serde")] -pub enum NodeKind { - Source { - typ: String, - port_names: HashMap, - }, - Processor { - typ: String, - }, - Sink { - typ: String, - port_names: HashMap, - }, -} - -#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)] -#[serde(crate = "dozer_types::serde")] -pub struct EdgeType { - pub from_port: PortHandle, - pub to_port: PortHandle, - pub schema: Schema, -} - -#[derive(Deserialize, Serialize, Debug, Clone)] -#[serde(crate = "dozer_types::serde")] -pub struct PipelineContract(daggy::Dag); - -#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)] -#[serde(crate = "dozer_types::serde")] -pub struct Contract { - pub version: usize, - pub pipeline: PipelineContract, -} - -impl Contract { - pub fn new( - version: usize, - dag_schemas: &DagSchemas, - connections: &[Connection], - ) -> Result { - let mut source_types = HashMap::new(); - for (node_index, node) in dag_schemas.graph().node_references() { - if let dozer_core::NodeKind::Source(_) = &node.kind { - let connection = connections - .iter() - .find(|connection| connection.name == node.handle.id) - .ok_or(BuildError::MissingConnection(node.handle.id.clone()))?; - let typ = connection.config.get_type_name(); - source_types.insert(node_index, typ); - } - } - - let graph = dag_schemas.graph(); - let pipeline = graph.map( - |node_index, node| { - let handle = node.handle.clone(); - let kind = match &node.kind { - dozer_core::NodeKind::Source(source) => { - let typ = source_types - .get(&node_index) - .expect("Source must have a type") - .to_string(); - let port_names = source - .get_output_ports() - .iter() - .map(|port| { - let port_name = source.get_output_port_name(&port.handle); - (port.handle, port_name) - }) - .collect(); - NodeKind::Source { typ, port_names } - } - dozer_core::NodeKind::Processor(processor) => NodeKind::Processor { - typ: processor.type_name(), - }, - dozer_core::NodeKind::Sink(sink) => { - let typ = sink.type_name(); - let port_names = sink - .get_input_ports() - .iter() - .map(|port| { - let port_name = sink.get_input_port_name(port); - (*port, port_name) - }) - .collect(); - NodeKind::Sink { typ, port_names } - } - }; - NodeType { handle, kind } - }, - |_, edge| EdgeType { - from_port: edge.output_port, - to_port: edge.input_port, - schema: edge.schema.clone(), - }, - ); - - Ok(Self { - version, - pipeline: PipelineContract(pipeline), - }) - } - - pub fn serialize(&self, path: &Path) -> Result<(), BuildError> { - serde_json_to_path(path, &self)?; - Ok(()) - } - - pub fn deserialize(path: &Path) -> Result { - serde_json_from_path(path) - } -} - -mod service; - -fn serde_json_to_path(path: impl AsRef, value: &impl Serialize) -> Result<(), BuildError> { - let file = OpenOptions::new() - .create(true) - .write(true) - .truncate(true) - .open(path.as_ref()) - .map_err(|e| BuildError::FileSystem(path.as_ref().into(), e))?; - serde_json::to_writer_pretty(file, value).map_err(BuildError::SerdeJson) -} - -fn serde_json_from_path(path: impl AsRef) -> Result -where - T: DeserializeOwned, -{ - let file = OpenOptions::new() - .read(true) - .open(path.as_ref()) - .map_err(|e| BuildError::FileSystem(path.as_ref().into(), e))?; - serde_json::from_reader(file).map_err(BuildError::FailedToLoadExistingContract) -} - -impl PartialEq for PipelineContract { - fn eq(&self, other: &PipelineContract) -> bool { - is_isomorphic_matching( - self.0.graph(), - other.0.graph(), - |left, right| { - left.handle == right.handle - && match (&left.kind, &right.kind) { - ( - NodeKind::Source { - typ: left_typ, - port_names: left_portnames, - }, - NodeKind::Source { - typ: right_typ, - port_names: right_portnames, - }, - ) => { - if left_typ != right_typ { - false - } else { - let mut left: Vec<_> = left_portnames.values().collect(); - left.sort(); - let mut right: Vec<_> = right_portnames.values().collect(); - right.sort(); - left == right - } - } - ( - NodeKind::Processor { typ: left_typ }, - NodeKind::Processor { typ: right_typ }, - ) => left_typ == right_typ, - ( - NodeKind::Sink { - typ: left_typ, - port_names: left_port_names, - }, - NodeKind::Sink { - typ: right_type, - port_names: right_port_names, - }, - ) => left_typ == right_type && left_port_names == right_port_names, - _ => false, - } - }, - |left, right| left.schema == right.schema, - ) - } -} diff --git a/dozer-cli/src/simple/build/contract/service.rs b/dozer-cli/src/simple/build/contract/service.rs deleted file mode 100644 index 8f289c5e15..0000000000 --- a/dozer-cli/src/simple/build/contract/service.rs +++ /dev/null @@ -1,335 +0,0 @@ -use std::{collections::HashMap, fmt::Display}; - -use dozer_core::{ - daggy, - petgraph::{ - dot, - visit::{EdgeRef, IntoEdgeReferences, IntoEdgesDirected, IntoNodeReferences}, - Direction, - }, -}; -use dozer_types::grpc_types::{conversions::map_schema, types::Schema}; - -use super::{Contract, NodeKind}; - -impl Contract { - pub fn get_source_schemas(&self, connection_name: &str) -> Option> { - // Find the source node. - for (node_index, node) in self.pipeline.0.node_references() { - if let NodeKind::Source { port_names, .. } = &node.kind { - if node.handle.id == connection_name { - let mut result = HashMap::new(); - for edge in self - .pipeline - .0 - .edges_directed(node_index, Direction::Outgoing) - { - let edge = edge.weight(); - let name = port_names - .get(&edge.from_port) - .expect("Every port name must have been added") - .clone(); - let schema = edge.schema.clone(); - result.insert(name, map_schema(schema)); - } - return Some(result); - } - } - } - None - } - - pub fn get_sink_table_schemas(&self, sink_name: &str) -> Option> { - // Find the sink node. - for (node_index, node) in self.pipeline.0.node_references() { - if let NodeKind::Sink { port_names, .. } = &node.kind { - if node.handle.id == sink_name { - let mut result = HashMap::new(); - for edge in self - .pipeline - .0 - .edges_directed(node_index, Direction::Incoming) - { - let edge = edge.weight(); - let name = port_names - .get(&edge.to_port) - .expect("Every port name must have been added") - .clone(); - let schema = edge.schema.clone(); - result.insert(name, map_schema(schema)); - } - return Some(result); - } - } - } - None - } - - pub fn get_graph_schemas(&self) -> HashMap { - let graph = self.create_ui_graph(); - let nodes = graph.into_graph().into_nodes_edges().0; - nodes - .into_iter() - .filter_map(|node| { - let node = node.weight; - node.output_schema - .map(|schema| (node.kind.to_string(), schema)) - }) - .collect() - } - - pub fn generate_dot(&self) -> String { - dot::Dot::new(&self.create_ui_graph()).to_string() - } - - fn create_ui_graph(&self) -> UiGraph { - let mut ui_graph = UiGraph::new(); - let mut pipeline_node_index_to_ui_node_index = HashMap::new(); - let mut pipeline_source_to_ui_node_index = HashMap::new(); - let mut pipeline_sink_table_to_ui_node_index = HashMap::new(); - - // Create nodes. - for (node_index, node) in self.pipeline.0.node_references() { - match &node.kind { - NodeKind::Source { typ, port_names } => { - // Create connection ui node. - let connection_node_index = ui_graph.add_node(UiNodeType { - kind: UiNodeKind::Connection { - typ: typ.clone(), - name: node.handle.id.clone(), - }, - output_schema: None, - }); - pipeline_node_index_to_ui_node_index.insert(node_index, connection_node_index); - - // Create source ui node. Schema comes from connection's outgoing edge. - for edge in self - .pipeline - .0 - .edges_directed(node_index, Direction::Outgoing) - { - if let std::collections::hash_map::Entry::Vacant(entry) = - pipeline_source_to_ui_node_index - .entry((node_index, edge.weight().from_port)) - { - let edge = edge.weight(); - let schema = edge.schema.clone(); - let source_node_index = ui_graph.add_node(UiNodeType { - kind: UiNodeKind::Source { - name: port_names[&edge.from_port].clone(), - }, - output_schema: Some(map_schema(schema)), - }); - entry.insert(source_node_index); - } - } - } - NodeKind::Processor { typ } => { - // Create processor ui node. Schema comes from the outgoing edge. - let mut edges = self - .pipeline - .0 - .edges_directed(node_index, Direction::Outgoing) - .collect::>(); - assert!( - edges.len() == 1, - "We only support visualizing processors with one output port" - ); - let edge = edges.remove(0); - - let processor_node_index = ui_graph.add_node(UiNodeType { - kind: UiNodeKind::Processor { - typ: typ.clone(), - name: node.handle.id.clone(), - }, - output_schema: Some(map_schema(edge.weight().schema.clone())), - }); - pipeline_node_index_to_ui_node_index.insert(node_index, processor_node_index); - } - NodeKind::Sink { typ, port_names } => { - // Create sink ui node. - let sink_node_index = ui_graph.add_node(UiNodeType { - kind: UiNodeKind::Sink { - name: node.handle.id.clone(), - typ: typ.clone(), - }, - output_schema: None, - }); - pipeline_node_index_to_ui_node_index.insert(node_index, sink_node_index); - - // Create sink table ui node. Schema comes from sink's ingoing edge. - for edge in self - .pipeline - .0 - .edges_directed(node_index, Direction::Incoming) - { - let edge = edge.weight(); - let schema = edge.schema.clone(); - let sink_table_node_index = ui_graph.add_node(UiNodeType { - kind: UiNodeKind::SinkTable { - name: port_names[&edge.to_port].clone(), - }, - output_schema: Some(map_schema(schema)), - }); - pipeline_sink_table_to_ui_node_index - .insert((node_index, edge.to_port), sink_table_node_index); - } - } - } - } - - // Create edges. - for edge in self.pipeline.0.edge_references() { - let from_node_index = edge.source(); - let to_node_index = edge.target(); - let from_ui_node_index = pipeline_node_index_to_ui_node_index[&from_node_index]; - let to_ui_node_index = pipeline_node_index_to_ui_node_index[&to_node_index]; - - let from_node = &self.pipeline.0[from_node_index]; - let from_ui_node_index = match &from_node.kind { - NodeKind::Source { .. } => { - let ui_source_node_index = pipeline_source_to_ui_node_index - [&(from_node_index, edge.weight().from_port)]; - // Connect ui connection node to ui source node. - ui_graph - .add_edge(from_ui_node_index, ui_source_node_index, UiEdgeType) - .unwrap(); - ui_source_node_index - } - _ => pipeline_node_index_to_ui_node_index[&from_node_index], - }; - - let to_node = &self.pipeline.0[to_node_index]; - let to_ui_node_index = match &to_node.kind { - NodeKind::Sink { .. } => { - let ui_sink_table_node_index = pipeline_sink_table_to_ui_node_index - [&(to_node_index, edge.weight().to_port)]; - // Connect ui sink table node to ui sink node. - ui_graph - .add_edge(ui_sink_table_node_index, to_ui_node_index, UiEdgeType) - .unwrap(); - ui_sink_table_node_index - } - _ => pipeline_node_index_to_ui_node_index[&to_node_index], - }; - - // Connect ui node. - ui_graph - .add_edge(from_ui_node_index, to_ui_node_index, UiEdgeType) - .unwrap(); - } - - remove_from_processor(&ui_graph) - } -} - -#[derive(Debug, Clone)] -struct UiNodeType { - kind: UiNodeKind, - output_schema: Option, -} - -impl Display for UiNodeType { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - self.kind.fmt(f) - } -} - -#[derive(Debug, Clone)] -enum UiNodeKind { - Connection { typ: String, name: String }, - Source { name: String }, - Processor { typ: String, name: String }, - Sink { typ: String, name: String }, - SinkTable { name: String }, -} - -impl Display for UiNodeKind { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - UiNodeKind::Connection { typ, name } => write!(f, "connection::{typ}::{name}"), - UiNodeKind::Source { name } => write!(f, "source::source::{name}"), - UiNodeKind::Processor { typ, name } => write!(f, "processor::{typ}::{name}"), - UiNodeKind::Sink { typ, name } => write!(f, "sink::{typ}::{name}"), - UiNodeKind::SinkTable { name } => write!(f, "sink::table::{name}"), - } - } -} - -#[derive(Debug)] -struct UiEdgeType; - -impl Display for UiEdgeType { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - write!(f, "") - } -} - -type UiGraph = daggy::Dag; - -fn remove_from_processor(graph: &UiGraph) -> UiGraph { - let mut output = UiGraph::new(); - - // Create nodes that are not "from". - let mut input_node_index_to_output_node_index = HashMap::new(); - for (node_index, node) in graph.node_references() { - if !is_from(&node.kind) { - let output_node_index = output.add_node(node.clone()); - input_node_index_to_output_node_index.insert(node_index, output_node_index); - } - } - - // Map "from" nodes to its only child. - let mut input_from_node_index_to_output_node_index = HashMap::new(); - for (node_index, node) in graph.node_references() { - if is_from(&node.kind) { - let mut edges = graph - .edges_directed(node_index, Direction::Outgoing) - .collect::>(); - assert!( - edges.len() == 1, - "We only support visualizing processors with one output port" - ); - let edge = edges.remove(0); - let output_node_index = input_node_index_to_output_node_index[&edge.target()]; - input_from_node_index_to_output_node_index.insert(node_index, output_node_index); - } - } - - // Create edges. Edges pointing to "from" nodes are pointed to its only child. Edges from "from" nodes are ignored. - for edge in graph.edge_references() { - let input_source_index = edge.source(); - let input_target_index = edge.target(); - if let Some(output_source_index) = - input_node_index_to_output_node_index.get(&input_source_index) - { - if let Some(output_target_index) = - input_node_index_to_output_node_index.get(&input_target_index) - { - output - .add_edge(*output_source_index, *output_target_index, UiEdgeType) - .unwrap(); - } else { - output - .add_edge( - *output_source_index, - input_from_node_index_to_output_node_index[&input_target_index], - UiEdgeType, - ) - .unwrap(); - } - } else { - // Ignore - } - } - - output -} - -fn is_from(node_kind: &UiNodeKind) -> bool { - if let UiNodeKind::Processor { typ, .. } = node_kind { - typ == "Table" - } else { - false - } -} diff --git a/dozer-cli/src/simple/build/mod.rs b/dozer-cli/src/simple/build/mod.rs deleted file mode 100644 index f4545d0791..0000000000 --- a/dozer-cli/src/simple/build/mod.rs +++ /dev/null @@ -1,2 +0,0 @@ -mod contract; -pub use contract::{Contract, PipelineContract}; diff --git a/dozer-cli/src/simple/executor.rs b/dozer-cli/src/simple/executor.rs deleted file mode 100644 index 5e22ffb5ef..0000000000 --- a/dozer-cli/src/simple/executor.rs +++ /dev/null @@ -1,88 +0,0 @@ -use dozer_core::shutdown::ShutdownReceiver; -use dozer_tracing::DozerMonitorContext; -use dozer_types::models::flags::Flags; -use dozer_types::models::sink::Sink; -use tokio::runtime::Runtime; - -use std::sync::Arc; - -use dozer_types::models::source::Source; -use dozer_types::models::udf_config::UdfConfig; - -use crate::pipeline::PipelineBuilder; -use dozer_core::executor::{DagExecutor, ExecutorOptions}; - -use dozer_types::models::connection::Connection; - -use crate::errors::OrchestrationError; - -pub struct Executor<'a> { - connections: &'a [Connection], - sources: &'a [Source], - sql: Option<&'a str>, - sinks: &'a [Sink], - labels: DozerMonitorContext, - udfs: &'a [UdfConfig], -} - -impl<'a> Executor<'a> { - // TODO: Refactor this to not require both `contract` and all of - // connections, sources and sql - #[allow(clippy::too_many_arguments)] - pub async fn new( - connections: &'a [Connection], - sources: &'a [Source], - sql: Option<&'a str>, - sinks: &'a [Sink], - labels: DozerMonitorContext, - udfs: &'a [UdfConfig], - ) -> Result, OrchestrationError> { - Ok(Executor { - connections, - sources, - sql, - sinks, - labels, - udfs, - }) - } - - pub async fn create_dag_executor( - self, - runtime: &Arc, - executor_options: ExecutorOptions, - shutdown: ShutdownReceiver, - flags: Flags, - ) -> Result { - let builder = PipelineBuilder::new( - self.connections, - self.sources, - self.sql, - self.sinks, - self.labels.clone(), - flags, - self.udfs, - ); - - let dag = builder.build(runtime, shutdown).await?; - let exec = DagExecutor::new(dag, executor_options).await?; - - Ok(exec) - } -} - -pub fn run_dag_executor( - runtime: &Arc, - dag_executor: DagExecutor, - shutdown: ShutdownReceiver, - labels: DozerMonitorContext, -) -> Result<(), OrchestrationError> { - let join_handle = runtime.block_on(dag_executor.start( - Box::pin(shutdown.create_shutdown_future()), - labels, - runtime.clone(), - ))?; - join_handle - .join() - .map_err(OrchestrationError::ExecutionError) -} diff --git a/dozer-cli/src/simple/helper.rs b/dozer-cli/src/simple/helper.rs deleted file mode 100644 index 340010c649..0000000000 --- a/dozer-cli/src/simple/helper.rs +++ /dev/null @@ -1,28 +0,0 @@ -use crate::console_helper::get_colored_text; -use crate::console_helper::PURPLE; -use crate::errors::OrchestrationError; -use dozer_types::log::info; -use dozer_types::models::config::default_home_dir; -use dozer_types::models::config::Config; -use dozer_types::models::sink::Sink; - -pub fn validate_config(config: &Config) -> Result<(), OrchestrationError> { - info!( - "Data directory: {}", - get_colored_text( - &config.home_dir.clone().unwrap_or_else(default_home_dir), - PURPLE - ) - ); - validate_sinks(&config.sinks)?; - - Ok(()) -} - -pub fn validate_sinks(sinks: &[Sink]) -> Result<(), OrchestrationError> { - if sinks.is_empty() { - return Err(OrchestrationError::EmptySinks); - } - - Ok(()) -} diff --git a/dozer-cli/src/simple/mod.rs b/dozer-cli/src/simple/mod.rs deleted file mode 100644 index 1d453045d0..0000000000 --- a/dozer-cli/src/simple/mod.rs +++ /dev/null @@ -1,6 +0,0 @@ -mod executor; -pub mod orchestrator; -pub use orchestrator::SimpleOrchestrator; -mod build; -pub use build::{Contract, PipelineContract}; -pub mod helper; diff --git a/dozer-cli/src/simple/orchestrator.rs b/dozer-cli/src/simple/orchestrator.rs deleted file mode 100644 index e587c8d7d4..0000000000 --- a/dozer-cli/src/simple/orchestrator.rs +++ /dev/null @@ -1,277 +0,0 @@ -use super::executor::{run_dag_executor, Executor}; -use super::Contract; -use crate::errors::{BuildError, OrchestrationError}; -use crate::home_dir::{BuildId, HomeDir}; -use crate::pipeline::connector_source::ConnectorSourceFactoryError; -use crate::pipeline::PipelineBuilder; -use crate::simple::build; -use crate::simple::helper::validate_config; -use crate::utils::get_executor_options; - -use crate::flatten_join_handle; -use camino::Utf8PathBuf; -use dozer_core::app::AppPipeline; -use dozer_core::dag_schemas::DagSchemas; -use dozer_core::event::EventHub; -use dozer_core::shutdown::ShutdownReceiver; -use dozer_tracing::DozerMonitorContext; -use dozer_types::constants::LOCK_FILE; -use futures::future::{select, Either}; - -use crate::console_helper::get_colored_text; -use crate::console_helper::GREEN; -use crate::console_helper::PURPLE; -use crate::console_helper::RED; -use dozer_core::errors::ExecutionError; -use dozer_ingestion::{get_connector, SourceSchema, TableInfo}; -use dozer_sql::builder::statement_to_pipeline; -use dozer_sql::errors::PipelineError; -use dozer_types::log::info; -use dozer_types::models::config::{default_home_dir, Config}; -use dozer_types::tracing::error; -use futures::stream::FuturesUnordered; -use futures::{FutureExt, StreamExt}; -use std::collections::{HashMap, HashSet}; -use std::fs; - -use std::sync::Arc; -use tokio::runtime::Runtime; -use tokio::sync::oneshot; - -#[derive(Clone)] -pub struct SimpleOrchestrator { - pub base_directory: Utf8PathBuf, - pub config: Config, - pub runtime: Arc, - pub labels: DozerMonitorContext, -} - -impl SimpleOrchestrator { - pub fn new( - base_directory: Utf8PathBuf, - config: Config, - runtime: Arc, - labels: DozerMonitorContext, - ) -> Self { - Self { - base_directory, - config, - runtime, - labels, - } - } - - pub fn home_dir(&self) -> Utf8PathBuf { - self.base_directory.join( - self.config - .home_dir - .clone() - .unwrap_or_else(default_home_dir), - ) - } - - pub fn lockfile_path(&self) -> Utf8PathBuf { - lockfile_path(self.base_directory.clone()) - } - - pub async fn run_apps( - &self, - shutdown: ShutdownReceiver, - api_notifier: Option>, - ) -> Result<(), OrchestrationError> { - let executor = Executor::new( - &self.config.connections, - &self.config.sources, - self.config.sql.as_deref(), - &self.config.sinks, - self.labels.clone(), - &self.config.udfs, - ) - .await?; - let dag_executor = executor - .create_dag_executor( - &self.runtime, - get_executor_options(&self.config), - shutdown.clone(), - self.config.flags.clone(), - ) - .await?; - - if let Some(api_notifier) = api_notifier { - api_notifier.send(()).expect("Failed to notify API server"); - } - - let labels = self.labels.clone(); - let runtime_clone = self.runtime.clone(); - let shutdown_clone = shutdown.clone(); - let pipeline_future = self.runtime.spawn_blocking(move || { - run_dag_executor(&runtime_clone, dag_executor, shutdown_clone, labels) - }); - - let mut futures = FuturesUnordered::new(); - futures.push(flatten_join_handle(pipeline_future).boxed()); - - while let Some(result) = futures.next().await { - result?; - } - Ok(()) - } - - #[allow(clippy::type_complexity)] - pub async fn list_connectors( - &self, - connections: HashSet, - ) -> Result, Vec)>, OrchestrationError> { - let mut schema_map = HashMap::new(); - for connection in self - .config - .connections - .iter() - .filter(|conn| connections.contains(&conn.name)) - { - // We're not really going to start ingestion, so passing `None` as state here is OK. - let mut connector = get_connector( - self.runtime.clone(), - EventHub::new(1), - connection.clone(), - None, - ) - .map_err(|e| ConnectorSourceFactoryError::Connector(e.into()))?; - let schema_tuples = connector - .list_all_schemas() - .await - .map_err(ConnectorSourceFactoryError::Connector)?; - schema_map.insert(connection.name.clone(), schema_tuples); - } - - Ok(schema_map) - } - - pub async fn build( - &self, - force: bool, - shutdown: ShutdownReceiver, - locked: bool, - ) -> Result<(), OrchestrationError> { - let home_dir = self.home_dir(); - let home_dir = HomeDir::new(home_dir); - - info!( - "Initializing app: {}", - get_colored_text(&self.config.app_name, PURPLE) - ); - if force { - self.clean()?; - } - validate_config(&self.config)?; - - let builder = PipelineBuilder::new( - &self.config.connections, - &self.config.sources, - self.config.sql.as_deref(), - &self.config.sinks, - self.labels.clone(), - self.config.flags.clone(), - &self.config.udfs, - ); - let dag = builder.build(&self.runtime, shutdown).await?; - // Populate schemas. - let dag_schemas = DagSchemas::new(dag).await?; - - // Get current contract. - let version = self.config.version as usize; - - let contract = build::Contract::new(version, &dag_schemas, &self.config.connections)?; - - let contract_path = self.lockfile_path(); - if locked { - let existing_contract = Contract::deserialize(contract_path.as_std_path()).ok(); - let Some(existing_contract) = existing_contract.as_ref() else { - return Err(OrchestrationError::LockedNoLockFile); - }; - - if &contract != existing_contract { - return Err(OrchestrationError::LockedOutdatedLockfile); - } - } - - home_dir - .create_build_dir_all(BuildId::first()) - .map_err(|(path, error)| BuildError::FileSystem(path.into(), error))?; - - contract.serialize(contract_path.as_std_path())?; - - Ok(()) - } - - // Cleaning the entire folder as there will be inconsistencies - // between pipeline, cache and generated proto files. - pub fn clean(&self) -> Result<(), OrchestrationError> { - let home_dir = self.home_dir(); - if home_dir.exists() { - fs::remove_dir_all(&home_dir) - .map_err(|e| ExecutionError::FileSystemError(home_dir.into_std_path_buf(), e))?; - }; - - Ok(()) - } - - pub async fn run_all( - &self, - shutdown: ShutdownReceiver, - locked: bool, - ) -> Result<(), OrchestrationError> { - let (tx, rx) = oneshot::channel::<()>(); - - self.build(false, shutdown.clone(), locked).await?; - - let dozer_pipeline = self.clone(); - let pipeline_shutdown = shutdown.clone(); - let pipeline_future = - async move { dozer_pipeline.run_apps(pipeline_shutdown, Some(tx)).await }.boxed(); - - match select(rx, pipeline_future).await { - Either::Left((result, pipeline_future)) => { - if result.is_err() { - // Pipeline panics before ready. Propagate the panic. - pipeline_future.await.unwrap(); - unreachable!("we must have panicked"); - } else { - Ok(()) - } - } - Either::Right((result, _)) => result, - } - } -} - -pub fn validate_sql(sql: String, runtime: Arc) -> Result<(), PipelineError> { - statement_to_pipeline( - &sql, - &mut AppPipeline::new_with_default_flags(), - None, - vec![], - runtime, - ) - .map_or_else( - |e| { - error!( - "[sql][{}] Transforms validation error: {}", - get_colored_text("X", RED), - e - ); - Err(e) - }, - |_| { - info!( - "[sql][{}] Transforms validation completed", - get_colored_text("✓", GREEN) - ); - Ok(()) - }, - ) -} - -pub fn lockfile_path(base_directory: Utf8PathBuf) -> Utf8PathBuf { - base_directory.join(LOCK_FILE) -} diff --git a/dozer-cli/src/tests.rs b/dozer-cli/src/tests.rs deleted file mode 100644 index 5c462b6db6..0000000000 --- a/dozer-cli/src/tests.rs +++ /dev/null @@ -1,71 +0,0 @@ -use crate::config_helper::add_file_content_to_config; -use dozer_types::models::config::Config; -use dozer_types::serde_yaml; -use dozer_types::serde_yaml::Mapping; - -#[test] -fn test_sql_merge_in_config() { - let query_a = "select * from table_a"; - let query_b = "select * from table_b"; - - let yaml = format!( - r#" - app_name: dozer-config-sample - version: 1 - sql: - {} - "#, - query_a - ); - - let mut combined_yaml = serde_yaml::Value::Mapping(Mapping::new()); - - add_file_content_to_config(&mut combined_yaml, "config.yaml", yaml.into()).unwrap(); - add_file_content_to_config(&mut combined_yaml, "query.sql", query_b.into()).unwrap(); - - let config = serde_yaml::from_value::(combined_yaml).unwrap(); - - assert_eq!(config.sql, Some(format!("{};{}", query_a, query_b))); -} - -#[test] -fn test_sql_from_single_sql_source_in_config() { - let query = "select * from table_b"; - - let yaml = r#" - app_name: dozer-config-sample - version: 1 - "#; - - let mut combined_yaml = serde_yaml::Value::Mapping(Mapping::new()); - - add_file_content_to_config(&mut combined_yaml, "config.yaml", yaml.into()).unwrap(); - add_file_content_to_config(&mut combined_yaml, "query.sql", query.into()).unwrap(); - - let config = serde_yaml::from_value::(combined_yaml).unwrap(); - - assert_eq!(config.sql, Some(query.to_string())); -} - -#[test] -fn test_sql_from_single_yaml_source_in_config() { - let query = "select * from table_b"; - - let yaml = format!( - r#" - app_name: dozer-config-sample - version: 1 - sql: - {} - "#, - query - ); - - let mut combined_yaml = serde_yaml::Value::Mapping(Mapping::new()); - - add_file_content_to_config(&mut combined_yaml, "config.yaml", yaml.into()).unwrap(); - - let config = serde_yaml::from_value::(combined_yaml).unwrap(); - - assert_eq!(config.sql, Some(query.to_string())); -} diff --git a/dozer-cli/src/ui/app/errors.rs b/dozer-cli/src/ui/app/errors.rs deleted file mode 100644 index 03985a67b8..0000000000 --- a/dozer-cli/src/ui/app/errors.rs +++ /dev/null @@ -1,62 +0,0 @@ -use crate::errors::{BuildError, CliError, OrchestrationError}; -use crate::ui::downloader::DownloaderError; -use dozer_core::errors::ExecutionError; -use dozer_sql::errors::PipelineError; - -use dozer_types::thiserror; -use dozer_types::thiserror::Error; -use zip::result::ZipError; - -#[derive(Error, Debug)] -pub enum AppUIError { - #[error("IO error: {0}")] - Io(#[from] std::io::Error), - #[error("Notify error: {0}")] - Notify(#[from] notify::Error), - #[error(transparent)] - CliError(#[from] CliError), - - #[error("Cannot pull docker image: {0}")] - CannotPullDockerImage(String), - #[error("Cannot run docker image: {0}")] - CannotRunDockerImage(String), - #[error("Docker not installed")] - DockerNotInstalled, - #[error("Cannot stop docker container: {0}")] - CannotStopDockerContainer(String), - #[error("Cannot remove docker container: {0}")] - CannotRemoveDockerContainer(String), - - #[error("Dozer is not initialized")] - NotInitialized, - #[error("Connection {0} not found")] - ConnectionNotFound(String), - #[error("Sink {0} not found")] - SinkNotFound(String), - #[error("Error in initializing app ui server: {0}")] - Transport(#[from] tonic::transport::Error), - #[error("Error in reading or extracting from Zip file: {0}")] - ZipError(#[from] ZipError), - #[error("Reqwest error: {0}")] - Reqwest(#[from] reqwest::Error), - #[error("Cannot start ui server: {0}")] - CannotStartUiServer(#[source] std::io::Error), - - #[error(transparent)] - Build(#[from] BuildError), - #[error(transparent)] - PipelineError(#[from] PipelineError), - #[error(transparent)] - ExecutionError(#[from] ExecutionError), - #[error(transparent)] - OrchestrationError(Box), - - #[error(transparent)] - DownloaderError(#[from] DownloaderError), -} - -impl From for AppUIError { - fn from(error: OrchestrationError) -> Self { - AppUIError::OrchestrationError(Box::new(error)) - } -} diff --git a/dozer-cli/src/ui/app/mod.rs b/dozer-cli/src/ui/app/mod.rs deleted file mode 100644 index 4e15b01ff4..0000000000 --- a/dozer-cli/src/ui/app/mod.rs +++ /dev/null @@ -1,67 +0,0 @@ -mod errors; -mod server; -mod state; -mod watcher; -use crate::ui::{ - app::{server::APP_UI_PORT, state::AppUIState}, - downloader::{self, LOCAL_APP_UI_DIR}, -}; -use dozer_core::shutdown::ShutdownReceiver; -use dozer_types::{grpc_types::app_ui::ConnectResponse, log::info}; -pub use errors::AppUIError; -use futures::stream::{AbortHandle, Abortable}; -use std::sync::Arc; -use tokio::runtime::Runtime; - -const APP_UI_WEB_PORT: u16 = 62888; - -pub async fn start_app_ui_server( - runtime: &Arc, - shutdown: ShutdownReceiver, - disable_ui: bool, -) -> Result<(), AppUIError> { - let (sender, receiver) = tokio::sync::broadcast::channel::(100); - let state = Arc::new(AppUIState::new()); - state.set_sender(sender.clone()).await; - // Ignore if build fails - let res = state.build(runtime.clone()).await; - if let Err(e) = res { - info!("Failed to build state : {}", e); - } - let state2: Arc = state.clone(); - if !disable_ui { - info!("Check if latest app ui code is available"); - let already_exist = downloader::validate_if_dozer_app_ui_code_exists(); - if !already_exist { - info!("There's no ui code folder, fetching latest app ui code"); - downloader::fetch_latest_dozer_app_ui_code().await?; - } - let react_app_server: actix_web::dev::Server = - downloader::start_react_app(APP_UI_WEB_PORT, LOCAL_APP_UI_DIR) - .map_err(AppUIError::CannotStartUiServer)?; - tokio::spawn(react_app_server); - let browser_url: String = format!("http://localhost:{}", APP_UI_WEB_PORT); - info!("Starting ui on : {}", browser_url); - if webbrowser::open(&browser_url).is_err() { - info!("Failed to open browser. "); - } - } - info!("Starting app ui server on port : {}", APP_UI_PORT); - let rshudown = shutdown.clone(); - tokio::spawn(async { - let (abort_handle, abort_registration) = AbortHandle::new_pair(); - tokio::spawn(async move { - rshudown.create_shutdown_future().await; - abort_handle.abort(); - }); - let res: Result<(), AppUIError> = - match Abortable::new(server::serve(receiver, state2), abort_registration).await { - Ok(result) => result.map_err(AppUIError::Transport), - Err(_) => Ok(()), - }; - - res.unwrap(); - }); - watcher::watch(runtime, state.clone(), shutdown).await?; - Ok(()) -} diff --git a/dozer-cli/src/ui/app/server.rs b/dozer-cli/src/ui/app/server.rs deleted file mode 100644 index 08680d4fd0..0000000000 --- a/dozer-cli/src/ui/app/server.rs +++ /dev/null @@ -1,193 +0,0 @@ -use dozer_types::{ - grpc_types::{ - app_ui::{ - code_service_server::{CodeService, CodeServiceServer}, - ConnectResponse, RunRequest, RunResponse, - }, - contract::{ - contract_service_server::{ContractService, ContractServiceServer}, - CommonRequest, DotResponse, SinkTablesRequest, SourcesRequest, - }, - types::SchemasResponse, - }, - log::info, -}; -use futures::stream::BoxStream; -use std::sync::Arc; -use tokio::sync::broadcast::Receiver; - -use super::state::AppUIState; -use dozer_types::tracing::Level; -use tokio_stream::wrappers::ReceiverStream; -use tonic::{Request, Response, Status}; -use tower_http::trace::{self, TraceLayer}; -pub const APP_UI_PORT: u16 = 4555; - -struct ContractServer { - state: Arc, -} - -#[tonic::async_trait] -impl ContractService for ContractServer { - async fn sources( - &self, - request: Request, - ) -> Result, Status> { - let req = request.into_inner(); - let res = self.state.get_source_schemas(req.connection_name).await; - match res { - Ok(res) => Ok(Response::new(res)), - Err(e) => Err(Status::internal(e.to_string())), - } - } - - async fn sink_tables( - &self, - request: Request, - ) -> Result, Status> { - let req = request.into_inner(); - let res = self.state.get_sink_table_schemas(req.sink_name).await; - match res { - Ok(res) => Ok(Response::new(res)), - Err(e) => Err(Status::internal(e.to_string())), - } - } - - async fn generate_dot( - &self, - _request: Request, - ) -> Result, Status> { - let state = self.state.clone(); - let res = state.generate_dot().await; - - match res { - Ok(res) => Ok(Response::new(res)), - Err(e) => Err(Status::internal(e.to_string())), - } - } - - async fn get_graph_schemas( - &self, - _request: Request, - ) -> Result, Status> { - let state = self.state.clone(); - let res = state.get_graph_schemas().await; - - match res { - Ok(res) => Ok(Response::new(res)), - Err(e) => Err(Status::internal(e.to_string())), - } - } -} - -struct AppUiServer { - receiver: Receiver, - state: Arc, -} - -impl AppUiServer { - pub fn new(receiver: Receiver, state: Arc) -> AppUiServer { - Self { receiver, state } - } - async fn start(&self, req: RunRequest) -> Result, Status> { - let state = self.state.clone(); - info!("Starting dozer"); - match state.run(req).await { - Ok(application_id) => Ok(Response::new(RunResponse { application_id })), - Err(e) => Err(Status::internal(e.to_string())), - } - } -} - -#[tonic::async_trait] -impl CodeService for AppUiServer { - type AppUIConnectStream = BoxStream<'static, Result>; - - async fn app_ui_connect( - &self, - _request: Request<()>, - ) -> Result, Status> { - let (tx, rx) = tokio::sync::mpsc::channel(1); - let mut receiver = self.receiver.resubscribe(); - - let initial_state = self.state.clone(); - tokio::spawn(async move { - let initial_state = initial_state.get_current().await; - if let Err(e) = tx - .send(Ok(ConnectResponse { - app_ui: Some(initial_state), - build: None, - })) - .await - { - info!("Error getting initial state"); - info!("{}", e.to_string()); - return {}; - } - loop { - let Ok(connect_response) = receiver.recv().await else { - break; - }; - if tx.send(Ok(connect_response)).await.is_err() { - break; - } - } - }); - let stream = ReceiverStream::new(rx); - - Ok(Response::new(Box::pin(stream) as Self::AppUIConnectStream)) - } - - async fn run(&self, request: Request) -> Result, Status> { - let req = request.into_inner(); - self.start(req).await - } - - async fn stop(&self, _request: Request<()>) -> Result, Status> { - let state = self.state.clone(); - info!("Stopping dozer"); - match state.stop().await { - Ok(()) => Ok(Response::new(())), - Err(e) => Err(Status::internal(e.to_string())), - } - } -} - -pub async fn serve( - receiver: Receiver, - state: Arc, -) -> Result<(), tonic::transport::Error> { - let addr = format!("0.0.0.0:{APP_UI_PORT}").parse().unwrap(); - let contract_server = ContractServer { - state: state.clone(), - }; - let app_ui_server = AppUiServer::new(receiver, state); - let contract_service = ContractServiceServer::new(contract_server); - let code_service = CodeServiceServer::new(app_ui_server); - // Enable CORS for local development - let contract_service = tonic_web::enable(contract_service); - let code_service = tonic_web::enable(code_service); - - let reflection_service = tonic_reflection::server::Builder::configure() - .register_encoded_file_descriptor_set( - dozer_types::grpc_types::contract::FILE_DESCRIPTOR_SET, - ) - .register_encoded_file_descriptor_set(dozer_types::grpc_types::app_ui::FILE_DESCRIPTOR_SET) - .build() - .unwrap(); - - tonic::transport::Server::builder() - .layer( - TraceLayer::new_for_http() - .make_span_with(trace::DefaultMakeSpan::new().level(Level::INFO)) - .on_response(trace::DefaultOnResponse::new().level(Level::INFO)) - .on_failure(trace::DefaultOnFailure::new().level(Level::ERROR)), - ) - .accept_http1(true) - .concurrency_limit_per_connection(32) - .add_service(contract_service) - .add_service(code_service) - .add_service(reflection_service) - .serve(addr) - .await -} diff --git a/dozer-cli/src/ui/app/state.rs b/dozer-cli/src/ui/app/state.rs deleted file mode 100644 index b4ee1e42fc..0000000000 --- a/dozer-cli/src/ui/app/state.rs +++ /dev/null @@ -1,390 +0,0 @@ -use std::{collections::HashMap, sync::Arc, thread::JoinHandle}; - -use clap::Parser; - -use dozer_core::shutdown::{self, ShutdownReceiver, ShutdownSender}; -use dozer_core::{dag_schemas::DagSchemas, Dag}; -use dozer_tracing::DozerMonitorContext; -use dozer_types::{ - grpc_types::{ - app_ui::{AppUi, AppUiResponse, BuildResponse, BuildStatus, ConnectResponse, RunRequest}, - contract::DotResponse, - types::SchemasResponse, - }, - log::info, - models::{ - api_config::{ApiConfig, AppGrpcOptions, GrpcApiOptions, RestApiOptions}, - api_security::ApiSecurity, - flags::Flags, - }, -}; -use tempfile::TempDir; -use tokio::{runtime::Runtime, sync::RwLock}; - -use super::AppUIError; -use crate::{ - cli::{init_config, init_dozer, types::Cli}, - errors::OrchestrationError, - pipeline::PipelineBuilder, - simple::{helper::validate_config, Contract, SimpleOrchestrator}, -}; -struct DozerAndContract { - dozer: SimpleOrchestrator, - contract: Option, -} - -pub struct ShutdownAndTempDir { - shutdown: ShutdownSender, - _temp_dir: TempDir, -} - -#[derive(Debug)] -pub enum BroadcastType { - Start, - Success, - Failed(String), -} -pub struct AppUIState { - dozer: RwLock>, - run_thread: RwLock>, - error_message: RwLock>, - sender: RwLock>>, -} - -impl Default for AppUIState { - fn default() -> Self { - Self::new() - } -} -impl AppUIState { - pub fn new() -> Self { - Self { - dozer: RwLock::new(None), - run_thread: RwLock::new(None), - sender: RwLock::new(None), - error_message: RwLock::new(None), - } - } - - async fn create_contract_if_missing(&self) -> Result<(), AppUIError> { - let mut dozer_and_contract_lock = self.dozer.write().await; - if let Some(dozer_and_contract) = dozer_and_contract_lock.as_mut() { - if dozer_and_contract.contract.is_none() { - let contract = create_contract(dozer_and_contract.dozer.clone()).await?; - dozer_and_contract.contract = Some(contract); - } - } - Ok(()) - } - - pub async fn set_sender(&self, sender: tokio::sync::broadcast::Sender) { - *self.sender.write().await = Some(sender); - } - - pub async fn broadcast(&self, broadcast_type: BroadcastType) { - let sender = self.sender.read().await; - info!("Broadcasting state: {:?}", broadcast_type); - if let Some(sender) = sender.as_ref() { - let res = match broadcast_type { - BroadcastType::Start => ConnectResponse { - app_ui: None, - build: Some(BuildResponse { - status: BuildStatus::BuildStart as i32, - message: None, - }), - }, - BroadcastType::Failed(msg) => ConnectResponse { - app_ui: None, - build: Some(BuildResponse { - status: BuildStatus::BuildFailed as i32, - message: Some(msg), - }), - }, - BroadcastType::Success => { - let res = self.get_current().await; - ConnectResponse { - app_ui: Some(res), - build: None, - } - } - }; - let _ = sender.send(res); - } - } - - pub async fn set_error_message(&self, error_message: Option) { - *self.error_message.write().await = error_message; - } - - pub async fn build(&self, runtime: Arc) -> Result<(), AppUIError> { - // Taking lock to ensure that we don't have multiple builds running at the same time - let mut lock = self.dozer.write().await; - - let cli = Cli::parse(); - let (config, _) = init_config( - cli.config_paths.clone(), - cli.config_token.clone(), - cli.config_overrides.clone(), - cli.ignore_pipe, - ) - .await?; - - let dozer = init_dozer(runtime, config, Default::default())?; - - let contract = create_contract(dozer.clone()).await; - *lock = Some(DozerAndContract { - dozer, - contract: match &contract { - Ok(contract) => Some(contract.clone()), - Err(_) => None, - }, - }); - if let Err(e) = &contract { - self.set_error_message(Some(e.to_string())).await; - } else { - self.set_error_message(None).await; - } - - contract - .map(|_| ()) - .map_err(|e| AppUIError::OrchestrationError(Box::new(e))) - } - pub async fn get_current(&self) -> AppUiResponse { - let dozer = self.dozer.read().await; - let app = dozer.as_ref().map(|dozer| { - let config = &dozer.dozer.config; - let connections_in_source: Vec = config - .sources - .iter() - .map(|source| source.connection.clone()) - .collect::>() - .into_iter() - .collect(); - let sink_names: Vec = - config.sinks.iter().map(|sink| sink.name.clone()).collect(); - - let enable_api_security = std::env::var("DOZER_MASTER_SECRET") - .ok() - .map(ApiSecurity::Jwt) - .as_ref() - .or(dozer.dozer.config.api.api_security.as_ref()) - .is_some(); - AppUi { - app_name: dozer.dozer.config.app_name.clone(), - connections: connections_in_source, - sink_names, - enable_api_security, - } - }); - AppUiResponse { - initialized: app.is_some(), - running: self.run_thread.read().await.is_some(), - error_message: self.error_message.read().await.as_ref().cloned(), - app, - } - } - - pub async fn get_sink_table_schemas( - &self, - sink_name: String, - ) -> Result { - self.create_contract_if_missing().await?; - let dozer = self.dozer.read().await; - let contract = get_contract(&dozer)?; - - contract - .get_sink_table_schemas(&sink_name) - .ok_or(AppUIError::SinkNotFound(sink_name)) - .map(|schemas| SchemasResponse { - schemas, - errors: HashMap::new(), - }) - } - pub async fn get_source_schemas( - &self, - connection_name: String, - ) -> Result { - self.create_contract_if_missing().await?; - let dozer = self.dozer.read().await; - let contract = get_contract(&dozer)?; - - contract - .get_source_schemas(&connection_name) - .ok_or(AppUIError::ConnectionNotFound(connection_name)) - .map(|schemas| SchemasResponse { - schemas, - errors: HashMap::new(), - }) - } - - pub async fn get_graph_schemas(&self) -> Result { - self.create_contract_if_missing().await?; - let dozer = self.dozer.read().await; - let contract = get_contract(&dozer)?; - - Ok(SchemasResponse { - schemas: contract.get_graph_schemas(), - errors: HashMap::new(), - }) - } - - pub async fn generate_dot(&self) -> Result { - self.create_contract_if_missing().await?; - let dozer = self.dozer.read().await; - let contract = get_contract(&dozer)?; - - Ok(DotResponse { - dot: contract.generate_dot(), - }) - } - - pub async fn run(&self, request: RunRequest) -> Result { - let dozer = self.dozer.read().await; - let dozer = &dozer.as_ref().ok_or(AppUIError::NotInitialized)?.dozer; - // kill if a handle already exists - self.stop().await?; - let temp_dir = tempfile::Builder::new() - .prefix("dozer_app_local") - .tempdir()?; - let temp_dir_path = temp_dir.path().to_str().unwrap(); - - let application_id = uuid::Uuid::new_v4().to_string(); - let (shutdown_sender, shutdown_receiver) = shutdown::new(&dozer.runtime); - let _handle = run( - dozer.clone(), - application_id.clone(), - request, - shutdown_receiver, - temp_dir_path, - )?; - - let mut lock = self.run_thread.write().await; - if let Some(shutdown_and_tempdir) = lock.take() { - shutdown_and_tempdir.shutdown.shutdown(); - } - let shutdown_and_tempdir = ShutdownAndTempDir { - shutdown: shutdown_sender, - _temp_dir: temp_dir, - }; - *lock = Some(shutdown_and_tempdir); - Ok(application_id) - } - - pub async fn stop(&self) -> Result<(), AppUIError> { - let mut lock = self.run_thread.write().await; - if let Some(shutdown_and_tempdir) = lock.take() { - shutdown_and_tempdir.shutdown.shutdown(); - shutdown_and_tempdir._temp_dir.close()?; - } - *lock = None; - Ok(()) - } -} - -fn get_contract(dozer_and_contract: &Option) -> Result<&Contract, AppUIError> { - dozer_and_contract - .as_ref() - .ok_or(AppUIError::NotInitialized)? - .contract - .as_ref() - .ok_or(AppUIError::NotInitialized) -} - -pub async fn create_contract(dozer: SimpleOrchestrator) -> Result { - let dag = create_dag(&dozer).await?; - let version = dozer.config.version; - let schemas = DagSchemas::new(dag).await?; - let contract = Contract::new(version as usize, &schemas, &dozer.config.connections)?; - Ok(contract) -} - -pub async fn create_dag(dozer: &SimpleOrchestrator) -> Result { - let builder = PipelineBuilder::new( - &dozer.config.connections, - &dozer.config.sources, - dozer.config.sql.as_deref(), - &dozer.config.sinks, - Default::default(), - Flags::default(), - &dozer.config.udfs, - ); - let (_shutdown_sender, shutdown_receiver) = shutdown::new(&dozer.runtime); - builder.build(&dozer.runtime, shutdown_receiver).await -} - -fn run( - dozer: SimpleOrchestrator, - application_id: String, - request: RunRequest, - shutdown_receiver: ShutdownReceiver, - temp_dir: &str, -) -> Result, OrchestrationError> { - let dozer = get_dozer_run_instance(dozer, application_id, request, temp_dir)?; - - validate_config(&dozer.config)?; - let runtime = dozer.runtime.clone(); - - let handle: JoinHandle<()> = std::thread::spawn(move || { - runtime.block_on(async move { dozer.run_all(shutdown_receiver, false).await.unwrap() }); - }); - - Ok(handle) -} - -fn get_dozer_run_instance( - mut dozer: SimpleOrchestrator, - application_id: String, - req: RunRequest, - temp_dir: &str, -) -> Result { - match req.request { - Some(dozer_types::grpc_types::app_ui::run_request::Request::Sql(req)) => { - //overwrite sql - dozer.config.sql = Some(req.sql); - dozer.config.sinks = vec![]; - } - Some(dozer_types::grpc_types::app_ui::run_request::Request::Source(_req)) => { - dozer.config.sql = None; - dozer.config.sinks = vec![]; - } - None => {} - }; - - override_api_config(&mut dozer.config.api); - - dozer.config.flags.enable_app_checkpoints = Some(false); - - dozer.config.home_dir = Some(temp_dir.to_string()); - - dozer.labels = DozerMonitorContext::new(application_id, Default::default(), false); - - Ok(dozer) -} - -fn override_api_config(api: &mut ApiConfig) { - override_rest_config(&mut api.rest); - override_grpc_config(&mut api.grpc); - override_app_grpc_config(&mut api.app_grpc); - api.pgwire.enabled = Some(true); -} - -fn override_rest_config(rest: &mut RestApiOptions) { - rest.host = Some("0.0.0.0".to_string()); - rest.port = Some(62885); - rest.cors = Some(true); - rest.enabled = Some(true); - rest.enable_sql = Some(true); -} - -fn override_grpc_config(grpc: &mut GrpcApiOptions) { - grpc.host = Some("0.0.0.0".to_string()); - grpc.port = Some(62887); - grpc.cors = Some(true); - grpc.web = Some(true); - grpc.enabled = Some(true); -} - -fn override_app_grpc_config(app_grpc: &mut AppGrpcOptions) { - app_grpc.port = Some(62997); - app_grpc.host = Some("0.0.0.0".to_string()); -} diff --git a/dozer-cli/src/ui/app/watcher.rs b/dozer-cli/src/ui/app/watcher.rs deleted file mode 100644 index f8ddb369c8..0000000000 --- a/dozer-cli/src/ui/app/watcher.rs +++ /dev/null @@ -1,80 +0,0 @@ -use std::{sync::Arc, time::Duration}; - -use super::{ - state::{AppUIState, BroadcastType}, - AppUIError, -}; - -use dozer_core::shutdown::ShutdownReceiver; -use dozer_types::log::info; -use notify::{RecursiveMode, Watcher}; -use notify_debouncer_full::new_debouncer; -use tokio::{runtime::Runtime, select}; - -pub async fn watch( - runtime: &Arc, - state: Arc, - shutdown: ShutdownReceiver, -) -> Result<(), AppUIError> { - // setup debouncer - let (tx, rx) = std::sync::mpsc::channel(); - - let dir: std::path::PathBuf = std::env::current_dir()?; - let mut debouncer = new_debouncer(Duration::from_millis(500), None, tx)?; - debouncer - .cache() - .add_root(dir.as_path(), RecursiveMode::Recursive); - let watcher = debouncer.watcher(); - - watcher.watch(dir.as_path(), RecursiveMode::NonRecursive)?; - - let additional_paths = vec![dir.join("sql")]; - - for path in additional_paths { - let _ = watcher.watch(path.as_path(), RecursiveMode::NonRecursive); - } - - let (async_sender, mut async_receiver) = tokio::sync::mpsc::channel(10); - - // Thread that adapts the sync watcher channel to an async channel - let adapter = runtime.spawn_blocking(move || loop { - let res = rx.recv(); - let Ok(msg) = res else { - break; - }; - let _ = async_sender.blocking_send(msg); - }); - - loop { - select! { - Some(msg) = async_receiver.recv() => match msg { - Ok(_events) => { - build(runtime.clone(), state.clone()).await; - } - Err(errors) => errors.iter().for_each(|error| info!("{error:?}")), - }, - // We are shutting down - _ = shutdown.create_shutdown_future() => break, - // The watcher quit - else => break - } - } - - // Drop the channels that may keep the adapter thread alive - drop(async_receiver); - drop(debouncer); - - let _ = adapter.await; - - Ok(()) -} - -async fn build(runtime: Arc, state: Arc) { - state.broadcast(BroadcastType::Start).await; - if let Err(res) = state.build(runtime).await { - let message = res.to_string(); - state.broadcast(BroadcastType::Failed(message)).await; - } else { - state.broadcast(BroadcastType::Success).await; - } -} diff --git a/dozer-cli/src/ui/downloader.rs b/dozer-cli/src/ui/downloader.rs deleted file mode 100644 index 12ed0f9791..0000000000 --- a/dozer-cli/src/ui/downloader.rs +++ /dev/null @@ -1,235 +0,0 @@ -use actix_files::NamedFile; -use dozer_types::log::info; -use dozer_types::thiserror; -use dozer_types::thiserror::Error; -use std::env; -use std::fs::remove_dir_all; -use std::fs::remove_file; -use std::fs::File; -use std::io::Read; -use std::io::Seek; -use std::io::Write; -use std::path::Path; -use std::path::PathBuf; -use zip::result::ZipError; -use zip::ZipArchive; - -use crate::actix_web; -use crate::actix_web::dev::Server; -use crate::actix_web::middleware; -use crate::actix_web::web; -use crate::actix_web::App; -use crate::actix_web::HttpRequest; -use crate::actix_web::HttpServer; - -#[derive(Error, Debug)] -pub enum DownloaderError { - #[error("IO error: {0}")] - Io(#[from] std::io::Error), - - #[error("Dozer is not initialized")] - NotInitialized, - #[error("Connection {0} not found")] - ConnectionNotFound(String), - #[error("Error in initializing live server: {0}")] - Transport(#[from] tonic::transport::Error), - #[error("Error in reading or extracting from Zip file: {0}")] - ZipError(#[from] ZipError), - #[error("Reqwest error: {0}")] - Reqwest(#[from] reqwest::Error), - #[error("Cannot start ui server: {0}")] - CannotStartUiServer(#[source] std::io::Error), -} -async fn fetch_dozer_ui(url: &str, folder_name: &str) -> Result<(), DownloaderError> { - // let url = "https://dozer-explorer.s3.ap-southeast-1.amazonaws.com/latest"; - let latest_url: &str = &format!("{}/latest", url); - let (key, existing_key, key_changed) = get_key_from_url(latest_url, folder_name).await?; - let zip_file_name = key.as_str(); - let prev_zip_file_name = existing_key.as_str(); - if key_changed { - info!("Downloading latest file: {}", zip_file_name); - let base_url = &format!("{}/", url); - let zip_url = &(base_url.to_owned() + zip_file_name); - if !prev_zip_file_name.is_empty() { - delete_file_if_present(prev_zip_file_name, folder_name)?; - } - get_zip_from_url(zip_url, folder_name, zip_file_name).await?; - } else { - info!("Current file is up to date"); - } - Ok(()) -} -pub const LOCAL_APP_UI_DIR: &str = "local-app-ui"; -pub async fn fetch_latest_dozer_app_ui_code() -> Result<(), DownloaderError> { - fetch_dozer_ui( - "https://dozer-app-ui-local.s3.ap-southeast-1.amazonaws.com", - LOCAL_APP_UI_DIR, - ) - .await -} - -pub fn validate_if_dozer_app_ui_code_exists() -> bool { - let directory_path = get_directory_path(); - let file_path = Path::new(&directory_path) - .join(LOCAL_APP_UI_DIR) - .join("contents"); - file_path.exists() -} - -pub const LIVE_APP_UI_DIR: &str = "live-app-ui"; -pub async fn fetch_latest_dozer_explorer_code() -> Result<(), DownloaderError> { - fetch_dozer_ui( - "https://dozer-explorer.s3.ap-southeast-1.amazonaws.com", - LIVE_APP_UI_DIR, - ) - .await -} - -// This function gets the latest keys from url and compares it with the existing key -// Returns the latest key, existing key and a boolean indicating if the key has changed -async fn get_key_from_url( - url: &str, - folder_name: &str, -) -> Result<(String, String, bool), DownloaderError> { - let response = reqwest::get(url).await?.error_for_status()?.text().await?; - let key = response.trim().to_string(); - let directory_path = format!("{}/{}/", get_directory_path(), folder_name); - - let file_path = format!("{}keys.txt", directory_path); //"/local-ui/keys.txt"; - let mut existing_key = String::new(); - if let Ok(mut file) = std::fs::File::open(&file_path) { - file.read_to_string(&mut existing_key)?; - } - let existing_key = existing_key.trim().to_string(); - - let key_changed = existing_key != key; - - if key_changed { - std::fs::create_dir_all(directory_path)?; - let mut file = std::fs::File::create(&file_path)?; - file.write_all(key.as_bytes())?; - } - - Ok((key, existing_key, key_changed)) -} - -// This function gets the latest zip from url and extracts the zip file to the local-ui directory -async fn get_zip_from_url( - url: &str, - folder_name: &str, - file_name: &str, -) -> Result<(), DownloaderError> { - // Download the ZIP file - let response = reqwest::get(url).await?.error_for_status()?.bytes().await?; - let mut temp_zip = tempfile::tempfile()?; - - // Save the downloaded ZIP content to a temporary file - temp_zip.write_all(&response)?; - - // Prepare paths and files for extraction - let directory_path = get_directory_path(); - let file_path = Path::new(&directory_path).join(folder_name).join(file_name); - let mut existing_zip = Vec::new(); - - // Read existing ZIP content if it exists - if let Ok(mut file) = File::open(&file_path) { - file.read_to_end(&mut existing_zip)?; - } - - // Create necessary directories - std::fs::create_dir_all(&directory_path)?; - std::fs::create_dir_all(folder_name)?; - std::fs::create_dir_all(file_path.parent().unwrap())?; - - // Save the downloaded ZIP content to the final location - let mut file = File::create(&file_path)?; - temp_zip.seek(std::io::SeekFrom::Start(0))?; - std::io::copy(&mut temp_zip, &mut file)?; - - // Extract the ZIP archive - let archive_file = File::open(&file_path)?; - let mut archive = ZipArchive::new(archive_file)?; - - let extraction_path = Path::new(&directory_path) - .join(folder_name) - .join("contents"); - for i in 0..archive.len() { - let mut file = archive.by_index(i)?; - let outpath = extraction_path.join(file.mangled_name()); - - if (file.name()).ends_with('/') { - std::fs::create_dir_all(&outpath)?; - } else if let Some(p) = outpath.parent() { - if !p.exists() { - std::fs::create_dir_all(p)?; - } - let mut outfile = File::create(&outpath)?; - std::io::copy(&mut file, &mut outfile)?; - } - } - - Ok(()) -} - -// This function deletes the zip files and the contents directory if key has changed -fn delete_file_if_present(file_name: &str, folder_name: &str) -> Result<(), std::io::Error> { - let directory_path = get_directory_path(); - let file_path = Path::new(&directory_path).join(folder_name).join(file_name); - info!("deleting file {:?}", file_path); - if file_path.exists() { - remove_file(file_path)?; - } - let contents_path = Path::new(&directory_path) - .join(folder_name) - .join("contents"); - if contents_path.exists() { - remove_dir_all(contents_path)?; - } - Ok(()) -} - -async fn index(_req: HttpRequest, data: web::Data) -> actix_web::Result { - let folder_name = &data.folder_name; - let build_path = get_build_path(folder_name); - let index_path = build_path.join("index.html"); - let path: PathBuf = index_path; // Update this with the actual path to your build folder - Ok(NamedFile::open(path)?) -} - -struct AppData { - folder_name: String, -} -//This function navigates to the react app and starts it -pub fn start_react_app(port: u16, folder_name: &str) -> Result { - let build_path = get_build_path(folder_name); - let static_path = build_path.join("static"); - let assets_path = build_path.join("assets"); - let my_folder_name = folder_name.to_owned(); - let server = HttpServer::new(move || { - App::new() - .wrap(middleware::Logger::default()) - .service(actix_files::Files::new("/static", &static_path).show_files_listing()) - .service(actix_files::Files::new("/assets", &assets_path).show_files_listing()) - .app_data(web::Data::new(AppData { - folder_name: my_folder_name.to_owned(), - })) - .route("/{anyname:.*}", web::get().to(index)) - }) - .bind(("0.0.0.0", port))? - .run(); - - Ok(server) -} - -fn get_directory_path() -> String { - let home_dir = env::var("HOME").unwrap_or_else(|_| ".".to_string()); - format!("{}/{}", home_dir, ".dozer") -} - -fn get_build_path(folder_name: &str) -> PathBuf { - let directory_path = get_directory_path(); - Path::new(&directory_path) - .join(folder_name) - .join("contents") - .join("build") -} diff --git a/dozer-cli/src/ui/mod.rs b/dozer-cli/src/ui/mod.rs deleted file mode 100644 index 90d4ae830f..0000000000 --- a/dozer-cli/src/ui/mod.rs +++ /dev/null @@ -1,2 +0,0 @@ -pub mod app; -pub mod downloader; diff --git a/dozer-cli/src/utils.rs b/dozer-cli/src/utils.rs deleted file mode 100644 index 2246972782..0000000000 --- a/dozer-cli/src/utils.rs +++ /dev/null @@ -1,34 +0,0 @@ -use dozer_core::executor::ExecutorOptions; -use dozer_types::models::{ - app_config::{default_app_buffer_size, default_error_threshold, default_event_hub_capacity}, - config::Config, -}; - -fn get_buffer_size(config: &Config) -> u32 { - config - .app - .app_buffer_size - .unwrap_or_else(default_app_buffer_size) -} - -fn get_error_threshold(config: &Config) -> u32 { - config - .app - .error_threshold - .unwrap_or_else(default_error_threshold) -} - -fn get_event_hub_capacity(config: &Config) -> usize { - config - .app - .event_hub_capacity - .unwrap_or_else(default_event_hub_capacity) -} - -pub fn get_executor_options(config: &Config) -> ExecutorOptions { - ExecutorOptions { - channel_buffer_sz: get_buffer_size(config) as usize, - error_threshold: Some(get_error_threshold(config)), - event_hub_capacity: get_event_hub_capacity(config), - } -} diff --git a/dozer-core/Cargo.toml b/dozer-core/Cargo.toml deleted file mode 100644 index afae7e6f70..0000000000 --- a/dozer-core/Cargo.toml +++ /dev/null @@ -1,27 +0,0 @@ -[package] -name = "dozer-core" -version = "0.4.0" -edition = "2021" -authors = ["getdozer/dozer-dev"] -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-types = { path = "../dozer-types/" } -dozer-tracing = { path = "../dozer-tracing/" } -bincode = { workspace = true } - -uuid = { version = "1.6.1", features = ["v1", "v4", "fast-rng"] } -crossbeam = "0.8.2" -daggy = { git = "https://github.com/getdozer/daggy", branch = "feat/try_map", features = [ - "serde-1", -] } -futures-util = "0.3.28" -async-stream = "0.3.5" -futures = "0.3.30" -tokio = { version = "1", features = ["full"] } -deno_core = { workspace = true, optional = true} - -[features] -javascript = ["dep:deno_core"] diff --git a/dozer-core/src/app.rs b/dozer-core/src/app.rs deleted file mode 100644 index 29597fca7a..0000000000 --- a/dozer-core/src/app.rs +++ /dev/null @@ -1,186 +0,0 @@ -use dozer_types::models::flags::{EnableProbabilisticOptimizations, Flags}; -use dozer_types::node::NodeHandle; - -use crate::appsource::{self, AppSourceManager}; -use crate::errors::ExecutionError; -use crate::node::{PortHandle, ProcessorFactory, SinkFactory}; -use crate::{Dag, Edge, Endpoint}; - -#[derive(Debug, Clone, PartialEq, Eq, Hash)] -pub struct PipelineEntryPoint { - source_name: String, - /// Target port. - port: PortHandle, -} - -impl PipelineEntryPoint { - pub fn new(source_name: String, port: PortHandle) -> Self { - Self { source_name, port } - } - - pub fn source_name(&self) -> &str { - &self.source_name - } -} - -#[derive(Debug)] -pub struct AppPipeline { - edges: Vec, - processors: Vec<(NodeHandle, Box)>, - sinks: Vec<(NodeHandle, Box)>, - entry_points: Vec<(NodeHandle, PipelineEntryPoint)>, - flags: PipelineFlags, -} - -impl AppPipeline { - fn create_handle(id: String) -> NodeHandle { - NodeHandle::new(None, id) - } - - pub fn add_processor(&mut self, proc: Box, id: String) { - self.processors.push((Self::create_handle(id), proc)); - } - - pub fn add_sink(&mut self, sink: Box, id: String) { - self.sinks.push((Self::create_handle(id), sink)); - } - - pub fn add_entry_point(&mut self, id: String, entry_point: PipelineEntryPoint) { - self.entry_points - .push((Self::create_handle(id), entry_point)); - } - - pub fn connect_nodes( - &mut self, - from: String, - from_port: PortHandle, - to: String, - to_port: PortHandle, - ) { - let edge = Edge::new( - Endpoint::new(NodeHandle::new(None, from), from_port), - Endpoint::new(NodeHandle::new(None, to), to_port), - ); - self.edges.push(edge); - } - - pub fn new(flags: PipelineFlags) -> Self { - Self { - processors: Vec::new(), - sinks: Vec::new(), - edges: Vec::new(), - entry_points: Vec::new(), - flags, - } - } - - pub fn new_with_default_flags() -> Self { - Self::new(Default::default()) - } - - pub fn get_entry_points_sources_names(&self) -> Vec { - self.entry_points - .iter() - .map(|(_, p)| p.source_name().to_string()) - .collect() - } - - pub fn flags(&self) -> &PipelineFlags { - &self.flags - } -} - -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct PipelineFlags { - pub enable_probabilistic_optimizations: EnableProbabilisticOptimizations, -} - -impl From<&Flags> for PipelineFlags { - fn from(flags: &Flags) -> Self { - Self { - enable_probabilistic_optimizations: flags.enable_probabilistic_optimizations.clone(), - } - } -} - -impl From for PipelineFlags { - fn from(flags: Flags) -> Self { - Self::from(&flags) - } -} - -impl Default for PipelineFlags { - fn default() -> Self { - Flags::default().into() - } -} - -pub struct App { - pipelines: Vec<(u16, AppPipeline)>, - app_counter: u16, - sources: AppSourceManager, -} - -impl App { - pub fn add_pipeline(&mut self, pipeline: AppPipeline) { - self.app_counter += 1; - self.pipelines.push((self.app_counter, pipeline)); - } - - pub fn into_dag(self) -> Result { - let mut dag = Dag::new(); - // (source name, target endpoint) - let mut entry_points: Vec<(String, Endpoint)> = Vec::new(); - - // Add all processors and sinks while collecting the entry points. - for (pipeline_id, pipeline) in self.pipelines { - for (handle, proc) in pipeline.processors { - dag.add_processor(NodeHandle::new(Some(pipeline_id), handle.id), proc); - } - for (handle, sink) in pipeline.sinks { - dag.add_sink(NodeHandle::new(Some(pipeline_id), handle.id), sink); - } - for edge in pipeline.edges { - dag.connect( - Endpoint::new( - NodeHandle::new(Some(pipeline_id), edge.from.node.id), - edge.from.port, - ), - Endpoint::new( - NodeHandle::new(Some(pipeline_id), edge.to.node.id), - edge.to.port, - ), - )?; - } - - for (handle, entry) in pipeline.entry_points { - entry_points.push(( - entry.source_name, - Endpoint::new(NodeHandle::new(Some(pipeline_id), handle.id), entry.port), - )); - } - } - - for (source, mapping) in self.sources.sources.into_iter().zip(&self.sources.mappings) { - let node_handle = NodeHandle::new(None, mapping.connection.clone()); - dag.add_source(node_handle, source); - } - - // Connect to all pipelines - for (source_name, target_endpoint) in entry_points { - let source_endpoint = - appsource::get_endpoint_from_mappings(&self.sources.mappings, &source_name)?; - dag.connect(source_endpoint, target_endpoint)?; - } - - Ok(dag) - } - - pub fn new(sources: AppSourceManager) -> Self { - Self { - pipelines: Vec::new(), - app_counter: 0, - sources, - } - } -} diff --git a/dozer-core/src/appsource.rs b/dozer-core/src/appsource.rs deleted file mode 100644 index c5e7eb55a4..0000000000 --- a/dozer-core/src/appsource.rs +++ /dev/null @@ -1,82 +0,0 @@ -use dozer_types::node::NodeHandle; - -use crate::errors::ExecutionError; -use crate::errors::ExecutionError::{ - AmbiguousSourceIdentifier, AppSourceConnectionAlreadyExists, InvalidSourceIdentifier, -}; -use crate::node::{PortHandle, SourceFactory}; -use crate::Endpoint; -use std::collections::HashMap; - -#[derive(Debug)] -pub struct AppSourceMappings { - pub connection: String, - /// From source name to output port handle. - pub mappings: HashMap, -} - -impl AppSourceMappings { - pub fn new(connection: String, mappings: HashMap) -> Self { - Self { - connection, - mappings, - } - } -} - -#[derive(Debug, Default)] -pub struct AppSourceManager { - pub(crate) sources: Vec>, - pub(crate) mappings: Vec, -} - -impl AppSourceManager { - pub fn add( - &mut self, - source: Box, - mapping: AppSourceMappings, - ) -> Result<(), ExecutionError> { - if self - .mappings - .iter() - .any(|existing_mapping| existing_mapping.connection == mapping.connection) - { - return Err(AppSourceConnectionAlreadyExists(mapping.connection)); - } - - self.sources.push(source); - self.mappings.push(mapping); - Ok(()) - } - - pub fn get_endpoint(&self, source_name: &str) -> Result { - get_endpoint_from_mappings(&self.mappings, source_name) - } - - pub fn new() -> Self { - Self::default() - } -} - -pub fn get_endpoint_from_mappings( - mappings: &[AppSourceMappings], - source_name: &str, -) -> Result { - let mut found: Vec = mappings - .iter() - .filter_map(|mapping| { - mapping.mappings.get(source_name).map(|output_port| { - Endpoint::new( - NodeHandle::new(None, mapping.connection.clone()), - *output_port, - ) - }) - }) - .collect(); - - match found.len() { - 0 => Err(InvalidSourceIdentifier(source_name.to_string())), - 1 => Ok(found.remove(0)), - _ => Err(AmbiguousSourceIdentifier(source_name.to_string())), - } -} diff --git a/dozer-core/src/builder_dag.rs b/dozer-core/src/builder_dag.rs deleted file mode 100644 index e2041af90b..0000000000 --- a/dozer-core/src/builder_dag.rs +++ /dev/null @@ -1,241 +0,0 @@ -use std::{ - collections::{hash_map::Entry, HashMap}, - fmt::Debug, -}; - -use daggy::{petgraph::visit::IntoNodeIdentifiers, NodeIndex}; -use dozer_types::{ - log::warn, - node::{NodeHandle, OpIdentifier}, -}; - -use crate::{ - dag_schemas::{DagHaveSchemas, DagSchemas, EdgeType}, - errors::ExecutionError, - event::EventHub, - node::{Processor, Sink, SinkFactory, Source}, - NodeKind as DagNodeKind, -}; - -#[derive(Debug)] -/// Node in the builder DAG. -pub struct NodeType { - /// The node handle. - pub handle: NodeHandle, - /// The node kind. - pub kind: NodeKind, -} - -#[derive(Debug)] -/// Node kind, source, processor or sink. Source has a checkpoint to start from. -pub enum NodeKind { - Source { - source: Box, - last_checkpoint: Option, - }, - Processor(Box), - Sink(Box), -} - -/// Builder DAG builds all the sources, processors and sinks. -/// It also asks each source if its possible to start from the given checkpoint. -/// If not possible, it resets metadata and updates the checkpoint. -#[derive(Debug)] -pub struct BuilderDag { - graph: daggy::Dag, - event_hub: EventHub, -} - -impl BuilderDag { - pub async fn new( - dag_schemas: DagSchemas, - event_hub_capacity: usize, - ) -> Result { - // Collect input output schemas. - let mut input_schemas = HashMap::new(); - let mut output_schemas = HashMap::new(); - for node_index in dag_schemas.graph().node_identifiers() { - input_schemas.insert(node_index, dag_schemas.get_node_input_schemas(node_index)); - output_schemas.insert(node_index, dag_schemas.get_node_output_schemas(node_index)); - } - - // Collect sources that may affect a node. - let mut affecting_sources = dag_schemas - .graph() - .node_identifiers() - .map(|node_index| dag_schemas.collect_ancestor_sources(node_index)) - .collect::>(); - - // Prepare nodes and edges for consuming. - let (nodes, edges) = dag_schemas.into_graph().into_graph().into_nodes_edges(); - let mut nodes = nodes - .into_iter() - .map(|node| Some(node.weight)) - .collect::>(); - - // Build the sinks and load checkpoint. - let event_hub = EventHub::new(event_hub_capacity); - let mut graph = daggy::Dag::new(); - let mut source_states = HashMap::new(); - let mut source_op_ids = HashMap::new(); - let mut source_id_to_sinks = HashMap::>::new(); - let mut node_index_map: HashMap = HashMap::new(); - for (node_index, node) in nodes.iter_mut().enumerate() { - if let Some((handle, sink)) = take_sink(node) { - let sources = std::mem::take(&mut affecting_sources[node_index]); - if sources.len() > 1 { - warn!("Multiple sources ({sources:?}) connected to same sink: {handle}"); - } - let source = sources.into_iter().next().expect("sink must have a source"); - - let node_index = NodeIndex::new(node_index); - let mut sink = sink - .build( - input_schemas - .remove(&node_index) - .expect("we collected all input schemas"), - event_hub.clone(), - ) - .await - .map_err(ExecutionError::Factory)?; - - let state = sink.get_source_state().map_err(ExecutionError::Sink)?; - if let Some(state) = state { - match source_states.entry(source.clone()) { - Entry::Occupied(entry) => { - if entry.get() != &state { - return Err(ExecutionError::SourceStateConflict(source)); - } - } - Entry::Vacant(entry) => { - entry.insert(state); - } - } - } - - let op_id = sink.get_latest_op_id().map_err(ExecutionError::Sink)?; - if let Some(op_id) = op_id { - match source_op_ids.entry(source.clone()) { - Entry::Occupied(mut entry) => { - *entry.get_mut() = op_id.min(*entry.get()); - } - Entry::Vacant(entry) => { - entry.insert(op_id); - } - } - } - - let new_node_index = graph.add_node(NodeType { - handle, - kind: NodeKind::Sink(sink), - }); - node_index_map.insert(node_index, new_node_index); - source_id_to_sinks - .entry(source) - .or_default() - .push(new_node_index); - } - } - - // Build sources, processors, and collect source states. - for (node_index, node) in nodes.iter_mut().enumerate() { - let Some(node) = node.take() else { - continue; - }; - let node_index = NodeIndex::new(node_index); - let node = match node.kind { - DagNodeKind::Source(source) => { - let source = source - .build( - output_schemas - .remove(&node_index) - .expect("we collected all output schemas"), - event_hub.clone(), - source_states.remove(&node.handle), - ) - .map_err(ExecutionError::Factory)?; - - // Write state to relevant sink. - let state = source - .serialize_state() - .await - .map_err(ExecutionError::Source)?; - let mut checkpoint = None; - for sink in source_id_to_sinks.remove(&node.handle).unwrap_or_default() { - let sink = &mut graph[sink]; - let sink_handle = &sink.handle; - let NodeKind::Sink(sink) = &mut sink.kind else { - unreachable!() - }; - sink.set_source_state(&state) - .map_err(ExecutionError::Sink)?; - if let Some(sink_checkpoint) = source_op_ids.remove(sink_handle) { - checkpoint = - Some(checkpoint.unwrap_or(sink_checkpoint).min(sink_checkpoint)); - } - } - - NodeType { - handle: node.handle, - kind: NodeKind::Source { - source, - last_checkpoint: checkpoint, - }, - } - } - DagNodeKind::Processor(processor) => { - let processor = processor - .build( - input_schemas - .remove(&node_index) - .expect("we collected all input schemas"), - output_schemas - .remove(&node_index) - .expect("we collected all output schemas"), - event_hub.clone(), - ) - .await - .map_err(ExecutionError::Factory)?; - NodeType { - handle: node.handle, - kind: NodeKind::Processor(processor), - } - } - DagNodeKind::Sink(_) => unreachable!(), - }; - let new_node_index = graph.add_node(node); - node_index_map.insert(node_index, new_node_index); - } - - // Connect the edges. - for edge in edges { - graph - .add_edge( - node_index_map[&edge.source()], - node_index_map[&edge.target()], - edge.weight, - ) - .expect("we know there's no loop"); - } - - Ok(BuilderDag { graph, event_hub }) - } - - pub fn graph(&self) -> &daggy::Dag { - &self.graph - } - - pub fn into_graph_and_event_hub(self) -> (daggy::Dag, EventHub) { - (self.graph, self.event_hub) - } -} - -fn take_sink(node: &mut Option) -> Option<(NodeHandle, Box)> { - let super::NodeType { handle, kind } = node.take()?; - if let super::NodeKind::Sink(sink) = kind { - Some((handle, sink)) - } else { - *node = Some(super::NodeType { handle, kind }); - None - } -} diff --git a/dozer-core/src/channels.rs b/dozer-core/src/channels.rs deleted file mode 100644 index 36951d8d40..0000000000 --- a/dozer-core/src/channels.rs +++ /dev/null @@ -1,9 +0,0 @@ -use dozer_types::types::TableOperation; - -pub trait ProcessorChannelForwarder { - /// Sends a operation to downstream nodes. Panics if the operation cannot be sent. - /// - /// We must panic instead of returning an error because this method will be called by `Processor::process`, - /// which only returns recoverable errors. - fn send(&mut self, op: TableOperation); -} diff --git a/dozer-core/src/dag_impl.rs b/dozer-core/src/dag_impl.rs deleted file mode 100644 index 4d284150fb..0000000000 --- a/dozer-core/src/dag_impl.rs +++ /dev/null @@ -1,399 +0,0 @@ -use daggy::petgraph::visit::{Bfs, EdgeRef, IntoEdges}; -use daggy::Walker; -use dozer_types::node::NodeHandle; - -use crate::errors::ExecutionError; -use crate::node::{PortHandle, ProcessorFactory, SinkFactory, SourceFactory}; -use std::collections::{HashMap, HashSet}; -use std::fmt::{Debug, Display}; - -pub const DEFAULT_PORT_HANDLE: u16 = 0xffff_u16; - -#[derive(Clone, Debug, PartialEq, Eq, Hash)] -pub struct Endpoint { - pub node: NodeHandle, - pub port: PortHandle, -} - -impl Endpoint { - pub fn new(node: NodeHandle, port: PortHandle) -> Self { - Self { node, port } - } -} - -#[derive(Clone, Debug, PartialEq, Eq, Hash)] -pub struct Edge { - pub from: Endpoint, - pub to: Endpoint, -} - -impl Edge { - pub fn new(from: Endpoint, to: Endpoint) -> Self { - Self { from, to } - } -} - -#[derive(Debug)] -/// A `SourceFactory`, `ProcessorFactory` or `SinkFactory`. -pub enum NodeKind { - Source(Box), - Processor(Box), - Sink(Box), -} - -#[derive(Debug)] -/// The node type of the description DAG. -pub struct NodeType { - /// The node handle, unique across the DAG. - pub handle: NodeHandle, - /// The node kind. - pub kind: NodeKind, -} - -impl Display for NodeType { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - write!(f, "{}", self.handle.id) - } -} - -#[derive(Debug, Clone, Copy)] -/// The edge type of the description DAG. -pub struct EdgeType { - pub from: PortHandle, - pub to: PortHandle, -} - -impl EdgeType { - pub fn new(from: PortHandle, to: PortHandle) -> Self { - Self { from, to } - } -} - -pub trait EdgeHavePorts { - fn output_port(&self) -> PortHandle; - fn input_port(&self) -> PortHandle; -} - -impl EdgeHavePorts for EdgeType { - fn output_port(&self) -> PortHandle { - self.from - } - - fn input_port(&self) -> PortHandle { - self.to - } -} - -#[derive(Debug)] -pub struct Dag { - /// The underlying graph. - graph: daggy::Dag, - /// Map from node handle to node index. - node_lookup_table: HashMap, - /// All edge indexes. - edge_indexes: HashSet, -} - -impl Default for Dag { - fn default() -> Self { - Self::new() - } -} - -impl Dag { - /// Creates an empty DAG. - pub fn new() -> Self { - Self { - graph: daggy::Dag::new(), - node_lookup_table: HashMap::new(), - edge_indexes: HashSet::new(), - } - } - - /// Returns the underlying daggy graph. - pub fn graph(&self) -> &daggy::Dag { - &self.graph - } - - /// Returns the underlying daggy graph. - pub fn into_graph(self) -> daggy::Dag { - self.graph - } - - /// Adds a source. Panics if the `handle` exists in the `Dag`. - pub fn add_source( - &mut self, - handle: NodeHandle, - source: Box, - ) -> daggy::NodeIndex { - self.add_node(handle, NodeKind::Source(source)) - } - - /// Adds a processor. Panics if the `handle` exists in the `Dag`. - pub fn add_processor( - &mut self, - handle: NodeHandle, - processor: Box, - ) -> daggy::NodeIndex { - self.add_node(handle, NodeKind::Processor(processor)) - } - - /// Adds a sink. Panics if the `handle` exists in the `Dag`. - pub fn add_sink(&mut self, handle: NodeHandle, sink: Box) -> daggy::NodeIndex { - self.add_node(handle, NodeKind::Sink(sink)) - } - - /// Adds an edge. Panics if there's already an edge from `from` to `to`. - /// - /// Returns an error if any of the port cannot be found or the edge would create a cycle. - pub fn connect(&mut self, from: Endpoint, to: Endpoint) -> Result<(), ExecutionError> { - let from_node_index = validate_endpoint(self, &from, PortDirection::Output)?; - let to_node_index = validate_endpoint(self, &to, PortDirection::Input)?; - self.connect_with_index(from_node_index, from.port, to_node_index, to.port) - } - - /// Adds an edge. Panics if there's already an edge from `from` to `to`. - /// - /// Returns an error if any of the port cannot be found or the edge would create a cycle. - pub fn connect_with_index( - &mut self, - from_node_index: daggy::NodeIndex, - output_port: PortHandle, - to_node_index: daggy::NodeIndex, - input_port: PortHandle, - ) -> Result<(), ExecutionError> { - validate_port_with_index(self, from_node_index, output_port, PortDirection::Output)?; - validate_port_with_index(self, to_node_index, input_port, PortDirection::Input)?; - let edge_index = self.graph.add_edge( - from_node_index, - to_node_index, - EdgeType::new(output_port, input_port), - )?; - - if !self.edge_indexes.insert(EdgeIndex { - from_node: from_node_index, - output_port, - to_node: to_node_index, - input_port, - }) { - panic!("An edge {edge_index:?} has already been inserted using specified edge handle"); - } - - Ok(()) - } - - /// Adds another whole `Dag` to `self`. Optionally under a namespace `ns`. - pub fn merge(&mut self, ns: Option, other: Dag) { - let (other_nodes, _) = other.graph.into_graph().into_nodes_edges(); - - // Insert nodes. - let mut other_node_index_to_self_node_index = vec![]; - for other_node in other_nodes.into_iter() { - let other_node = other_node.weight; - let self_node_handle = - NodeHandle::new(ns.or(other_node.handle.ns), other_node.handle.id.clone()); - let self_node_index = self.add_node(self_node_handle.clone(), other_node.kind); - other_node_index_to_self_node_index.push(self_node_index); - } - - // Insert edges. - for other_edge_index in other.edge_indexes.into_iter() { - let self_from_node = - other_node_index_to_self_node_index[other_edge_index.from_node.index()]; - let self_to_node = - other_node_index_to_self_node_index[other_edge_index.to_node.index()]; - self.connect_with_index( - self_from_node, - other_edge_index.output_port, - self_to_node, - other_edge_index.input_port, - ) - .expect("BUG in DAG"); - } - } - - /// Returns an iterator over all node handles. - pub fn node_handles(&self) -> impl Iterator { - self.nodes().map(|node| &node.handle) - } - - /// Returns an iterator over all nodes. - pub fn nodes(&self) -> impl Iterator { - self.graph.raw_nodes().iter().map(|node| &node.weight) - } - - /// Returns an iterator over source handles and sources. - pub fn sources(&self) -> impl Iterator { - self.nodes().flat_map(|node| { - if let NodeKind::Source(source) = &node.kind { - Some((&node.handle, &**source)) - } else { - None - } - }) - } - - /// Returns an iterator over processor handles and processors. - pub fn processors(&self) -> impl Iterator { - self.nodes().flat_map(|node| { - if let NodeKind::Processor(processor) = &node.kind { - Some((&node.handle, &**processor)) - } else { - None - } - }) - } - - /// Returns an iterator over sink handles and sinks. - pub fn sinks(&self) -> impl Iterator { - self.nodes().flat_map(|node| { - if let NodeKind::Sink(sink) = &node.kind { - Some((&node.handle, &**sink)) - } else { - None - } - }) - } - - /// Returns an iterator over all edge handles. - pub fn edge_handles(&self) -> Vec { - let get_endpoint = |node_index: daggy::NodeIndex, port_handle| { - let node = &self.graph[node_index]; - Endpoint { - node: node.handle.clone(), - port: port_handle, - } - }; - - self.edge_indexes - .iter() - .map(|edge_index| { - Edge::new( - get_endpoint(edge_index.from_node, edge_index.output_port), - get_endpoint(edge_index.to_node, edge_index.input_port), - ) - }) - .collect() - } - - /// Finds the node by its handle. - pub fn node_kind_from_handle(&self, handle: &NodeHandle) -> &NodeKind { - &self.graph[self.node_index(handle)].kind - } - - /// Returns an iterator over node handles that are connected to the given node handle. - pub fn edges_from_handle(&self, handle: &NodeHandle) -> impl Iterator { - let node_index = self.node_index(handle); - self.graph - .edges(node_index) - .map(|edge| &self.graph[edge.target()].handle) - } - - /// Returns an iterator over endpoints that are connected to the given endpoint. - pub fn edges_from_endpoint<'a>( - &'a self, - node_handle: &'a NodeHandle, - port_handle: PortHandle, - ) -> impl Iterator { - self.graph - .edges(self.node_index(node_handle)) - .filter(move |edge| edge.weight().from == port_handle) - .map(|edge| (&self.graph[edge.target()].handle, edge.weight().to)) - } - - /// Returns an iterator over all node handles reachable from `start` in a breadth-first search. - pub fn bfs(&self, start: &NodeHandle) -> impl Iterator { - let start = self.node_index(start); - - Bfs::new(self.graph.graph(), start) - .iter(self.graph.graph()) - .map(|node_index| &self.graph[node_index].handle) - } -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] -struct EdgeIndex { - from_node: daggy::NodeIndex, - output_port: PortHandle, - to_node: daggy::NodeIndex, - input_port: PortHandle, -} - -impl Dag { - fn add_node(&mut self, handle: NodeHandle, kind: NodeKind) -> daggy::NodeIndex { - let node_index = self.graph.add_node(NodeType { - handle: handle.clone(), - kind, - }); - if let Some(node_index) = self.node_lookup_table.insert(handle, node_index) { - panic!("A node {node_index:?} has already been inserted using specified node handle"); - } - node_index - } - - fn node_index(&self, node_handle: &NodeHandle) -> daggy::NodeIndex { - *self - .node_lookup_table - .get(node_handle) - .unwrap_or_else(|| panic!("Node handle {node_handle:?} not found in dag")) - } -} - -#[derive(Clone, Debug, PartialEq, Eq, Hash)] -enum PortDirection { - Input, - Output, -} - -fn validate_endpoint( - dag: &Dag, - endpoint: &Endpoint, - direction: PortDirection, -) -> Result { - let node_index = dag.node_index(&endpoint.node); - validate_port_with_index(dag, node_index, endpoint.port, direction)?; - Ok(node_index) -} - -fn validate_port_with_index( - dag: &Dag, - node_index: daggy::NodeIndex, - port: PortHandle, - direction: PortDirection, -) -> Result<(), ExecutionError> { - let node = &dag.graph[node_index]; - if !contains_port(&node.kind, direction, port)? { - return Err(ExecutionError::InvalidPortHandle(port)); - } - Ok(()) -} - -fn contains_port( - node: &NodeKind, - direction: PortDirection, - port: PortHandle, -) -> Result { - Ok(match node { - NodeKind::Processor(p) => { - if direction == PortDirection::Output { - p.get_output_ports().iter().any(|e| e == &port) - } else { - p.get_input_ports().contains(&port) - } - } - NodeKind::Sink(s) => { - if direction == PortDirection::Output { - false - } else { - s.get_input_ports().contains(&port) - } - } - NodeKind::Source(s) => { - if direction == PortDirection::Output { - s.get_output_ports().iter().any(|e| e.handle == port) - } else { - false - } - } - }) -} diff --git a/dozer-core/src/dag_schemas.rs b/dozer-core/src/dag_schemas.rs deleted file mode 100644 index cd90f1cf8b..0000000000 --- a/dozer-core/src/dag_schemas.rs +++ /dev/null @@ -1,527 +0,0 @@ -use crate::errors::ExecutionError; -use crate::{Dag, EdgeHavePorts, NodeKind}; - -use crate::node::{OutputPortType, PortHandle}; -use daggy::petgraph::graph::EdgeReference; -use daggy::petgraph::visit::{EdgeRef, IntoEdges, IntoEdgesDirected, IntoNodeReferences, Topo}; -use daggy::petgraph::Direction; -use daggy::{NodeIndex, Walker}; -use dozer_types::log::{error, info}; -use dozer_types::node::NodeHandle; -use dozer_types::serde::{Deserialize, Serialize}; -use dozer_types::types::Schema; -use std::collections::{HashMap, HashSet}; -use std::fmt::Debug; - -use super::node::OutputPortDef; -use super::{EdgeType as DagEdgeType, NodeType}; - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(crate = "dozer_types::serde")] -pub struct EdgeType { - pub output_port: PortHandle, - pub input_port: PortHandle, - pub schema: Schema, - pub edge_kind: EdgeKind, -} - -impl EdgeType { - pub fn new( - output_port: PortHandle, - input_port: PortHandle, - schema: Schema, - edge_kind: EdgeKind, - ) -> Self { - Self { - output_port, - input_port, - schema, - edge_kind, - } - } -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(crate = "dozer_types::serde")] -pub enum EdgeKind { - FromSource { - port_type: OutputPortType, - port_name: String, - }, - FromProcessor, -} - -pub trait EdgeHaveSchema: EdgeHavePorts { - fn schema(&self) -> &Schema; -} - -impl EdgeHavePorts for EdgeType { - fn output_port(&self) -> PortHandle { - self.output_port - } - - fn input_port(&self) -> PortHandle { - self.input_port - } -} - -impl EdgeHaveSchema for EdgeType { - fn schema(&self) -> &Schema { - &self.schema - } -} - -#[derive(Debug)] -/// `DagSchemas` is a `Dag` with validated schema on the edge. -pub struct DagSchemas { - graph: daggy::Dag, -} - -impl DagSchemas { - pub fn into_graph(self) -> daggy::Dag { - self.graph - } - - pub fn graph(&self) -> &daggy::Dag { - &self.graph - } -} - -impl DagSchemas { - /// Validate and populate the schemas, the resultant DAG will have the exact same structure as the input DAG, - /// with validated schema information on the edges. - pub async fn new(dag: Dag) -> Result { - validate_connectivity(&dag); - - match populate_schemas(dag.into_graph()).await { - Ok(graph) => { - info!("[pipeline] Validation completed"); - Ok(Self { graph }) - } - Err(e) => { - error!("[pipeline] Validation error: {}", e); - Err(e) - } - } - } - - pub fn collect_ancestor_sources(&self, node_index: NodeIndex) -> HashSet { - let mut sources = HashSet::new(); - collect_ancestor_sources_recursive(self, node_index, &mut sources); - sources - } -} - -fn collect_ancestor_sources_recursive( - dag: &DagSchemas, - node_index: NodeIndex, - sources: &mut HashSet, -) { - for edge in dag.graph().edges_directed(node_index, Direction::Incoming) { - let source_node_index = edge.source(); - let source_node = &dag.graph()[source_node_index]; - if matches!(source_node.kind, NodeKind::Source(_)) { - sources.insert(source_node.handle.clone()); - } - collect_ancestor_sources_recursive(dag, source_node_index, sources); - } -} - -pub trait DagHaveSchemas { - type NodeType; - type EdgeType: EdgeHaveSchema; - - fn graph(&self) -> &daggy::Dag; - - fn get_node_input_schemas(&self, node_index: NodeIndex) -> HashMap { - let mut schemas = HashMap::new(); - - for edge in self.graph().edges_directed(node_index, Direction::Incoming) { - let edge = edge.weight(); - let schema = edge.schema(); - schemas.insert(edge.input_port(), schema.clone()); - } - - schemas - } - - fn get_node_output_schemas(&self, node_index: NodeIndex) -> HashMap { - let mut schemas = HashMap::new(); - - for edge in self.graph().edges(node_index) { - let edge = edge.weight(); - let schema = edge.schema(); - schemas.insert(edge.output_port(), schema.clone()); - } - - schemas - } -} - -impl DagHaveSchemas for DagSchemas { - type NodeType = NodeType; - type EdgeType = EdgeType; - - fn graph(&self) -> &daggy::Dag { - &self.graph - } -} - -fn validate_connectivity(dag: &Dag) { - // Every source or processor has at least one outgoing edge. - for (node_index, node) in dag.graph().node_references() { - match &node.kind { - NodeKind::Source(_) | NodeKind::Processor(_) => { - if dag.graph().edges(node_index).count() == 0 { - panic!("Node {} has no outgoing edge", node.handle); - } - } - NodeKind::Sink(_) => {} - } - } - - // Processor and sink has at least one input port. Every input port has exactly one incoming edge. - for (node_index, node) in dag.graph().node_references() { - let mut input_ports = match &node.kind { - NodeKind::Source(_) => continue, - NodeKind::Processor(processor) => processor.get_input_ports(), - NodeKind::Sink(sink) => sink.get_input_ports(), - }; - if input_ports.is_empty() { - panic!("Node {} has no input port", node.handle); - } - - input_ports.sort(); - - let mut connected_input_ports = dag - .graph() - .edges_directed(node_index, Direction::Incoming) - .map(|edge| edge.weight().to) - .collect::>(); - connected_input_ports.sort(); - - if input_ports != connected_input_ports { - panic!( - "Node {} has input ports {input_ports:?}, but the incoming edges are {connected_input_ports:?}", - node.handle - ); - } - } -} - -/// In topological order, pass output schemas to downstream nodes' input schemas. -async fn populate_schemas( - dag: daggy::Dag, -) -> Result, ExecutionError> { - let mut edges = vec![None; dag.graph().edge_count()]; - - for node_index in Topo::new(&dag).iter(&dag) { - let node = &dag.graph()[node_index]; - - match &node.kind { - NodeKind::Source(source) => { - let ports = source.get_output_ports(); - - for edge in dag.graph().edges(node_index) { - let port = edge.weight().from; - let port_type = find_output_port_type(&ports, edge); - let port_name = source.get_output_port_name(&port); - let schema = source - .get_output_schema(&port) - .map_err(ExecutionError::Factory)?; - create_edge( - &mut edges, - edge, - EdgeKind::FromSource { - port_type, - port_name, - }, - schema, - ); - } - } - - NodeKind::Processor(processor) => { - let input_schemas = - validate_input_schemas(&dag, &edges, node_index, processor.get_input_ports())?; - - for edge in dag.graph().edges(node_index) { - let schema = processor - .get_output_schema(&edge.weight().from, &input_schemas) - .await - .map_err(ExecutionError::Factory)?; - create_edge(&mut edges, edge, EdgeKind::FromProcessor, schema); - } - } - - NodeKind::Sink(sink) => { - let input_schemas = - validate_input_schemas(&dag, &edges, node_index, sink.get_input_ports())?; - sink.prepare(input_schemas) - .map_err(ExecutionError::Factory)?; - } - } - } - - Ok(dag.map_owned( - |_, node| node, - |edge, _| edges[edge.index()].take().expect("We traversed every edge"), - )) -} - -fn find_output_port_type( - ports: &[OutputPortDef], - edge: EdgeReference, -) -> OutputPortType { - let handle = edge.weight().from; - for port in ports { - if port.handle == handle { - return port.typ; - } - } - panic!("BUG: port {handle} not found") -} - -fn create_edge( - edges: &mut [Option], - edge: EdgeReference, - edge_kind: EdgeKind, - schema: Schema, -) { - let edge_ref = &mut edges[edge.id().index()]; - debug_assert!(edge_ref.is_none()); - *edge_ref = Some(EdgeType::new( - edge.weight().from, - edge.weight().to, - schema, - edge_kind, - )); -} - -fn validate_input_schemas( - dag: &daggy::Dag, - edge_and_contexts: &[Option], - node_index: NodeIndex, - input_ports: Vec, -) -> Result, ExecutionError> { - let node_handle = &dag.graph()[node_index].handle; - - let mut input_schemas = HashMap::new(); - for edge in dag.graph().edges_directed(node_index, Direction::Incoming) { - let port_handle = edge.weight().to; - - let edge = edge_and_contexts[edge.id().index()].as_ref().expect( - "This edge has been created from the source node because we traverse in topological order" - ); - - if input_schemas - .insert(port_handle, edge.schema.clone()) - .is_some() - { - return Err(ExecutionError::DuplicateInput { - node: node_handle.clone(), - port: port_handle, - }); - } - } - - for port in input_ports { - if !input_schemas.contains_key(&port) { - return Err(ExecutionError::MissingInput { - node: node_handle.clone(), - port, - }); - } - } - Ok(input_schemas) -} - -#[cfg(test)] -mod tests { - use dozer_types::node::NodeHandle; - - use super::*; - - use crate::{ - tests::{ - processors::{ConnectivityTestProcessorFactory, NoInputPortProcessorFactory}, - sinks::{ConnectivityTestSinkFactory, NoInputPortSinkFactory}, - sources::ConnectivityTestSourceFactory, - }, - DEFAULT_PORT_HANDLE, - }; - - #[test] - #[should_panic] - fn source_with_no_outgoing_edge_should_panic() { - let mut dag = Dag::new(); - dag.add_source( - NodeHandle::new(None, "source".to_string()), - Box::new(ConnectivityTestSourceFactory), - ); - validate_connectivity(&dag); - } - - #[test] - #[should_panic] - fn processor_with_no_outgoing_edge_should_panic() { - let mut dag = Dag::new(); - let source = dag.add_source( - NodeHandle::new(None, "source".to_string()), - Box::new(ConnectivityTestSourceFactory), - ); - let processor = dag.add_processor( - NodeHandle::new(None, "processor".to_string()), - Box::new(ConnectivityTestProcessorFactory), - ); - dag.connect_with_index(source, DEFAULT_PORT_HANDLE, processor, DEFAULT_PORT_HANDLE) - .unwrap(); - validate_connectivity(&dag); - } - - #[test] - #[should_panic] - fn sink_with_no_input_port_should_panic() { - let mut dag = Dag::new(); - dag.add_sink( - NodeHandle::new(None, "sink".to_string()), - Box::new(NoInputPortSinkFactory), - ); - validate_connectivity(&dag); - } - - #[test] - #[should_panic] - fn processor_with_no_input_port_should_panic() { - let mut dag = Dag::new(); - let processor = dag.add_processor( - NodeHandle::new(None, "processor".to_string()), - Box::new(NoInputPortProcessorFactory), - ); - let sink = dag.add_sink( - NodeHandle::new(None, "sink".to_string()), - Box::new(ConnectivityTestSinkFactory), - ); - dag.connect_with_index(processor, DEFAULT_PORT_HANDLE, sink, DEFAULT_PORT_HANDLE) - .unwrap(); - validate_connectivity(&dag); - } - - #[test] - #[should_panic] - fn sink_with_unconnected_input_port_should_panic() { - let mut dag = Dag::new(); - dag.add_sink( - NodeHandle::new(None, "sink".to_string()), - Box::new(ConnectivityTestSinkFactory), - ); - validate_connectivity(&dag); - } - - #[test] - #[should_panic] - fn sink_with_over_connected_input_port_should_panic() { - let mut dag = Dag::new(); - let source1 = dag.add_source( - NodeHandle::new(None, "source1".to_string()), - Box::new(ConnectivityTestSourceFactory), - ); - let source2 = dag.add_source( - NodeHandle::new(None, "source2".to_string()), - Box::new(ConnectivityTestSourceFactory), - ); - let sink = dag.add_sink( - NodeHandle::new(None, "sink".to_string()), - Box::new(ConnectivityTestSinkFactory), - ); - dag.connect_with_index(source1, DEFAULT_PORT_HANDLE, sink, DEFAULT_PORT_HANDLE) - .unwrap(); - dag.connect_with_index(source2, DEFAULT_PORT_HANDLE, sink, DEFAULT_PORT_HANDLE) - .unwrap(); - validate_connectivity(&dag); - } - - #[test] - #[should_panic] - fn processor_with_unconnected_input_port_should_panic() { - let mut dag = Dag::new(); - let processor = dag.add_processor( - NodeHandle::new(None, "processor".to_string()), - Box::new(ConnectivityTestProcessorFactory), - ); - let sink = dag.add_sink( - NodeHandle::new(None, "sink".to_string()), - Box::new(ConnectivityTestSinkFactory), - ); - dag.connect_with_index(processor, DEFAULT_PORT_HANDLE, sink, DEFAULT_PORT_HANDLE) - .unwrap(); - validate_connectivity(&dag); - } - - #[test] - #[should_panic] - fn processor_with_over_connected_input_port_should_panic() { - let mut dag = Dag::new(); - let source1 = dag.add_source( - NodeHandle::new(None, "source1".to_string()), - Box::new(ConnectivityTestSourceFactory), - ); - let source2 = dag.add_source( - NodeHandle::new(None, "source2".to_string()), - Box::new(ConnectivityTestSourceFactory), - ); - let processor = dag.add_processor( - NodeHandle::new(None, "processor".to_string()), - Box::new(ConnectivityTestProcessorFactory), - ); - let sink = dag.add_sink( - NodeHandle::new(None, "sink".to_string()), - Box::new(ConnectivityTestSinkFactory), - ); - dag.connect_with_index(source1, DEFAULT_PORT_HANDLE, processor, DEFAULT_PORT_HANDLE) - .unwrap(); - dag.connect_with_index(source2, DEFAULT_PORT_HANDLE, processor, DEFAULT_PORT_HANDLE) - .unwrap(); - dag.connect_with_index(processor, DEFAULT_PORT_HANDLE, sink, DEFAULT_PORT_HANDLE) - .unwrap(); - validate_connectivity(&dag); - } - - #[test] - fn validate_source_sink_dag() { - let mut dag = Dag::new(); - let source = dag.add_source( - NodeHandle::new(None, "source".to_string()), - Box::new(ConnectivityTestSourceFactory), - ); - let sink = dag.add_sink( - NodeHandle::new(None, "sink".to_string()), - Box::new(ConnectivityTestSinkFactory), - ); - dag.connect_with_index(source, DEFAULT_PORT_HANDLE, sink, DEFAULT_PORT_HANDLE) - .unwrap(); - validate_connectivity(&dag); - } - - #[test] - fn validate_link_shaped_dag() { - let mut dag = Dag::new(); - let source = dag.add_source( - NodeHandle::new(None, "source".to_string()), - Box::new(ConnectivityTestSourceFactory), - ); - let processor = dag.add_processor( - NodeHandle::new(None, "processor1".to_string()), - Box::new(ConnectivityTestProcessorFactory), - ); - let sink = dag.add_sink( - NodeHandle::new(None, "sink".to_string()), - Box::new(ConnectivityTestSinkFactory), - ); - dag.connect_with_index(source, DEFAULT_PORT_HANDLE, processor, DEFAULT_PORT_HANDLE) - .unwrap(); - dag.connect_with_index(processor, DEFAULT_PORT_HANDLE, sink, DEFAULT_PORT_HANDLE) - .unwrap(); - validate_connectivity(&dag); - } -} diff --git a/dozer-core/src/error_manager.rs b/dozer-core/src/error_manager.rs deleted file mode 100644 index 3f876ef739..0000000000 --- a/dozer-core/src/error_manager.rs +++ /dev/null @@ -1,42 +0,0 @@ -use std::sync::atomic::AtomicU32; - -use dozer_types::tracing::error_span; -use dozer_types::{errors::internal::BoxedError, log::error}; - -/// `ErrorManager` records and counts the number of errors happened. -/// -/// It panics when an error threshold is set and reached. -#[derive(Debug)] -pub struct ErrorManager { - threshold: Option, - count: AtomicU32, -} - -impl ErrorManager { - pub fn new_threshold(threshold: u32) -> Self { - Self { - threshold: Some(threshold), - count: AtomicU32::new(0), - } - } - - pub fn new_unlimited() -> Self { - Self { - threshold: None, - count: AtomicU32::new(0), - } - } - - pub fn report(&self, error: BoxedError) { - let err_span = error_span!("reported error", error = true, e = error); - let _error_guard = err_span.enter(); - error!("{}", error); - - let count = self.count.fetch_add(1, std::sync::atomic::Ordering::SeqCst); - if let Some(threshold) = self.threshold { - if count >= threshold { - panic!("Error threshold reached: {}", threshold); - } - } - } -} diff --git a/dozer-core/src/errors.rs b/dozer-core/src/errors.rs deleted file mode 100644 index 72d8493045..0000000000 --- a/dozer-core/src/errors.rs +++ /dev/null @@ -1,66 +0,0 @@ -use std::path::PathBuf; - -use crate::node::PortHandle; -use dozer_types::errors::internal::BoxedError; -use dozer_types::errors::types::{DeserializationError, SerializationError}; -use dozer_types::node::NodeHandle; -use dozer_types::thiserror::Error; -use dozer_types::{bincode, thiserror}; - -#[derive(Error, Debug)] -pub enum ExecutionError { - #[error("Adding this edge would have created a cycle")] - WouldCycle, - #[error("Invalid port handle: {0}")] - InvalidPortHandle(PortHandle), - #[error("Missing input for node {node} on port {port}")] - MissingInput { node: NodeHandle, port: PortHandle }, - #[error("Duplicate input for node {node} on port {port}")] - DuplicateInput { node: NodeHandle, port: PortHandle }, - #[error("Cannot send to channel")] - CannotSendToChannel, - #[error("Cannot receive from channel")] - CannotReceiveFromChannel, - #[error("Cannot spawn worker thread: {0}")] - CannotSpawnWorkerThread(#[source] std::io::Error), - #[error("Invalid source name {0}")] - InvalidSourceIdentifier(String), - #[error("Ambiguous source name {0}")] - AmbiguousSourceIdentifier(String), - #[error("Invalid AppSource connection {0}. Already exists.")] - AppSourceConnectionAlreadyExists(String), - #[error("Factory error: {0}")] - Factory(#[source] BoxedError), - #[error("Failed to restore record writer: {0}")] - RestoreRecordWriter(#[source] DeserializationError), - #[error("Source error: {0}")] - Source(#[source] BoxedError), - #[error("Sink error: {0}")] - Sink(#[source] BoxedError), - #[error("State of {0} is not consistent across sinks")] - SourceStateConflict(NodeHandle), - #[error("File system error {0:?}: {1}")] - FileSystemError(PathBuf, #[source] std::io::Error), - #[error("Checkpoint writer thread panicked")] - CheckpointWriterThreadPanicked, - #[error("Cannot deserialize checkpoint: {0}")] - CorruptedCheckpoint(#[source] bincode::error::DecodeError), - #[error("Source {0} cannot restart. You have to clean data from previous runs by running `dozer clean`")] - SourceCannotRestart(NodeHandle), - #[error("Failed to create checkpoint: {0}")] - FailedToCreateCheckpoint(BoxedError), - #[error("Failed to serialize record writer: {0}")] - SerializeRecordWriter(#[source] SerializationError), -} - -impl From> for ExecutionError { - fn from(_: crossbeam::channel::SendError) -> Self { - ExecutionError::CannotSendToChannel - } -} - -impl From> for ExecutionError { - fn from(_: daggy::WouldCycle) -> Self { - ExecutionError::WouldCycle - } -} diff --git a/dozer-core/src/executor/execution_dag.rs b/dozer-core/src/executor/execution_dag.rs deleted file mode 100644 index f492bd4995..0000000000 --- a/dozer-core/src/executor/execution_dag.rs +++ /dev/null @@ -1,254 +0,0 @@ -use std::{ - collections::{hash_map::Entry, HashMap}, - fmt::Debug, - sync::Arc, -}; - -use crate::{ - builder_dag::{BuilderDag, NodeKind}, - dag_schemas::EdgeKind, - error_manager::ErrorManager, - errors::ExecutionError, - event::EventHub, - executor_operation::ExecutorOperation, - forwarder::SenderWithPortMapping, - hash_map_to_vec::insert_vec_element, - node::{OutputPortType, PortHandle}, - record_store::{create_record_writer, RecordWriter}, -}; -use crossbeam::channel::{bounded, Receiver, Sender}; -use daggy::petgraph::{ - visit::{EdgeRef, IntoEdges, IntoEdgesDirected}, - Direction, -}; -use dozer_tracing::DozerMonitorContext; -use dozer_types::node::NodeHandle; -use tokio::sync::Mutex; - -#[derive(Debug)] -pub struct NodeType { - pub handle: NodeHandle, - pub kind: Option, -} - -type SharedRecordWriter = Arc>>>; - -#[derive(Debug, Clone)] -pub struct EdgeType { - /// Output port handle. - pub output_port: PortHandle, - /// Edge kind. - pub edge_kind: EdgeKind, - /// The sender for data flowing downstream. Edges that have same source and target node share the same sender. - pub sender: Sender, - /// The record writer for persisting data for downstream queries, if persistency is needed. Different edges with the same output port share the same record writer. - pub record_writer: SharedRecordWriter, - /// Input port handle. - pub input_port: PortHandle, - /// The receiver from receiving data from upstream. Edges that have same source and target node share the same receiver. - pub receiver: Receiver, -} - -#[derive(Debug)] -pub struct ExecutionDag { - /// Nodes will be moved into execution threads. - graph: daggy::Dag, - initial_epoch_id: u64, - error_manager: Arc, - labels: DozerMonitorContext, - event_hub: EventHub, -} - -impl ExecutionDag { - pub fn new( - builder_dag: BuilderDag, - labels: DozerMonitorContext, - channel_buffer_sz: usize, - error_threshold: Option, - ) -> Result { - // We only create record writer once for every output port. Every `HashMap` in this `Vec` tracks if a node's output ports already have the record writer created. - let mut all_record_writers = vec![ - HashMap::::new(); - builder_dag.graph().node_count() - ]; - // We only create channel once for every pair of nodes. - let mut channels = HashMap::< - (daggy::NodeIndex, daggy::NodeIndex), - (Sender, Receiver), - >::new(); - - // Create new edges. - let mut edges = vec![]; - for builder_dag_edge in builder_dag.graph().raw_edges().iter() { - let source_node_index = builder_dag_edge.source(); - let target_node_index = builder_dag_edge.target(); - let edge = &builder_dag_edge.weight; - let output_port = edge.output_port; - let edge_kind = edge.edge_kind.clone(); - - // Create or get record writer. - let record_writer = - match all_record_writers[source_node_index.index()].entry(output_port) { - Entry::Vacant(entry) => { - let record_writer = match &edge_kind { - EdgeKind::FromSource { - port_type: OutputPortType::StatefulWithPrimaryKeyLookup, - .. - } => Some( - create_record_writer(edge.schema.clone()) - .map_err(ExecutionError::RestoreRecordWriter)?, - ), - _ => None, - }; - let record_writer = Arc::new(Mutex::new(record_writer)); - entry.insert(record_writer).clone() - } - Entry::Occupied(entry) => entry.get().clone(), - }; - - // Create or get channel. - let (sender, receiver) = match channels.entry((source_node_index, target_node_index)) { - Entry::Vacant(entry) => { - let (sender, receiver) = bounded(channel_buffer_sz); - entry.insert((sender.clone(), receiver.clone())); - (sender, receiver) - } - Entry::Occupied(entry) => entry.get().clone(), - }; - - // Create edge. - let edge = EdgeType { - output_port, - edge_kind, - sender, - record_writer, - input_port: edge.input_port, - receiver, - }; - edges.push(Some(edge)); - } - - // Create new graph. - let (graph, event_hub) = builder_dag.into_graph_and_event_hub(); - let graph = graph.map_owned( - |_, node| NodeType { - handle: node.handle, - kind: Some(node.kind), - }, - |edge_index, _| { - edges[edge_index.index()] - .take() - .expect("We created all edges") - }, - ); - Ok(ExecutionDag { - graph, - initial_epoch_id: 0, - error_manager: Arc::new(if let Some(threshold) = error_threshold { - ErrorManager::new_threshold(threshold) - } else { - ErrorManager::new_unlimited() - }), - labels, - event_hub, - }) - } - - pub fn graph(&self) -> &daggy::Dag { - &self.graph - } - - pub fn node_weight_mut(&mut self, node_index: daggy::NodeIndex) -> &mut NodeType { - &mut self.graph[node_index] - } - - pub fn initial_epoch_id(&self) -> u64 { - self.initial_epoch_id - } - - pub fn error_manager(&self) -> &Arc { - &self.error_manager - } - - pub fn labels(&self) -> &DozerMonitorContext { - &self.labels - } - - pub fn event_hub(&self) -> &EventHub { - &self.event_hub - } - - pub fn collect_senders(&self, node_index: daggy::NodeIndex) -> Vec { - // Map from target node index to `SenderWithPortMapping`. - let mut senders = HashMap::::new(); - for edge in self.graph.edges(node_index) { - match senders.entry(edge.target()) { - Entry::Vacant(entry) => { - let port_mapping = - [(edge.weight().output_port, vec![edge.weight().input_port])] - .into_iter() - .collect(); - entry.insert(SenderWithPortMapping { - sender: edge.weight().sender.clone(), - port_mapping, - }); - } - Entry::Occupied(mut entry) => { - insert_vec_element( - &mut entry.get_mut().port_mapping, - edge.weight().output_port, - edge.weight().input_port, - ); - } - } - } - senders.into_values().collect() - } - - pub async fn collect_record_writers( - &mut self, - node_index: daggy::NodeIndex, - ) -> HashMap> { - let edge_indexes = self - .graph - .edges(node_index) - .map(|edge| edge.id()) - .collect::>(); - - let mut record_writers = HashMap::new(); - for edge_index in edge_indexes { - let edge = self - .graph - .edge_weight_mut(edge_index) - .expect("We don't modify graph structure, only modify the edge weight"); - - if let Entry::Vacant(entry) = record_writers.entry(edge.output_port) { - // This interior mutability is to work around `Arc`. Other parts of this function is correctly marked `mut`. - if let Some(record_writer) = edge.record_writer.lock().await.take() { - entry.insert(record_writer); - } - } - } - - record_writers - } - - pub fn collect_receivers( - &self, - node_index: daggy::NodeIndex, - ) -> (Vec, Vec>) { - // Map from source node index to source node handle and the receiver to receiver from source. - let mut handles_and_receivers = - HashMap::)>::new(); - for edge in self.graph.edges_directed(node_index, Direction::Incoming) { - let source_node_index = edge.source(); - if let Entry::Vacant(entry) = handles_and_receivers.entry(source_node_index) { - entry.insert(( - self.graph[source_node_index].handle.clone(), - edge.weight().receiver.clone(), - )); - } - } - handles_and_receivers.into_values().unzip() - } -} diff --git a/dozer-core/src/executor/mod.rs b/dozer-core/src/executor/mod.rs deleted file mode 100644 index 2b3b2a7a55..0000000000 --- a/dozer-core/src/executor/mod.rs +++ /dev/null @@ -1,181 +0,0 @@ -use crate::builder_dag::{BuilderDag, NodeKind}; -use crate::dag_schemas::DagSchemas; -use crate::errors::ExecutionError; -use crate::Dag; - -use daggy::petgraph::visit::IntoNodeIdentifiers; - -use dozer_tracing::DozerMonitorContext; -use futures::Future; -use std::fmt::Debug; -use std::sync::Arc; -use std::thread::JoinHandle; -use std::thread::{self, Builder}; -use std::time::Duration; -use tokio::runtime::Runtime; - -#[derive(Debug, Clone)] -pub struct ExecutorOptions { - pub channel_buffer_sz: usize, - pub event_hub_capacity: usize, - pub error_threshold: Option, -} - -impl Default for ExecutorOptions { - fn default() -> Self { - Self { - channel_buffer_sz: 20_000, - event_hub_capacity: 100, - error_threshold: Some(0), - } - } -} - -mod execution_dag; -mod name; -mod node; -mod processor_node; -mod receiver_loop; -mod sink_node; -mod source_node; - -use node::Node; -use processor_node::ProcessorNode; -use sink_node::SinkNode; - -use self::execution_dag::ExecutionDag; -use self::source_node::{create_source_node, SourceNode}; - -pub struct DagExecutor { - builder_dag: BuilderDag, - options: ExecutorOptions, -} - -pub struct DagExecutorJoinHandle { - join_handles: Vec>>, -} - -impl DagExecutor { - pub async fn new(dag: Dag, options: ExecutorOptions) -> Result { - let dag_schemas = DagSchemas::new(dag).await?; - - let builder_dag = BuilderDag::new(dag_schemas, options.event_hub_capacity).await?; - - Ok(Self { - builder_dag, - options, - }) - } - - pub async fn validate(dag: Dag) -> Result<(), ExecutionError> { - DagSchemas::new(dag).await?; - Ok(()) - } - - pub async fn start( - self, - shutdown: F, - labels: DozerMonitorContext, - runtime: Arc, - ) -> Result { - // Construct execution dag. - let mut execution_dag = ExecutionDag::new( - self.builder_dag, - labels, - self.options.channel_buffer_sz, - self.options.error_threshold, - )?; - let node_indexes = execution_dag.graph().node_identifiers().collect::>(); - - // Start the threads. - let source_node = - create_source_node(&mut execution_dag, &self.options, shutdown, runtime.clone()).await; - let mut join_handles = vec![start_source(source_node)?]; - for node_index in node_indexes { - let Some(node) = execution_dag.graph()[node_index].kind.as_ref() else { - continue; - }; - match node { - NodeKind::Source { .. } => unreachable!("We already started the source node"), - NodeKind::Processor(_) => { - let processor_node = ProcessorNode::new(&mut execution_dag, node_index).await; - join_handles.push(start_processor(processor_node)?); - } - NodeKind::Sink(_) => { - let sink_node = SinkNode::new(&mut execution_dag, node_index); - join_handles.push(start_sink(sink_node)?); - } - } - } - - Ok(DagExecutorJoinHandle { join_handles }) - } -} - -impl DagExecutorJoinHandle { - pub fn join(mut self) -> Result<(), ExecutionError> { - loop { - let Some(finished) = self - .join_handles - .iter() - .enumerate() - .find_map(|(i, handle)| handle.is_finished().then_some(i)) - else { - thread::sleep(Duration::from_millis(250)); - - continue; - }; - let handle = self.join_handles.swap_remove(finished); - handle.join().unwrap()?; - - if self.join_handles.is_empty() { - return Ok(()); - } - } - } -} - -fn start_source( - source: SourceNode, -) -> Result>, ExecutionError> { - let handle = Builder::new() - .name("sources".into()) - .spawn(move || match source.run() { - Ok(()) => Ok(()), - // Channel disconnection means the source listener has quit. - // Maybe it quit gracefully so we don't need to propagate the error. - Err(e) => { - if let ExecutionError::Source(e) = &e { - if let Some(ExecutionError::CannotSendToChannel) = e.downcast_ref() { - return Ok(()); - } - } - Err(e) - } - }) - .map_err(ExecutionError::CannotSpawnWorkerThread)?; - - Ok(handle) -} - -fn start_processor( - processor: ProcessorNode, -) -> Result>, ExecutionError> { - Builder::new() - .name(processor.handle().to_string()) - .spawn(move || { - processor.run()?; - Ok(()) - }) - .map_err(ExecutionError::CannotSpawnWorkerThread) -} - -fn start_sink(sink: SinkNode) -> Result>, ExecutionError> { - Builder::new() - .name(sink.handle().to_string()) - .spawn(|| { - sink.run()?; - Ok(()) - }) - .map_err(ExecutionError::CannotSpawnWorkerThread) -} diff --git a/dozer-core/src/executor/name.rs b/dozer-core/src/executor/name.rs deleted file mode 100644 index 16686e8b14..0000000000 --- a/dozer-core/src/executor/name.rs +++ /dev/null @@ -1,5 +0,0 @@ -use std::borrow::Cow; - -pub trait Name { - fn name(&self) -> Cow; -} diff --git a/dozer-core/src/executor/node.rs b/dozer-core/src/executor/node.rs deleted file mode 100644 index 2e841a5a1d..0000000000 --- a/dozer-core/src/executor/node.rs +++ /dev/null @@ -1,18 +0,0 @@ -use std::fmt::Debug; - -use crate::errors::ExecutionError; - -use super::receiver_loop::ReceiverLoop; - -/// A node in the execution DAG. -pub trait Node { - /// Runs the node. - fn run(self) -> Result<(), ExecutionError>; -} - -impl Node for T { - fn run(self) -> Result<(), ExecutionError> { - let initial_epoch_id = self.initial_epoch_id(); - self.receiver_loop(initial_epoch_id) - } -} diff --git a/dozer-core/src/executor/processor_node.rs b/dozer-core/src/executor/processor_node.rs deleted file mode 100644 index bc345aa9b8..0000000000 --- a/dozer-core/src/executor/processor_node.rs +++ /dev/null @@ -1,129 +0,0 @@ -use std::sync::Arc; -use std::{borrow::Cow, mem::swap}; - -use crossbeam::channel::Receiver; -use daggy::NodeIndex; -use dozer_types::node::{NodeHandle, OpIdentifier}; -use dozer_types::types::TableOperation; - -use crate::epoch::Epoch; -use crate::error_manager::ErrorManager; -use crate::executor_operation::ExecutorOperation; -use crate::{ - builder_dag::NodeKind, errors::ExecutionError, forwarder::ChannelManager, node::Processor, -}; - -use super::{execution_dag::ExecutionDag, name::Name, receiver_loop::ReceiverLoop}; - -/// A processor in the execution DAG. -#[derive(Debug)] -pub struct ProcessorNode { - /// Node handle in description DAG. - node_handle: NodeHandle, - /// The epoch id the processor was constructed for. - initial_epoch_id: u64, - /// Input node handles. - node_handles: Vec, - /// Input data channels. - receivers: Vec>, - /// The processor. - processor: Box, - /// This node's output channel manager, for forwarding data, writing metadata and writing port state. - channel_manager: ChannelManager, - /// The error manager, for reporting non-fatal errors. - error_manager: Arc, -} - -impl ProcessorNode { - pub async fn new(dag: &mut ExecutionDag, node_index: NodeIndex) -> Self { - let node = dag.node_weight_mut(node_index); - let Some(kind) = node.kind.take() else { - panic!("Must pass in a node") - }; - let node_handle = node.handle.clone(); - let NodeKind::Processor(processor) = kind else { - panic!("Must pass in a processor node"); - }; - - let (node_handles, receivers) = dag.collect_receivers(node_index); - - let senders = dag.collect_senders(node_index); - let record_writers = dag.collect_record_writers(node_index).await; - - let channel_manager = ChannelManager::new( - node_handle.clone(), - record_writers, - senders, - dag.error_manager().clone(), - ); - - Self { - node_handle, - initial_epoch_id: dag.initial_epoch_id(), - node_handles, - receivers, - processor, - channel_manager, - error_manager: dag.error_manager().clone(), - } - } - - pub fn handle(&self) -> &NodeHandle { - &self.node_handle - } -} - -impl Name for ProcessorNode { - fn name(&self) -> Cow { - Cow::Owned(self.node_handle.to_string()) - } -} - -impl ReceiverLoop for ProcessorNode { - fn initial_epoch_id(&self) -> u64 { - self.initial_epoch_id - } - - fn receivers(&mut self) -> Vec> { - let mut result = vec![]; - swap(&mut self.receivers, &mut result); - result - } - - fn receiver_name(&self, index: usize) -> Cow { - Cow::Owned(self.node_handles[index].to_string()) - } - - fn on_op(&mut self, _index: usize, op: TableOperation) -> Result<(), ExecutionError> { - if let Err(e) = self.processor.process(op, &mut self.channel_manager) { - self.error_manager.report(e); - } - Ok(()) - } - - fn on_commit(&mut self, epoch: Epoch) -> Result<(), ExecutionError> { - if let Err(e) = self.processor.commit(&epoch) { - self.error_manager.report(e); - } - - self.channel_manager.send_commit(epoch) - } - - fn on_terminate(&mut self) -> Result<(), ExecutionError> { - self.channel_manager.send_terminate() - } - - fn on_snapshotting_started(&mut self, connection_name: String) -> Result<(), ExecutionError> { - self.channel_manager - .send_snapshotting_started(connection_name) - } - - fn on_snapshotting_done( - &mut self, - connection_name: String, - id: Option, - ) -> Result<(), ExecutionError> { - self.channel_manager - .send_snapshotting_done(connection_name, id) - } -} diff --git a/dozer-core/src/executor/receiver_loop.rs b/dozer-core/src/executor/receiver_loop.rs deleted file mode 100644 index f4f3bf7b8c..0000000000 --- a/dozer-core/src/executor/receiver_loop.rs +++ /dev/null @@ -1,353 +0,0 @@ -use std::borrow::Cow; - -use crossbeam::channel::{Receiver, Select}; -use dozer_types::{log::debug, node::OpIdentifier, types::TableOperation}; - -use crate::{epoch::Epoch, errors::ExecutionError, executor_operation::ExecutorOperation}; - -use super::name::Name; - -/// Common code for processor and sink nodes. -/// -/// They both select from their input channels, and respond to "op", "commit", and terminate. -pub trait ReceiverLoop: Name { - /// Returns the epoch id that this node was constructed for. - fn initial_epoch_id(&self) -> u64; - /// Returns input channels to this node. Will be called exactly once in [`receiver_loop`]. - fn receivers(&mut self) -> Vec>; - /// Returns the name of the receiver at `index`. Used for logging. - fn receiver_name(&self, index: usize) -> Cow; - /// Responds to `op` from the receiver at `index`. - fn on_op(&mut self, index: usize, op: TableOperation) -> Result<(), ExecutionError>; - /// Responds to `commit` of `epoch`. - fn on_commit(&mut self, epoch: Epoch) -> Result<(), ExecutionError>; - /// Responds to `terminate`. - fn on_terminate(&mut self) -> Result<(), ExecutionError>; - /// Responds to `SnapshottingStarted`. - fn on_snapshotting_started(&mut self, connection_name: String) -> Result<(), ExecutionError>; - /// Responds to `SnapshottingDone`. - fn on_snapshotting_done( - &mut self, - connection_name: String, - id: Option, - ) -> Result<(), ExecutionError>; - - /// The loop implementation, calls [`on_op`], [`on_commit`] and [`on_terminate`] at appropriate times. - fn receiver_loop(mut self, initial_epoch_id: u64) -> Result<(), ExecutionError> - where - Self: Sized, - { - let receivers = self.receivers(); - debug_assert!( - !receivers.is_empty(), - "Processor or sink must have at least 1 incoming edge" - ); - let mut is_terminated = vec![false; receivers.len()]; - - let mut commits_received: usize = 0; - let mut epoch_id = initial_epoch_id; - - let mut sel = init_select(&receivers); - loop { - let index = sel.ready(); - let op = receivers[index] - .recv() - .map_err(|_| ExecutionError::CannotReceiveFromChannel)?; - - match op { - ExecutorOperation::Op { op } => { - self.on_op(index, op)?; - } - ExecutorOperation::Commit { epoch } => { - assert_eq!(epoch.common_info.id, epoch_id); - commits_received += 1; - sel.remove(index); - - if commits_received == receivers.len() { - self.on_commit(epoch)?; - epoch_id += 1; - commits_received = 0; - sel = init_select(&receivers); - } - } - ExecutorOperation::Terminate => { - is_terminated[index] = true; - sel.remove(index); - debug!( - "[{}] Received Terminate request from {}", - self.name(), - self.receiver_name(index) - ); - if is_terminated.iter().all(|value| *value) { - self.on_terminate()?; - debug!("[{}] Quit", self.name()); - return Ok(()); - } - } - ExecutorOperation::SnapshottingStarted { connection_name } => { - self.on_snapshotting_started(connection_name)?; - } - ExecutorOperation::SnapshottingDone { - connection_name, - id, - } => { - self.on_snapshotting_done(connection_name, id)?; - } - } - } - } -} - -pub(crate) fn init_select(receivers: &Vec>) -> Select { - let mut sel = Select::new(); - for r in receivers { - sel.recv(r); - } - sel -} - -#[cfg(test)] -mod tests { - use std::{cell::RefCell, mem::swap, rc::Rc, sync::Arc, time::SystemTime}; - - use crossbeam::channel::{unbounded, Sender}; - use dozer_types::{ - node::{NodeHandle, SourceState, SourceStates}, - types::{Field, Operation, Record}, - }; - - use crate::DEFAULT_PORT_HANDLE; - - use super::*; - - #[derive(Clone)] - struct TestReceiverLoopState { - ops: Vec<(usize, TableOperation)>, - commits: Vec, - snapshotting_started: Vec, - snapshotting_done: Vec<(String, Option)>, - num_terminations: usize, - } - - struct TestReceiverLoop { - receivers: Vec>, - state: Rc>, - } - - impl Name for TestReceiverLoop { - fn name(&self) -> Cow { - Cow::Borrowed("TestReceiverLoop") - } - } - - impl ReceiverLoop for TestReceiverLoop { - fn initial_epoch_id(&self) -> u64 { - 0 - } - - fn receivers(&mut self) -> Vec> { - let mut result = vec![]; - swap(&mut self.receivers, &mut result); - result - } - - fn receiver_name(&self, index: usize) -> Cow { - Cow::Owned(format!("receiver_{index}")) - } - - fn on_op(&mut self, index: usize, op: TableOperation) -> Result<(), ExecutionError> { - self.state.borrow_mut().ops.push((index, op)); - Ok(()) - } - - fn on_commit(&mut self, epoch: Epoch) -> Result<(), ExecutionError> { - self.state.borrow_mut().commits.push(epoch); - Ok(()) - } - - fn on_terminate(&mut self) -> Result<(), ExecutionError> { - self.state.borrow_mut().num_terminations += 1; - Ok(()) - } - - fn on_snapshotting_started( - &mut self, - connection_name: String, - ) -> Result<(), ExecutionError> { - self.state - .borrow_mut() - .snapshotting_started - .push(connection_name); - Ok(()) - } - - fn on_snapshotting_done( - &mut self, - connection_name: String, - state: Option, - ) -> Result<(), ExecutionError> { - self.state - .borrow_mut() - .snapshotting_done - .push((connection_name, state)); - Ok(()) - } - } - - impl TestReceiverLoop { - fn new( - num_receivers: usize, - ) -> ( - TestReceiverLoop, - Vec>, - Rc>, - ) { - let (senders, receivers) = (0..num_receivers).map(|_| unbounded()).unzip(); - let state = Rc::new(RefCell::new(TestReceiverLoopState { - ops: vec![], - commits: vec![], - snapshotting_started: vec![], - snapshotting_done: vec![], - num_terminations: 0, - })); - ( - TestReceiverLoop { - receivers, - state: state.clone(), - }, - senders, - state, - ) - } - } - - #[test] - fn receiver_loop_stops_on_terminate() { - let (test_loop, senders, state) = TestReceiverLoop::new(2); - let test_loop = Box::new(test_loop); - senders[0].send(ExecutorOperation::Terminate).unwrap(); - senders[1].send(ExecutorOperation::Terminate).unwrap(); - test_loop.receiver_loop(0).unwrap(); - assert_eq!(state.borrow().num_terminations, 1); - } - - #[test] - fn receiver_loop_forwards_snapshotting_done() { - let connection_name = "test_connection".to_string(); - let (test_loop, senders, state) = TestReceiverLoop::new(2); - senders[0] - .send(ExecutorOperation::SnapshottingDone { - connection_name: connection_name.clone(), - id: None, - }) - .unwrap(); - senders[0].send(ExecutorOperation::Terminate).unwrap(); - senders[1].send(ExecutorOperation::Terminate).unwrap(); - test_loop.receiver_loop(0).unwrap(); - let snapshotting_done = state.borrow().snapshotting_done.clone(); - assert_eq!(snapshotting_done, vec![(connection_name, None)]) - } - - #[test] - fn receiver_loop_forwards_op() { - let (test_loop, senders, state) = TestReceiverLoop::new(2); - let record = Record::new(vec![Field::Int(1)]); - senders[0] - .send(ExecutorOperation::Op { - op: TableOperation::without_id( - Operation::Insert { - new: record.clone(), - }, - DEFAULT_PORT_HANDLE, - ), - }) - .unwrap(); - senders[0].send(ExecutorOperation::Terminate).unwrap(); - senders[1].send(ExecutorOperation::Terminate).unwrap(); - test_loop.receiver_loop(0).unwrap(); - assert_eq!( - state.borrow().ops, - vec![( - 0, - TableOperation::without_id(Operation::Insert { new: record }, DEFAULT_PORT_HANDLE,) - )] - ); - } - - #[test] - fn receiver_loop_increases_epoch_id() { - let (test_loop, senders, state) = TestReceiverLoop::new(2); - let mut source_states = SourceStates::default(); - source_states.insert( - NodeHandle::new(None, "0".to_string()), - SourceState::NotStarted, - ); - source_states.insert( - NodeHandle::new(None, "1".to_string()), - SourceState::NotStarted, - ); - let source_states = Arc::new(source_states); - let decision_instant = SystemTime::now(); - let mut epoch0 = Epoch::new(0, source_states.clone(), decision_instant); - let mut epoch1 = Epoch::new(0, source_states, decision_instant); - senders[0] - .send(ExecutorOperation::Commit { - epoch: epoch0.clone(), - }) - .unwrap(); - senders[1] - .send(ExecutorOperation::Commit { - epoch: epoch1.clone(), - }) - .unwrap(); - epoch0.common_info.id = 1; - epoch1.common_info.id = 1; - senders[0] - .send(ExecutorOperation::Commit { - epoch: epoch0.clone(), - }) - .unwrap(); - senders[1] - .send(ExecutorOperation::Commit { - epoch: epoch1.clone(), - }) - .unwrap(); - senders[0].send(ExecutorOperation::Terminate).unwrap(); - senders[1].send(ExecutorOperation::Terminate).unwrap(); - test_loop.receiver_loop(0).unwrap(); - - let state = state.borrow(); - assert_eq!(state.commits[0].common_info.id, 0); - assert_eq!(state.commits[0].decision_instant, decision_instant); - assert_eq!(state.commits[1].common_info.id, 1); - assert_eq!(state.commits[1].decision_instant, decision_instant); - } - - #[test] - #[should_panic] - fn receiver_loop_panics_on_inconsistent_commit_epoch() { - let (test_loop, senders, _) = TestReceiverLoop::new(2); - let mut source_states = SourceStates::new(); - source_states.insert( - NodeHandle::new(None, "0".to_string()), - SourceState::NotStarted, - ); - source_states.insert( - NodeHandle::new(None, "1".to_string()), - SourceState::NotStarted, - ); - let source_states = Arc::new(source_states); - let decision_instant = SystemTime::now(); - let epoch0 = Epoch::new(0, source_states.clone(), decision_instant); - let epoch1 = Epoch::new(1, source_states, decision_instant); - senders[0] - .send(ExecutorOperation::Commit { epoch: epoch0 }) - .unwrap(); - senders[1] - .send(ExecutorOperation::Commit { epoch: epoch1 }) - .unwrap(); - senders[0].send(ExecutorOperation::Terminate).unwrap(); - senders[1].send(ExecutorOperation::Terminate).unwrap(); - test_loop.receiver_loop(0).unwrap(); - } -} diff --git a/dozer-core/src/executor/sink_node.rs b/dozer-core/src/executor/sink_node.rs deleted file mode 100644 index 79fa4fb38c..0000000000 --- a/dozer-core/src/executor/sink_node.rs +++ /dev/null @@ -1,508 +0,0 @@ -use crossbeam::channel::{Receiver, Sender, TryRecvError}; -use daggy::NodeIndex; -use dozer_tracing::{ - constants::{ - ConnectorEntityType, DOZER_METER_NAME, OPERATION_TYPE_LABEL, PIPELINE_LATENCY_GAUGE_NAME, - SINK_OPERATION_COUNTER_NAME, TABLE_LABEL, TOTAL_LATENCY_HISTOGRAM_NAME, - }, - emit_event, - opentelemetry_metrics::{Counter, Gauge, Histogram}, - DozerMonitorContext, -}; -use dozer_types::{epoch::SourceTime, log::warn}; -use dozer_types::{ - log::debug, - node::{NodeHandle, OpIdentifier}, - tracing::error, - types::{Operation, TableOperation}, -}; -use std::{ - borrow::Cow, - mem::swap, - sync::Arc, - time::{Duration, Instant}, - usize, -}; -use tokio::sync::broadcast; - -use crate::{ - builder_dag::NodeKind, epoch::Epoch, error_manager::ErrorManager, errors::ExecutionError, - event::Event, executor_operation::ExecutorOperation, node::Sink, -}; - -use super::execution_dag::ExecutionDag; -use super::{name::Name, receiver_loop::ReceiverLoop}; - -const DEFAULT_FLUSH_INTERVAL: Duration = Duration::from_millis(20); - -struct FlushScheduler { - receiver: Receiver, - sender: Sender<()>, - next_schedule: Option, - next_schedule_from: Instant, - loop_interval: Duration, -} - -impl FlushScheduler { - fn run(&mut self) { - loop { - // If we have nothing scheduled, block until we get a schedule - let mut next_schedule = if self.next_schedule.is_none() { - match self.receiver.recv() { - Ok(v) => Some(v), - Err(_) => return, - } - } else { - None - }; - - // Keep postponing the schedule while there are messages - while let Some(sched) = match self.receiver.try_recv() { - Ok(next) => Some(next), - Err(TryRecvError::Empty) => None, - Err(TryRecvError::Disconnected) => return, - } { - next_schedule = Some(sched); - } - - if let Some(next) = next_schedule { - self.next_schedule = Some(next); - self.next_schedule_from = Instant::now(); - } - - let Some(schedule) = self.next_schedule else { - continue; - }; - - let elapsed = self.next_schedule_from.elapsed(); - if elapsed >= schedule { - let Ok(_) = self.sender.send(()) else { - return; - }; - self.next_schedule = None; - } else { - let time_to_next_schedule = schedule - elapsed; - std::thread::sleep(self.loop_interval.min(time_to_next_schedule)); - } - } - } -} - -/// A sink in the execution DAG. -#[derive(Debug)] -pub struct SinkNode { - /// Node handle in description DAG. - node_handle: NodeHandle, - /// The epoch id the sink was constructed for. - initial_epoch_id: u64, - /// Input node handles. - node_handles: Vec, - /// Input data channels. - receivers: Vec>, - /// The sink. - sink: Box, - /// The error manager, for reporting non-fatal errors. - error_manager: Arc, - /// The metrics labels. - labels: DozerMonitorContext, - - max_flush_interval: Duration, - - ops_since_flush: u64, - last_op_if_commit: Option, - flush_scheduled_on_next_commit: bool, - flush_scheduler_sender: Sender, - should_flush_receiver: Receiver<()>, - - event_sender: broadcast::Sender, - metrics: SinkMetrics, - source_times: Option>, -} - -#[derive(Debug)] - -pub struct SinkMetrics { - sink_counter: Counter, - latency_gauge: Gauge, - total_latency_hist: Histogram, -} - -impl SinkNode { - pub fn new(dag: &mut ExecutionDag, node_index: NodeIndex) -> Self { - let node = dag.node_weight_mut(node_index); - let Some(kind) = node.kind.take() else { - panic!("Must pass in a node") - }; - let node_handle = node.handle.clone(); - let NodeKind::Sink(sink) = kind else { - panic!("Must pass in a sink node"); - }; - - let (node_handles, receivers) = dag.collect_receivers(node_index); - - let meter = dozer_tracing::global::meter(DOZER_METER_NAME); - let sink_counter = meter - .u64_counter(SINK_OPERATION_COUNTER_NAME) - .with_description("No of operations in the sink node") - .init(); - let latency_gauge = meter - .f64_gauge(PIPELINE_LATENCY_GAUGE_NAME) - .with_description("Mesasures latency between commits") - .init(); - - let total_latency_hist = meter - .u64_histogram(TOTAL_LATENCY_HISTOGRAM_NAME) - .with_description("Measures total latency between commit on source and commit on sink") - .init(); - - let max_flush_interval = sink - .max_batch_duration_ms() - .map_or(DEFAULT_FLUSH_INTERVAL, Duration::from_millis); - let (schedule_sender, schedule_receiver) = crossbeam::channel::bounded(10); - let (should_flush_sender, should_flush_receiver) = crossbeam::channel::bounded(0); - let mut scheduler = FlushScheduler { - receiver: schedule_receiver, - sender: should_flush_sender, - next_schedule: None, - next_schedule_from: Instant::now(), - loop_interval: max_flush_interval / 5, - }; - - std::thread::spawn(move || scheduler.run()); - let source_times = sink.supports_batching().then(Vec::new); - - Self { - node_handle, - initial_epoch_id: dag.initial_epoch_id(), - node_handles, - receivers, - sink, - error_manager: dag.error_manager().clone(), - labels: dag.labels().clone(), - last_op_if_commit: None, - flush_scheduled_on_next_commit: false, - flush_scheduler_sender: schedule_sender, - should_flush_receiver, - event_sender: dag.event_hub().sender.clone(), - max_flush_interval, - ops_since_flush: 0, - source_times, - metrics: SinkMetrics { - sink_counter, - latency_gauge, - total_latency_hist, - }, - } - } - - pub fn handle(&self) -> &NodeHandle { - &self.node_handle - } - - fn flush(&mut self, epoch: Epoch) -> Result<(), ExecutionError> { - if let Err(e) = self.sink.flush_batch() { - self.error_manager.report(e); - } - self.ops_since_flush = 0; - self.flush_scheduler_sender - .send(self.max_flush_interval) - .unwrap(); - let _ = self.event_sender.send(Event::SinkFlushed { - node: self.node_handle.clone(), - epoch, - }); - if let Some(source_times) = self.source_times.as_mut() { - let mut labels = self.labels.attrs().clone(); - labels.push(dozer_tracing::KeyValue::new( - TABLE_LABEL, - self.node_handle.id.clone(), - )); - for time in source_times.drain(..) { - if let Some(elapsed) = time.elapsed_millis() { - self.metrics.total_latency_hist.record(elapsed, &labels); - } - } - } - Ok(()) - } -} - -impl Name for SinkNode { - fn name(&self) -> Cow { - Cow::Owned(self.node_handle.to_string()) - } -} - -struct Select<'a> { - op_receivers: &'a [Receiver], - flush_receiver: &'a Receiver<()>, - inner: crossbeam::channel::Select<'a>, - flush_idx: usize, -} - -enum ReceiverMsg { - Op(usize, ExecutorOperation), - Flush, -} - -impl<'a> Select<'a> { - fn new( - op_receivers: &'a [Receiver], - flush_receiver: &'a Receiver<()>, - ) -> Self { - let mut inner = crossbeam::channel::Select::new(); - for recv in op_receivers { - let _ = inner.recv(recv); - } - let flush_idx = inner.recv(flush_receiver); - Self { - inner, - flush_idx, - op_receivers, - flush_receiver, - } - } - - fn remove(&mut self, idx: usize) { - self.inner.remove(idx); - } - - fn reinit(&mut self) { - self.inner = crossbeam::channel::Select::new(); - for recv in self.op_receivers { - let _ = self.inner.recv(recv); - } - self.flush_idx = self.inner.recv(self.flush_receiver); - } - - fn recv(&mut self) -> Result { - let msg = self.inner.select(); - let index = msg.index(); - let res = if index == self.flush_idx { - msg.recv(self.flush_receiver).map(|_| ReceiverMsg::Flush) - } else { - msg.recv(&self.op_receivers[index]) - .map(|op| ReceiverMsg::Op(index, op)) - }; - res.map_err(|_| ExecutionError::CannotReceiveFromChannel) - } -} - -impl ReceiverLoop for SinkNode { - fn initial_epoch_id(&self) -> u64 { - self.initial_epoch_id - } - - fn receivers(&mut self) -> Vec> { - let mut result = vec![]; - swap(&mut self.receivers, &mut result); - result - } - - fn receiver_name(&self, index: usize) -> Cow { - Cow::Owned(self.node_handles[index].to_string()) - } - - fn receiver_loop(mut self, initial_epoch_id: u64) -> Result<(), ExecutionError> { - // This is just copied from ReceiverLoop - let receivers = self.receivers(); - let should_flush_receiver = { - // Take the receiver. This is fine, as long as we exclusively use the - // returned receiver and not the one in `self`. - let (_, mut tmp_recv) = crossbeam::channel::bounded(0); - swap(&mut self.should_flush_receiver, &mut tmp_recv); - tmp_recv - }; - debug_assert!( - !receivers.is_empty(), - "Processor or sink must have at least 1 incoming edge" - ); - let mut is_terminated = vec![false; receivers.len()]; - - let mut commits_received: usize = 0; - let mut epoch_id = initial_epoch_id; - - self.flush_scheduler_sender - .send(self.max_flush_interval) - .unwrap(); - let mut sel = Select::new(&receivers, &should_flush_receiver); - loop { - let ReceiverMsg::Op(index, op) = sel.recv()? else { - if let Some(epoch) = self.last_op_if_commit.take() { - self.flush(epoch)?; - } else { - self.flush_scheduled_on_next_commit = true; - } - continue; - }; - - match op { - ExecutorOperation::Op { op } => { - self.on_op(index, op)?; - } - ExecutorOperation::Commit { epoch } => { - assert_eq!(epoch.common_info.id, epoch_id); - commits_received += 1; - sel.remove(index); - - if commits_received == receivers.len() { - self.on_commit(epoch)?; - epoch_id += 1; - commits_received = 0; - sel.reinit(); - } - } - ExecutorOperation::Terminate => { - is_terminated[index] = true; - sel.remove(index); - debug!( - "[{}] Received Terminate request from {}", - self.name(), - self.receiver_name(index) - ); - if is_terminated.iter().all(|value| *value) { - self.on_terminate()?; - debug!("[{}] Quit", self.name()); - return Ok(()); - } - } - ExecutorOperation::SnapshottingStarted { connection_name } => { - emit_event( - &connection_name, - &ConnectorEntityType::Connector, - &self.labels, - "snapshotting_started", - ); - self.on_snapshotting_started(connection_name)?; - } - ExecutorOperation::SnapshottingDone { - connection_name, - id, - } => { - emit_event( - &connection_name, - &ConnectorEntityType::Connector, - &self.labels, - "snapshotting_done", - ); - self.on_snapshotting_done(connection_name, id)?; - } - } - } - } - - fn on_op(&mut self, _index: usize, op: TableOperation) -> Result<(), ExecutionError> { - self.last_op_if_commit = None; - let mut labels = self.labels.attrs(); - - labels.push(dozer_tracing::KeyValue::new( - TABLE_LABEL, - self.node_handle.id.clone(), - )); - - let op_str = match &op.op { - Operation::Insert { .. } => "insert", - Operation::Delete { .. } => "delete", - Operation::Update { .. } => "update", - Operation::BatchInsert { .. } => "insert", - }; - labels.push(dozer_tracing::KeyValue::new(OPERATION_TYPE_LABEL, op_str)); - let counter_number: u64 = match &op.op { - Operation::BatchInsert { new } => new.len() as u64, - _ => 1, - }; - self.ops_since_flush += counter_number; - - if let Err(e) = self.sink.process(op) { - self.error_manager.report(e); - } - - self.metrics.sink_counter.add(counter_number, &labels); - Ok(()) - } - - fn on_commit(&mut self, epoch: Epoch) -> Result<(), ExecutionError> { - // debug!("[{}] Checkpointing - {}", self.node_handle, epoch); - if let Err(e) = self.sink.commit(&epoch) { - self.error_manager.report(e); - } - self.last_op_if_commit = Some(epoch.clone()); - - match epoch.decision_instant.elapsed() { - Ok(duration) => { - let mut labels = self.labels.attrs().clone(); - labels.push(dozer_tracing::KeyValue::new( - TABLE_LABEL, - self.node_handle.id.clone(), - )); - self.metrics - .latency_gauge - .record(duration.as_secs_f64(), &labels); - } - Err(e) => { - error!("error recording pipeline_latench {:?}", e); - } - } - - if let Some(source_time) = epoch.source_time { - if let Some(source_times) = self.source_times.as_mut() { - source_times.push(source_time); - } else { - let mut labels = self.labels.attrs().clone(); - labels.push(dozer_tracing::KeyValue::new( - TABLE_LABEL, - self.node_handle.id.clone(), - )); - if let Some(elapsed) = source_time.elapsed_millis() { - self.metrics.total_latency_hist.record(elapsed, &labels); - } else { - warn!("Recorded total latency < 0. Source clock and system clock are out of sync."); - } - } - } - - if self - .sink - .preferred_batch_size() - .is_some_and(|batch_size| self.ops_since_flush >= batch_size) - || self.flush_scheduled_on_next_commit - { - self.flush(epoch)?; - self.flush_scheduled_on_next_commit = false; - } - - Ok(()) - } - - fn on_terminate(&mut self) -> Result<(), ExecutionError> { - Ok(()) - } - - fn on_snapshotting_started(&mut self, connection_name: String) -> Result<(), ExecutionError> { - if let Err(e) = self.sink.on_source_snapshotting_started(connection_name) { - self.error_manager.report(e); - } - Ok(()) - } - - fn on_snapshotting_done( - &mut self, - connection_name: String, - id: Option, - ) -> Result<(), ExecutionError> { - if let Err(e) = self.sink.on_source_snapshotting_done(connection_name, id) { - self.error_manager.report(e); - } - - // record 0 as pipeline latency when snapshotting is done - // this is to initialize the value for metrics - let mut labels = self.labels.attrs().clone(); - labels.push(dozer_tracing::KeyValue::new( - TABLE_LABEL, - self.node_handle.id.clone(), - )); - self.metrics.latency_gauge.record(0_f64, &labels); - - Ok(()) - } -} diff --git a/dozer-core/src/executor/source_node/mod.rs b/dozer-core/src/executor/source_node/mod.rs deleted file mode 100644 index 0fa913c004..0000000000 --- a/dozer-core/src/executor/source_node/mod.rs +++ /dev/null @@ -1,235 +0,0 @@ -use std::{fmt::Debug, future::Future, pin::pin, sync::Arc, time::SystemTime}; - -use daggy::petgraph::visit::IntoNodeIdentifiers; -use dozer_types::{ - log::debug, models::ingestion_types::TransactionInfo, node::OpIdentifier, types::TableOperation, -}; -use dozer_types::{models::ingestion_types::IngestionMessage, node::SourceState}; -use futures::{future::Either, StreamExt}; -use tokio::{ - runtime::Runtime, - sync::mpsc::{channel, Receiver, Sender}, -}; - -use crate::{ - builder_dag::NodeKind, - epoch::Epoch, - errors::ExecutionError, - executor_operation::ExecutorOperation, - forwarder::ChannelManager, - node::{PortHandle, Source}, -}; - -use super::{execution_dag::ExecutionDag, node::Node, ExecutorOptions}; - -/// The source operation collector. -#[derive(Debug)] -pub struct SourceNode { - /// To decide when to emit `Commit`, we keep track of source state. - sources: Vec, - /// Structs for running a source. - source_runners: Vec, - /// Receivers from sources. - receivers: Vec>, - /// The current epoch id. - epoch_id: u64, - /// The shutdown future. - shutdown: F, - /// The runtime to run the source in. - runtime: Arc, -} - -impl Node for SourceNode { - fn run(mut self) -> Result<(), ExecutionError> { - let mut handles = vec![]; - for mut source_runner in self.source_runners { - handles.push(Some(self.runtime.spawn(async move { - source_runner - .source - .start(source_runner.sender, source_runner.last_checkpoint) - .await - }))); - } - let mut num_running_sources = handles.len(); - - let mut stream = pin!(stream::receivers_stream(self.receivers)); - loop { - let next = stream.next(); - let next = pin!(next); - match self - .runtime - .block_on(futures::future::select(self.shutdown, next)) - { - Either::Left((_, _)) => { - send_to_all_nodes(&self.sources, ExecutorOperation::Terminate)?; - return Ok(()); - } - Either::Right((next, shutdown)) => { - let next = next.expect("We return just when the stream ends"); - self.shutdown = shutdown; - let index = next.0; - let Some((port, message)) = next.1 else { - debug!("[{}] quit", self.sources[index].channel_manager.owner().id); - match self.runtime.block_on( - handles[index] - .take() - .expect("Shouldn't receive message from dropped receiver"), - ) { - Ok(Ok(())) => { - num_running_sources -= 1; - if num_running_sources == 0 { - send_to_all_nodes(&self.sources, ExecutorOperation::Terminate)?; - return Ok(()); - } - continue; - } - Ok(Err(e)) => return Err(ExecutionError::Source(e)), - Err(e) => { - panic!("Source panicked: {e}"); - } - } - }; - let source = &mut self.sources[index]; - match message { - IngestionMessage::OperationEvent { op, id, .. } => { - source.state = SourceState::NonRestartable; - source - .channel_manager - .send_op(TableOperation { op, id, port })?; - } - IngestionMessage::TransactionInfo(info) => match info { - TransactionInfo::Commit { id, source_time } => { - if let Some(id) = id { - source.state = SourceState::Restartable(id); - } else { - source.state = SourceState::NonRestartable; - } - - let source_states = Arc::new( - self.sources - .iter() - .map(|source| { - ( - source.channel_manager.owner().clone(), - source.state.clone(), - ) - }) - .collect(), - ); - let mut epoch = - Epoch::new(self.epoch_id, source_states, SystemTime::now()); - if let Some(st) = source_time { - epoch = epoch.with_source_time(st); - } - send_to_all_nodes( - &self.sources, - ExecutorOperation::Commit { epoch }, - )?; - self.epoch_id += 1; - } - TransactionInfo::SnapshottingStarted => { - source.channel_manager.send_snapshotting_started( - source.channel_manager.owner().id.clone(), - )?; - } - TransactionInfo::SnapshottingDone { id } => { - source.channel_manager.send_snapshotting_done( - source.channel_manager.owner().id.clone(), - id, - )?; - } - }, - } - } - } - } - } -} - -#[derive(Debug)] -struct RunningSource { - channel_manager: ChannelManager, - state: SourceState, -} - -#[derive(Debug)] -struct SourceRunner { - source: Box, - last_checkpoint: Option, - sender: Sender<(PortHandle, IngestionMessage)>, -} - -/// Returns if the operation is sent successfully. -fn send_to_all_nodes( - sources: &[RunningSource], - op: ExecutorOperation, -) -> Result<(), ExecutionError> { - for source in sources { - source.channel_manager.send_non_op(op.clone())?; - } - Ok(()) -} - -pub async fn create_source_node( - dag: &mut ExecutionDag, - options: &ExecutorOptions, - shutdown: F, - runtime: Arc, -) -> SourceNode { - let mut sources = vec![]; - let mut source_runners = vec![]; - let mut receivers = vec![]; - - let node_indices = dag.graph().node_identifiers().collect::>(); - for node_index in node_indices { - let node = dag.graph()[node_index] - .kind - .as_ref() - .expect("Each node should only be visited once"); - if !matches!(node, NodeKind::Source { .. }) { - continue; - } - let node = dag.node_weight_mut(node_index); - let node_handle = node.handle.clone(); - let NodeKind::Source { - source, - last_checkpoint, - } = node.kind.take().unwrap() - else { - continue; - }; - - let senders = dag.collect_senders(node_index); - let record_writers = dag.collect_record_writers(node_index).await; - let channel_manager = ChannelManager::new( - node_handle, - record_writers, - senders, - dag.error_manager().clone(), - ); - sources.push(RunningSource { - channel_manager, - state: SourceState::NotStarted, - }); - - let (sender, receiver) = channel(options.channel_buffer_sz); - // let (sender, receiver) = channel(1); - source_runners.push(SourceRunner { - source, - last_checkpoint, - sender, - }); - receivers.push(receiver); - } - - SourceNode { - sources, - source_runners, - receivers, - epoch_id: dag.initial_epoch_id(), - shutdown, - runtime, - } -} - -mod stream; diff --git a/dozer-core/src/executor/source_node/stream.rs b/dozer-core/src/executor/source_node/stream.rs deleted file mode 100644 index 5f1a0d73b3..0000000000 --- a/dozer-core/src/executor/source_node/stream.rs +++ /dev/null @@ -1,37 +0,0 @@ -use async_stream::stream; -use futures::{future::select_all, Stream}; -use tokio::sync::mpsc::Receiver; - -/// A convenient way of getting a self-referential struct. -async fn receive_or_drop( - index: usize, - mut receiver: Receiver, -) -> (usize, Option<(Receiver, T)>) { - (index, receiver.recv().await.map(|item| (receiver, item))) -} - -/// This is not simply the merge of `ReceiverStream` because we need to know if the source has quit. -pub fn receivers_stream(receivers: Vec>) -> impl Stream)> { - let mut futures = receivers - .into_iter() - .enumerate() - .map(|(index, receiver)| Box::pin(receive_or_drop(index, receiver))) - .collect::>(); - - stream! { - while !futures.is_empty() { - match select_all(futures).await { - ((index, Some((receiver, item))), _, remaining) => { - yield (index, Some(item)); - futures = remaining; - // Can we somehow remove the allocation here? - futures.push(Box::pin(receive_or_drop(index, receiver))); - } - ((index, None), _, remaining) => { - yield (index, None); - futures = remaining; - } - } - } - } -} diff --git a/dozer-core/src/executor_operation.rs b/dozer-core/src/executor_operation.rs deleted file mode 100644 index 44600b6fef..0000000000 --- a/dozer-core/src/executor_operation.rs +++ /dev/null @@ -1,21 +0,0 @@ -use dozer_types::{node::OpIdentifier, types::TableOperation}; - -use crate::epoch::Epoch; - -#[derive(Clone, Debug)] -pub enum ExecutorOperation { - Op { - op: TableOperation, - }, - Commit { - epoch: Epoch, - }, - Terminate, - SnapshottingStarted { - connection_name: String, - }, - SnapshottingDone { - connection_name: String, - id: Option, - }, -} diff --git a/dozer-core/src/forwarder.rs b/dozer-core/src/forwarder.rs deleted file mode 100644 index df9be44e1e..0000000000 --- a/dozer-core/src/forwarder.rs +++ /dev/null @@ -1,142 +0,0 @@ -use crate::channels::ProcessorChannelForwarder; -use crate::epoch::Epoch; -use crate::error_manager::ErrorManager; -use crate::errors::ExecutionError; -use crate::executor_operation::ExecutorOperation; -use crate::node::PortHandle; -use crate::record_store::RecordWriter; - -use crossbeam::channel::Sender; -use dozer_types::log::debug; -use dozer_types::node::{NodeHandle, OpIdentifier}; -use dozer_types::types::TableOperation; -use std::collections::HashMap; -use std::ops::Deref; -use std::sync::Arc; - -#[derive(Debug)] -pub struct SenderWithPortMapping { - pub sender: Sender, - /// From output port to input port. - pub port_mapping: HashMap>, -} - -impl SenderWithPortMapping { - pub fn send_op(&self, mut op: TableOperation) -> Result<(), ExecutionError> { - let Some(ports) = self.port_mapping.get(&op.port) else { - // Downstream node is not interested in data from this port. - return Ok(()); - }; - - if let Some((last_port, ports)) = ports.split_last() { - for port in ports { - let mut op = op.clone(); - op.port = *port; - self.sender.send(ExecutorOperation::Op { op })?; - } - op.port = *last_port; - self.sender.send(ExecutorOperation::Op { op })?; - } - Ok(()) - } -} - -#[derive(Debug)] -pub struct ChannelManager { - owner: NodeHandle, - record_writers: HashMap>, - senders: Vec, - error_manager: Arc, -} - -impl ChannelManager { - #[inline] - pub fn send_op(&mut self, mut op: TableOperation) -> Result<(), ExecutionError> { - if let Some(writer) = self.record_writers.get_mut(&op.port) { - match writer.write(op.op) { - Ok(new_op) => op.op = new_op, - Err(e) => { - self.error_manager.report(e.into()); - return Ok(()); - } - } - } - - if let Some((last_sender, senders)) = self.senders.split_last() { - for sender in senders { - sender.send_op(op.clone())?; - } - last_sender.send_op(op)?; - } - - Ok(()) - } - - /// Send anything that's not an `ExecutorOperation::Op`. - pub fn send_non_op(&self, op: ExecutorOperation) -> Result<(), ExecutionError> { - assert!(!matches!(op, ExecutorOperation::Op { .. })); - if let Some((last_sender, senders)) = self.senders.split_last() { - for sender in senders { - sender.sender.send(op.clone())?; - } - last_sender.sender.send(op)?; - } - - Ok(()) - } - - pub fn send_terminate(&self) -> Result<(), ExecutionError> { - self.send_non_op(ExecutorOperation::Terminate) - } - - pub fn send_snapshotting_started(&self, connection_name: String) -> Result<(), ExecutionError> { - self.send_non_op(ExecutorOperation::SnapshottingStarted { connection_name }) - } - - pub fn send_snapshotting_done( - &self, - connection_name: String, - id: Option, - ) -> Result<(), ExecutionError> { - self.send_non_op(ExecutorOperation::SnapshottingDone { - connection_name, - id, - }) - } - - pub fn send_commit(&mut self, epoch: Epoch) -> Result<(), ExecutionError> { - debug!( - "[{}] Checkpointing - {}: {:?}", - self.owner, - epoch.common_info.id, - epoch.common_info.source_states.deref() - ); - - self.send_non_op(ExecutorOperation::Commit { epoch }) - } - - pub fn owner(&self) -> &NodeHandle { - &self.owner - } - - pub fn new( - owner: NodeHandle, - record_writers: HashMap>, - senders: Vec, - error_manager: Arc, - ) -> Self { - Self { - owner, - record_writers, - senders, - error_manager, - } - } -} - -impl ProcessorChannelForwarder for ChannelManager { - fn send(&mut self, op: TableOperation) { - self.send_op(op) - .unwrap_or_else(|e| panic!("Failed to send operation: {e}")) - } -} diff --git a/dozer-core/src/hash_map_to_vec.rs b/dozer-core/src/hash_map_to_vec.rs deleted file mode 100644 index cd4af673f5..0000000000 --- a/dozer-core/src/hash_map_to_vec.rs +++ /dev/null @@ -1,18 +0,0 @@ -use std::{ - collections::{hash_map::Entry, HashMap}, - hash::Hash, -}; - -pub fn insert_vec_element(map: &mut HashMap>, key: K, value: V) -where - K: Eq + Hash, -{ - match map.entry(key) { - Entry::Occupied(mut entry) => { - entry.get_mut().push(value); - } - Entry::Vacant(entry) => { - entry.insert(vec![value]); - } - } -} diff --git a/dozer-core/src/lib.rs b/dozer-core/src/lib.rs deleted file mode 100644 index 33f8663d9a..0000000000 --- a/dozer-core/src/lib.rs +++ /dev/null @@ -1,23 +0,0 @@ -pub mod app; -pub mod appsource; -mod builder_dag; -pub mod channels; -mod dag_impl; -pub use dag_impl::*; -pub mod dag_schemas; -mod error_manager; -pub mod errors; -pub mod executor; -pub mod executor_operation; -pub mod forwarder; -mod hash_map_to_vec; -pub mod node; -pub mod record_store; -pub mod shutdown; -pub use tokio; - -#[cfg(test)] -pub mod tests; - -pub use daggy::{self, petgraph}; -pub use dozer_types::{epoch, event}; diff --git a/dozer-core/src/node.rs b/dozer-core/src/node.rs deleted file mode 100644 index d9eb136b10..0000000000 --- a/dozer-core/src/node.rs +++ /dev/null @@ -1,147 +0,0 @@ -use crate::channels::ProcessorChannelForwarder; -use crate::epoch::Epoch; -use crate::event::EventHub; - -use dozer_types::errors::internal::BoxedError; -use dozer_types::models::ingestion_types::IngestionMessage; -use dozer_types::node::OpIdentifier; -use dozer_types::serde::{Deserialize, Serialize}; -use dozer_types::tonic::async_trait; -use dozer_types::types::{Schema, TableOperation}; -use std::collections::HashMap; -use std::fmt::{Debug, Display, Formatter}; -use tokio::sync::mpsc::Sender; - -pub use dozer_types::types::PortHandle; - -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(crate = "dozer_types::serde")] -pub enum OutputPortType { - Stateless, - StatefulWithPrimaryKeyLookup, -} - -impl Display for OutputPortType { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - match self { - OutputPortType::Stateless => f.write_str("Stateless"), - OutputPortType::StatefulWithPrimaryKeyLookup { .. } => { - f.write_str("StatefulWithPrimaryKeyLookup") - } - } - } -} - -#[derive(Debug, Clone)] -pub struct OutputPortDef { - pub handle: PortHandle, - pub typ: OutputPortType, -} - -impl OutputPortDef { - pub fn new(handle: PortHandle, typ: OutputPortType) -> Self { - Self { handle, typ } - } -} - -pub trait SourceFactory: Send + Sync + Debug { - fn get_output_schema(&self, port: &PortHandle) -> Result; - fn get_output_port_name(&self, port: &PortHandle) -> String; - fn get_output_ports(&self) -> Vec; - fn build( - &self, - output_schemas: HashMap, - event_hub: EventHub, - state: Option>, - ) -> Result, BoxedError>; -} - -#[async_trait] -pub trait Source: Send + Sync + Debug { - async fn serialize_state(&self) -> Result, BoxedError>; - - async fn start( - &mut self, - sender: Sender<(PortHandle, IngestionMessage)>, - last_checkpoint: Option, - ) -> Result<(), BoxedError>; -} - -#[async_trait] -pub trait ProcessorFactory: Send + Sync + Debug { - async fn get_output_schema( - &self, - output_port: &PortHandle, - input_schemas: &HashMap, - ) -> Result; - fn get_input_ports(&self) -> Vec; - fn get_output_ports(&self) -> Vec; - async fn build( - &self, - input_schemas: HashMap, - output_schemas: HashMap, - event_hub: EventHub, - ) -> Result, BoxedError>; - fn type_name(&self) -> String; - fn id(&self) -> String; -} - -pub trait Processor: Send + Sync + Debug { - fn commit(&self, epoch_details: &Epoch) -> Result<(), BoxedError>; - fn process( - &mut self, - op: TableOperation, - fw: &mut dyn ProcessorChannelForwarder, - ) -> Result<(), BoxedError>; -} - -#[async_trait] -pub trait SinkFactory: Send + Sync + Debug { - fn get_input_ports(&self) -> Vec; - fn get_input_port_name(&self, port: &PortHandle) -> String; - fn prepare(&self, input_schemas: HashMap) -> Result<(), BoxedError>; - async fn build( - &self, - input_schemas: HashMap, - event_hub: EventHub, - ) -> Result, BoxedError>; - fn type_name(&self) -> String; -} - -pub trait Sink: Send + Debug { - fn commit(&mut self, epoch_details: &Epoch) -> Result<(), BoxedError>; - fn process(&mut self, op: TableOperation) -> Result<(), BoxedError>; - - fn on_source_snapshotting_started(&mut self, connection_name: String) - -> Result<(), BoxedError>; - fn on_source_snapshotting_done( - &mut self, - connection_name: String, - id: Option, - ) -> Result<(), BoxedError>; - - // Pipeline state management. - fn set_source_state(&mut self, source_state: &[u8]) -> Result<(), BoxedError>; - fn get_source_state(&mut self) -> Result>, BoxedError>; - fn get_latest_op_id(&mut self) -> Result, BoxedError>; - - fn preferred_batch_size(&self) -> Option { - None - } - - fn max_batch_duration_ms(&self) -> Option { - None - } - - /// If the Sink batches operations, flush the batch to the store when this method is called. - /// This method is guaranteed to only be called on commit boundaries - fn flush_batch(&mut self) -> Result<(), BoxedError> { - Ok(()) - } - - /// If this returns `true`, [Sink::commit] is assumed to not commit the committed - /// transaction to the sink's remote - fn supports_batching(&self) -> bool { - false - } -} diff --git a/dozer-core/src/record_store.rs b/dozer-core/src/record_store.rs deleted file mode 100644 index cff53fb58d..0000000000 --- a/dozer-core/src/record_store.rs +++ /dev/null @@ -1,87 +0,0 @@ -use dozer_types::errors::types::DeserializationError; -use dozer_types::thiserror::Error; -use dozer_types::types::{Operation, Record, Schema}; -use std::collections::HashMap; -use std::fmt::{Debug, Formatter}; - -#[derive(Debug, Error)] -pub enum RecordWriterError { - #[error("Record not found")] - RecordNotFound, -} - -pub trait RecordWriter: Send + Sync { - fn write(&mut self, op: Operation) -> Result; -} - -impl Debug for dyn RecordWriter { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - f.write_str("RecordWriter") - } -} - -pub fn create_record_writer(schema: Schema) -> Result, DeserializationError> { - let writer = Box::new(PrimaryKeyLookupRecordWriter::new(schema)?); - Ok(writer) -} - -#[derive(Debug)] -pub(crate) struct PrimaryKeyLookupRecordWriter { - schema: Schema, - index: HashMap, Record>, -} - -impl PrimaryKeyLookupRecordWriter { - pub(crate) fn new(schema: Schema) -> Result { - debug_assert!( - !schema.primary_index.is_empty(), - "PrimaryKeyLookupRecordWriter can only be used with a schema that has a primary key." - ); - - Ok(Self { - schema, - index: Default::default(), - }) - } -} - -impl RecordWriter for PrimaryKeyLookupRecordWriter { - fn write(&mut self, op: Operation) -> Result { - match op { - Operation::Insert { new } => { - let new_key = new.get_key(&self.schema.primary_index); - self.index.insert(new_key, new.clone()); - Ok(Operation::Insert { new }) - } - Operation::Delete { mut old } => { - let old_key = old.get_key(&self.schema.primary_index); - old = self - .index - .remove_entry(&old_key) - .ok_or(RecordWriterError::RecordNotFound)? - .1; - Ok(Operation::Delete { old }) - } - Operation::Update { mut old, new } => { - let old_key = old.get_key(&self.schema.primary_index); - old = self - .index - .remove_entry(&old_key) - .ok_or(RecordWriterError::RecordNotFound)? - .1; - let new_key = new.get_key(&self.schema.primary_index); - self.index.insert(new_key, new.clone()); - Ok(Operation::Update { old, new }) - } - Operation::BatchInsert { new } => { - let mut new_records = Vec::with_capacity(new.len()); - for record in new { - let new_key = record.get_key(&self.schema.primary_index); - self.index.insert(new_key, record.clone()); - new_records.push(record); - } - Ok(Operation::BatchInsert { new: new_records }) - } - } - } -} diff --git a/dozer-core/src/shutdown.rs b/dozer-core/src/shutdown.rs deleted file mode 100644 index 1d676ba4b9..0000000000 --- a/dozer-core/src/shutdown.rs +++ /dev/null @@ -1,55 +0,0 @@ -use std::sync::{ - atomic::{AtomicBool, Ordering}, - Arc, -}; - -use futures_util::Future; -use tokio::{ - runtime::Runtime, - sync::watch::{channel, Receiver, Sender}, -}; - -#[derive(Debug)] -pub struct ShutdownSender(Receiver<()>); - -impl ShutdownSender { - pub fn shutdown(self) { - let _ = self.0; - } -} - -#[derive(Debug, Clone)] -pub struct ShutdownReceiver { - sender: Arc>, - running: Arc, -} - -impl ShutdownReceiver { - pub fn get_running_flag(&self) -> Arc { - self.running.clone() - } - - pub fn create_shutdown_future(&self) -> impl Future { - wait_shutdown(self.sender.clone()) - } -} - -async fn wait_shutdown(sender: Arc>) { - sender.closed().await; -} - -pub fn new(runtime: &Runtime) -> (ShutdownSender, ShutdownReceiver) { - let (sender, receiver) = channel(()); - let sender = Arc::new(sender); - let running = Arc::new(AtomicBool::new(true)); - runtime.spawn(wait_and_set_running_flag(sender.clone(), running.clone())); - ( - ShutdownSender(receiver), - ShutdownReceiver { sender, running }, - ) -} - -async fn wait_and_set_running_flag(sender: Arc>, flag: Arc) { - sender.closed().await; - flag.store(false, Ordering::SeqCst); -} diff --git a/dozer-core/src/tests/app.rs b/dozer-core/src/tests/app.rs deleted file mode 100644 index 5c49c7e499..0000000000 --- a/dozer-core/src/tests/app.rs +++ /dev/null @@ -1,299 +0,0 @@ -use super::run_dag; -use crate::app::{App, AppPipeline, PipelineEntryPoint}; -use crate::appsource::{AppSourceManager, AppSourceMappings}; -use crate::event::EventHub; -use crate::node::{OutputPortDef, PortHandle, Source, SourceFactory}; -use crate::tests::dag_base_run::{ - NoopJoinProcessorFactory, NOOP_JOIN_LEFT_INPUT_PORT, NOOP_JOIN_RIGHT_INPUT_PORT, -}; -use crate::tests::sinks::{CountingSinkFactory, COUNTING_SINK_INPUT_PORT}; -use crate::tests::sources::{ - DualPortGeneratorSourceFactory, GeneratorSourceFactory, - DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_1, DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_2, - GENERATOR_SOURCE_OUTPUT_PORT, -}; -use crate::{Edge, Endpoint, DEFAULT_PORT_HANDLE}; -use dozer_types::errors::internal::BoxedError; -use dozer_types::node::NodeHandle; -use dozer_types::types::Schema; - -use std::collections::HashMap; -use std::sync::atomic::AtomicBool; -use std::sync::Arc; - -#[derive(Debug)] -struct NoneSourceFactory {} -impl SourceFactory for NoneSourceFactory { - fn get_output_schema(&self, _port: &PortHandle) -> Result { - todo!() - } - - fn get_output_port_name(&self, _port: &PortHandle) -> String { - todo!() - } - - fn get_output_ports(&self) -> Vec { - todo!() - } - - fn build( - &self, - _output_schemas: HashMap, - _event_hub: EventHub, - _state: Option>, - ) -> Result, BoxedError> { - todo!() - } -} - -#[test] -fn test_apps_source_manager_connection_exists() { - let mut asm = AppSourceManager::new(); - let _r = asm.add( - Box::new(NoneSourceFactory {}), - AppSourceMappings::new( - "conn1".to_string(), - vec![("table1".to_string(), 1_u16)].into_iter().collect(), - ), - ); - let r = asm.add( - Box::new(NoneSourceFactory {}), - AppSourceMappings::new( - "conn1".to_string(), - vec![("table2".to_string(), 1_u16)].into_iter().collect(), - ), - ); - assert!(r.is_err()); -} - -#[test] -fn test_apps_source_manager_lookup() { - let mut asm = AppSourceManager::new(); - asm.add( - Box::new(NoneSourceFactory {}), - AppSourceMappings::new( - "conn1".to_string(), - vec![("table1".to_string(), 1_u16)].into_iter().collect(), - ), - ) - .unwrap(); - - let r = asm.get_endpoint("table1").unwrap(); - assert_eq!(r.node.id, "conn1"); - assert_eq!(r.port, 1_u16); - - let r = asm.get_endpoint("Non-existent source"); - assert!(r.is_err()); - - // Insert another source - asm.add( - Box::new(NoneSourceFactory {}), - AppSourceMappings::new( - "conn2".to_string(), - vec![("table2".to_string(), 2_u16)].into_iter().collect(), - ), - ) - .unwrap(); - - let r = asm.get_endpoint("table3"); - assert!(r.is_err()); - - let r = asm.get_endpoint("table1").unwrap(); - assert_eq!(r.node.id, "conn1"); - assert_eq!(r.port, 1_u16); - - let r = asm.get_endpoint("table2").unwrap(); - assert_eq!(r.node.id, "conn2"); - assert_eq!(r.port, 2_u16); -} - -#[test] -fn test_apps_source_manager_lookup_multiple_ports() { - let mut asm = AppSourceManager::new(); - asm.add( - Box::new(NoneSourceFactory {}), - AppSourceMappings::new( - "conn1".to_string(), - vec![("table1".to_string(), 1_u16), ("table2".to_string(), 2_u16)] - .into_iter() - .collect(), - ), - ) - .unwrap(); - - let r = asm.get_endpoint("table1").unwrap(); - assert_eq!(r.node.id, "conn1"); - assert_eq!(r.port, 1_u16); - - let r = asm.get_endpoint("table2").unwrap(); - assert_eq!(r.node.id, "conn1"); - assert_eq!(r.port, 2_u16); -} - -#[test] -fn test_app_dag() { - let latch = Arc::new(AtomicBool::new(true)); - - let mut asm = AppSourceManager::new(); - asm.add( - Box::new(DualPortGeneratorSourceFactory::new( - 10_000, - latch.clone(), - true, - )), - AppSourceMappings::new( - "postgres".to_string(), - vec![ - ( - "users_postgres".to_string(), - DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_1, - ), - ( - "transactions".to_string(), - DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_2, - ), - ] - .into_iter() - .collect(), - ), - ) - .unwrap(); - - asm.add( - Box::new(GeneratorSourceFactory::new(10_000, latch.clone(), true)), - AppSourceMappings::new( - "snowflake".to_string(), - vec![("users_snowflake".to_string(), GENERATOR_SOURCE_OUTPUT_PORT)] - .into_iter() - .collect(), - ), - ) - .unwrap(); - - let mut app = App::new(asm); - - let mut p1 = AppPipeline::new_with_default_flags(); - p1.add_processor(Box::new(NoopJoinProcessorFactory {}), "join".to_string()); - p1.add_entry_point( - "join".to_string(), - PipelineEntryPoint::new("users_postgres".to_string(), NOOP_JOIN_LEFT_INPUT_PORT), - ); - p1.add_entry_point( - "join".to_string(), - PipelineEntryPoint::new("transactions".to_string(), NOOP_JOIN_RIGHT_INPUT_PORT), - ); - p1.add_sink( - Box::new(CountingSinkFactory::new(20_000, latch.clone())), - "sink".to_string(), - ); - p1.connect_nodes( - "join".to_string(), - DEFAULT_PORT_HANDLE, - "sink".to_string(), - COUNTING_SINK_INPUT_PORT, - ); - - app.add_pipeline(p1); - - let mut p2 = AppPipeline::new_with_default_flags(); - p2.add_processor(Box::new(NoopJoinProcessorFactory {}), "join".to_string()); - p2.add_entry_point( - "join".to_string(), - PipelineEntryPoint::new("users_snowflake".to_string(), NOOP_JOIN_LEFT_INPUT_PORT), - ); - p2.add_entry_point( - "join".to_string(), - PipelineEntryPoint::new("transactions".to_string(), NOOP_JOIN_RIGHT_INPUT_PORT), - ); - p2.add_sink( - Box::new(CountingSinkFactory::new(20_000, latch)), - "sink".to_string(), - ); - p2.connect_nodes( - "join".to_string(), - DEFAULT_PORT_HANDLE, - "sink".to_string(), - COUNTING_SINK_INPUT_PORT, - ); - - app.add_pipeline(p2); - - let dag = app.into_dag().unwrap(); - let edges = dag.edge_handles(); - - assert!(edges.iter().any(|e| *e - == Edge::new( - Endpoint::new( - NodeHandle::new(None, "postgres".to_string()), - DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_1 - ), - Endpoint::new( - NodeHandle::new(Some(1), "join".to_string()), - NOOP_JOIN_LEFT_INPUT_PORT - ) - ))); - - assert!(edges.iter().any(|e| *e - == Edge::new( - Endpoint::new( - NodeHandle::new(None, "postgres".to_string()), - DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_2 - ), - Endpoint::new( - NodeHandle::new(Some(1), "join".to_string()), - NOOP_JOIN_RIGHT_INPUT_PORT - ) - ))); - - assert!(edges.iter().any(|e| *e - == Edge::new( - Endpoint::new( - NodeHandle::new(None, "snowflake".to_string()), - GENERATOR_SOURCE_OUTPUT_PORT - ), - Endpoint::new( - NodeHandle::new(Some(2), "join".to_string()), - NOOP_JOIN_LEFT_INPUT_PORT - ) - ))); - - assert!(edges.iter().any(|e| *e - == Edge::new( - Endpoint::new( - NodeHandle::new(None, "postgres".to_string()), - DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_2 - ), - Endpoint::new( - NodeHandle::new(Some(2), "join".to_string()), - NOOP_JOIN_RIGHT_INPUT_PORT - ) - ))); - - assert!(edges.iter().any(|e| *e - == Edge::new( - Endpoint::new( - NodeHandle::new(Some(1), "join".to_string()), - DEFAULT_PORT_HANDLE - ), - Endpoint::new( - NodeHandle::new(Some(1), "sink".to_string()), - COUNTING_SINK_INPUT_PORT - ) - ))); - - assert!(edges.iter().any(|e| *e - == Edge::new( - Endpoint::new( - NodeHandle::new(Some(2), "join".to_string()), - DEFAULT_PORT_HANDLE - ), - Endpoint::new( - NodeHandle::new(Some(2), "sink".to_string()), - COUNTING_SINK_INPUT_PORT - ) - ))); - - assert_eq!(edges.len(), 6); - - run_dag(dag).unwrap(); -} diff --git a/dozer-core/src/tests/checkpoint_ns.rs b/dozer-core/src/tests/checkpoint_ns.rs deleted file mode 100644 index 14c6bef429..0000000000 --- a/dozer-core/src/tests/checkpoint_ns.rs +++ /dev/null @@ -1,72 +0,0 @@ -use super::run_dag; -use crate::tests::dag_base_run::NoopJoinProcessorFactory; -use crate::tests::sinks::{CountingSinkFactory, COUNTING_SINK_INPUT_PORT}; -use crate::tests::sources::{GeneratorSourceFactory, GENERATOR_SOURCE_OUTPUT_PORT}; -use crate::{Dag, Endpoint, DEFAULT_PORT_HANDLE}; -use dozer_types::node::NodeHandle; -use std::sync::atomic::AtomicBool; -use std::sync::Arc; - -#[test] -fn test_checkpoint_consistency_ns() { - const MESSAGES_COUNT: u64 = 25_000; - - let mut dag = Dag::new(); - - let sources: Vec = vec![ - NodeHandle::new(None, "src1".to_string()), - NodeHandle::new(None, "src2".to_string()), - NodeHandle::new(None, "src3".to_string()), - NodeHandle::new(None, "src4".to_string()), - NodeHandle::new(None, "src5".to_string()), - ]; - - let latch = Arc::new(AtomicBool::new(true)); - - for src_handle in &sources { - dag.add_source( - src_handle.clone(), - Box::new(GeneratorSourceFactory::new( - MESSAGES_COUNT, - latch.clone(), - true, - )), - ); - } - - // Create sources.len()-1 sub dags - for i in 0..sources.len() - 1 { - let mut child_dag = Dag::new(); - - let proc_handle = NodeHandle::new(None, "proc".to_string()); - let sink_handle = NodeHandle::new(None, "sink".to_string()); - - child_dag.add_processor(proc_handle.clone(), Box::new(NoopJoinProcessorFactory {})); - child_dag.add_sink( - sink_handle.clone(), - Box::new(CountingSinkFactory::new(MESSAGES_COUNT * 2, latch.clone())), - ); - child_dag - .connect( - Endpoint::new(proc_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, COUNTING_SINK_INPUT_PORT), - ) - .unwrap(); - - // Merge the DAG with the parent dag - dag.merge(Some(i as u16), child_dag); - - dag.connect( - Endpoint::new(sources[i].clone(), GENERATOR_SOURCE_OUTPUT_PORT), - Endpoint::new(NodeHandle::new(Some(i as u16), "proc".to_string()), 1), - ) - .unwrap(); - dag.connect( - Endpoint::new(sources[i + 1].clone(), GENERATOR_SOURCE_OUTPUT_PORT), - Endpoint::new(NodeHandle::new(Some(i as u16), "proc".to_string()), 2), - ) - .unwrap(); - } - - run_dag(dag).unwrap(); -} diff --git a/dozer-core/src/tests/dag_base_create_errors.rs b/dozer-core/src/tests/dag_base_create_errors.rs deleted file mode 100644 index 45d07a733f..0000000000 --- a/dozer-core/src/tests/dag_base_create_errors.rs +++ /dev/null @@ -1,286 +0,0 @@ -use crate::event::EventHub; -use crate::node::{ - OutputPortDef, OutputPortType, PortHandle, Processor, ProcessorFactory, Source, SourceFactory, -}; -use crate::{Dag, Endpoint, DEFAULT_PORT_HANDLE}; - -use crate::tests::dag_base_run::NoopProcessorFactory; -use crate::tests::sinks::{CountingSinkFactory, COUNTING_SINK_INPUT_PORT}; -use crate::tests::sources::{GeneratorSourceFactory, GENERATOR_SOURCE_OUTPUT_PORT}; - -use dozer_types::errors::internal::BoxedError; -use dozer_types::node::NodeHandle; -use dozer_types::tonic::async_trait; -use dozer_types::types::{FieldDefinition, FieldType, Schema, SourceDefinition}; - -use std::collections::HashMap; -use std::sync::atomic::AtomicBool; -use std::sync::Arc; - -use super::run_dag; - -#[derive(Debug)] -struct CreateErrSourceFactory { - panic: bool, -} - -impl CreateErrSourceFactory { - pub fn new(panic: bool) -> Self { - Self { panic } - } -} - -impl SourceFactory for CreateErrSourceFactory { - fn get_output_schema(&self, _port: &PortHandle) -> Result { - Ok(Schema::default() - .field( - FieldDefinition::new( - "id".to_string(), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - true, - ) - .clone()) - } - - fn get_output_port_name(&self, _port: &PortHandle) -> String { - "error".to_string() - } - - fn get_output_ports(&self) -> Vec { - vec![OutputPortDef::new( - DEFAULT_PORT_HANDLE, - OutputPortType::Stateless, - )] - } - - fn build( - &self, - _output_schemas: HashMap, - _event_hub: EventHub, - _state: Option>, - ) -> Result, BoxedError> { - if self.panic { - panic!("Generated error"); - } else { - Err("Generated Error".to_string().into()) - } - } -} - -#[test] -#[should_panic] -fn test_create_src_err() { - let count: u64 = 1_000_000; - - let mut dag = Dag::new(); - let latch = Arc::new(AtomicBool::new(true)); - - let source_handle = NodeHandle::new(Some(1), 1.to_string()); - let proc_handle = NodeHandle::new(Some(1), 2.to_string()); - let sink_handle = NodeHandle::new(Some(1), 3.to_string()); - - dag.add_source( - source_handle.clone(), - Box::new(CreateErrSourceFactory::new(false)), - ); - dag.add_processor(proc_handle.clone(), Box::new(NoopProcessorFactory {})); - dag.add_sink( - sink_handle.clone(), - Box::new(CountingSinkFactory::new(count, latch)), - ); - - dag.connect( - Endpoint::new(source_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(proc_handle.clone(), DEFAULT_PORT_HANDLE), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, COUNTING_SINK_INPUT_PORT), - ) - .unwrap(); - - run_dag(dag).unwrap(); -} - -#[test] -#[should_panic] -fn test_create_src_panic() { - let count: u64 = 1_000_000; - - let mut dag = Dag::new(); - let latch = Arc::new(AtomicBool::new(true)); - - let source_handle = NodeHandle::new(Some(1), 1.to_string()); - let proc_handle = NodeHandle::new(Some(1), 2.to_string()); - let sink_handle = NodeHandle::new(Some(1), 3.to_string()); - - dag.add_source( - source_handle.clone(), - Box::new(CreateErrSourceFactory::new(true)), - ); - dag.add_processor(proc_handle.clone(), Box::new(NoopProcessorFactory {})); - dag.add_sink( - sink_handle.clone(), - Box::new(CountingSinkFactory::new(count, latch)), - ); - - dag.connect( - Endpoint::new(source_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(proc_handle.clone(), DEFAULT_PORT_HANDLE), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, COUNTING_SINK_INPUT_PORT), - ) - .unwrap(); - - run_dag(dag).unwrap(); -} - -#[derive(Debug)] -struct CreateErrProcessorFactory { - panic: bool, -} - -impl CreateErrProcessorFactory { - pub fn new(panic: bool) -> Self { - Self { panic } - } -} - -#[async_trait] -impl ProcessorFactory for CreateErrProcessorFactory { - fn type_name(&self) -> String { - "CreateErr".to_owned() - } - - async fn get_output_schema( - &self, - _port: &PortHandle, - _input_schemas: &HashMap, - ) -> Result { - Ok(Schema::default() - .field( - FieldDefinition::new( - "id".to_string(), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - true, - ) - .clone()) - } - - fn get_input_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - fn get_output_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - async fn build( - &self, - _input_schemas: HashMap, - _output_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - if self.panic { - panic!("Generated error"); - } else { - Err("Generated Error".to_string().into()) - } - } - - fn id(&self) -> String { - "CreateErr".to_owned() - } -} - -#[test] -#[should_panic] -fn test_create_proc_err() { - let count: u64 = 1_000_000; - - let mut dag = Dag::new(); - let latch = Arc::new(AtomicBool::new(true)); - - let source_handle = NodeHandle::new(Some(1), 1.to_string()); - let proc_handle = NodeHandle::new(Some(1), 2.to_string()); - let sink_handle = NodeHandle::new(Some(1), 3.to_string()); - - dag.add_source( - source_handle.clone(), - Box::new(GeneratorSourceFactory::new(count, latch.clone(), false)), - ); - dag.add_processor( - proc_handle.clone(), - Box::new(CreateErrProcessorFactory::new(false)), - ); - dag.add_sink( - sink_handle.clone(), - Box::new(CountingSinkFactory::new(count, latch)), - ); - - dag.connect( - Endpoint::new(source_handle, GENERATOR_SOURCE_OUTPUT_PORT), - Endpoint::new(proc_handle.clone(), DEFAULT_PORT_HANDLE), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, COUNTING_SINK_INPUT_PORT), - ) - .unwrap(); - - run_dag(dag).unwrap(); -} - -#[test] -#[should_panic] -fn test_create_proc_panic() { - let count: u64 = 1_000_000; - - let mut dag = Dag::new(); - let latch = Arc::new(AtomicBool::new(true)); - - let source_handle = NodeHandle::new(Some(1), 1.to_string()); - let proc_handle = NodeHandle::new(Some(1), 2.to_string()); - let sink_handle = NodeHandle::new(Some(1), 3.to_string()); - - dag.add_source( - source_handle.clone(), - Box::new(GeneratorSourceFactory::new(count, latch.clone(), false)), - ); - dag.add_processor( - proc_handle.clone(), - Box::new(CreateErrProcessorFactory::new(true)), - ); - dag.add_sink( - sink_handle.clone(), - Box::new(CountingSinkFactory::new(count, latch)), - ); - - dag.connect( - Endpoint::new(source_handle, GENERATOR_SOURCE_OUTPUT_PORT), - Endpoint::new(proc_handle.clone(), DEFAULT_PORT_HANDLE), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, COUNTING_SINK_INPUT_PORT), - ) - .unwrap(); - - run_dag(dag).unwrap(); -} diff --git a/dozer-core/src/tests/dag_base_errors.rs b/dozer-core/src/tests/dag_base_errors.rs deleted file mode 100644 index e8a2bed55a..0000000000 --- a/dozer-core/src/tests/dag_base_errors.rs +++ /dev/null @@ -1,575 +0,0 @@ -use crate::channels::ProcessorChannelForwarder; -use crate::epoch::Epoch; -use crate::event::EventHub; -use crate::node::{ - OutputPortDef, OutputPortType, PortHandle, Processor, ProcessorFactory, Sink, SinkFactory, - Source, SourceFactory, -}; -use crate::tests::dag_base_run::NoopProcessorFactory; -use crate::tests::sinks::{CountingSinkFactory, COUNTING_SINK_INPUT_PORT}; -use crate::tests::sources::{GeneratorSourceFactory, GENERATOR_SOURCE_OUTPUT_PORT}; -use crate::{Dag, Endpoint, DEFAULT_PORT_HANDLE}; -use dozer_types::errors::internal::BoxedError; -use dozer_types::models::ingestion_types::{IngestionMessage, TransactionInfo}; -use dozer_types::node::{NodeHandle, OpIdentifier}; -use dozer_types::tonic::async_trait; -use dozer_types::types::{ - Field, FieldDefinition, FieldType, Operation, Record, Schema, SourceDefinition, TableOperation, -}; -use tokio::sync::mpsc::Sender; - -use std::collections::HashMap; -use std::panic; - -use std::sync::atomic::AtomicBool; -use std::sync::Arc; - -use super::run_dag; - -// Test when error is generated by a processor - -#[derive(Debug)] -struct ErrorProcessorFactory { - err_on: u64, - panic: bool, -} - -#[async_trait] -impl ProcessorFactory for ErrorProcessorFactory { - fn type_name(&self) -> String { - "Error".to_owned() - } - - async fn get_output_schema( - &self, - _output_port: &PortHandle, - input_schemas: &HashMap, - ) -> Result { - Ok(input_schemas.get(&DEFAULT_PORT_HANDLE).unwrap().clone()) - } - - fn get_input_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - fn get_output_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - async fn build( - &self, - _input_schemas: HashMap, - _output_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - Ok(Box::new(ErrorProcessor { - err_on: self.err_on, - count: 0, - panic: self.panic, - })) - } - - fn id(&self) -> String { - "Error".to_owned() - } -} - -#[derive(Debug)] -struct ErrorProcessor { - err_on: u64, - count: u64, - panic: bool, -} - -impl Processor for ErrorProcessor { - fn commit(&self, _epoch: &Epoch) -> Result<(), BoxedError> { - Ok(()) - } - - fn process( - &mut self, - mut op: TableOperation, - fw: &mut dyn ProcessorChannelForwarder, - ) -> Result<(), BoxedError> { - self.count += 1; - if self.count == self.err_on { - if self.panic { - panic!("Generated error"); - } else { - return Err("Uknown".to_string().into()); - } - } - - op.port = DEFAULT_PORT_HANDLE; - fw.send(op); - Ok(()) - } -} - -#[test] -#[should_panic] -fn test_run_dag_proc_err_panic() { - let count: u64 = 1_000_000; - - let mut dag = Dag::new(); - let latch = Arc::new(AtomicBool::new(true)); - - let source_handle = NodeHandle::new(None, 1.to_string()); - let proc_handle = NodeHandle::new(Some(1), 1.to_string()); - let sink_handle = NodeHandle::new(Some(1), 2.to_string()); - - dag.add_source( - source_handle.clone(), - Box::new(GeneratorSourceFactory::new(count, latch.clone(), false)), - ); - dag.add_processor( - proc_handle.clone(), - Box::new(ErrorProcessorFactory { - err_on: 800_000, - panic: true, - }), - ); - dag.add_sink( - sink_handle.clone(), - Box::new(CountingSinkFactory::new(count, latch)), - ); - - dag.connect( - Endpoint::new(source_handle, GENERATOR_SOURCE_OUTPUT_PORT), - Endpoint::new(proc_handle.clone(), DEFAULT_PORT_HANDLE), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, COUNTING_SINK_INPUT_PORT), - ) - .unwrap(); - - run_dag(dag).unwrap(); -} - -#[test] -#[should_panic] -fn test_run_dag_proc_err_2() { - let count: u64 = 1_000_000; - - let mut dag = Dag::new(); - let latch = Arc::new(AtomicBool::new(true)); - - let source_handle = NodeHandle::new(None, 1.to_string()); - let proc_handle = NodeHandle::new(Some(1), 1.to_string()); - let proc_err_handle = NodeHandle::new(Some(1), 2.to_string()); - let sink_handle = NodeHandle::new(Some(1), 3.to_string()); - - dag.add_source( - source_handle.clone(), - Box::new(GeneratorSourceFactory::new(count, latch.clone(), false)), - ); - dag.add_processor(proc_handle.clone(), Box::new(NoopProcessorFactory {})); - - dag.add_processor( - proc_err_handle.clone(), - Box::new(ErrorProcessorFactory { - err_on: 800_000, - panic: false, - }), - ); - - dag.add_sink( - sink_handle.clone(), - Box::new(CountingSinkFactory::new(count, latch)), - ); - - dag.connect( - Endpoint::new(source_handle, GENERATOR_SOURCE_OUTPUT_PORT), - Endpoint::new(proc_handle.clone(), DEFAULT_PORT_HANDLE), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(proc_err_handle.clone(), DEFAULT_PORT_HANDLE), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_err_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, COUNTING_SINK_INPUT_PORT), - ) - .unwrap(); - - run_dag(dag).unwrap(); -} - -#[test] -#[should_panic] -fn test_run_dag_proc_err_3() { - let count: u64 = 1_000_000; - - let mut dag = Dag::new(); - let latch = Arc::new(AtomicBool::new(true)); - - let source_handle = NodeHandle::new(None, 1.to_string()); - let proc_handle = NodeHandle::new(Some(1), 1.to_string()); - let proc_err_handle = NodeHandle::new(Some(1), 2.to_string()); - let sink_handle = NodeHandle::new(Some(1), 3.to_string()); - - dag.add_source( - source_handle.clone(), - Box::new(GeneratorSourceFactory::new(count, latch.clone(), false)), - ); - - dag.add_processor( - proc_err_handle.clone(), - Box::new(ErrorProcessorFactory { - err_on: 800_000, - panic: false, - }), - ); - - dag.add_processor(proc_handle.clone(), Box::new(NoopProcessorFactory {})); - - dag.add_sink( - sink_handle.clone(), - Box::new(CountingSinkFactory::new(count, latch)), - ); - - dag.connect( - Endpoint::new(source_handle, GENERATOR_SOURCE_OUTPUT_PORT), - Endpoint::new(proc_err_handle.clone(), DEFAULT_PORT_HANDLE), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_err_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(proc_handle.clone(), DEFAULT_PORT_HANDLE), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, COUNTING_SINK_INPUT_PORT), - ) - .unwrap(); - - run_dag(dag).unwrap(); -} - -// Test when error is generated by a source - -#[derive(Debug)] -pub(crate) struct ErrGeneratorSourceFactory { - count: u64, - err_at: u64, -} - -impl ErrGeneratorSourceFactory { - pub fn new(count: u64, err_at: u64) -> Self { - Self { count, err_at } - } -} - -impl SourceFactory for ErrGeneratorSourceFactory { - fn get_output_schema(&self, _port: &PortHandle) -> Result { - Ok(Schema::default() - .field( - FieldDefinition::new( - "id".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - true, - ) - .field( - FieldDefinition::new( - "value".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone()) - } - - fn get_output_port_name(&self, _port: &PortHandle) -> String { - "error".to_string() - } - - fn get_output_ports(&self) -> Vec { - vec![OutputPortDef::new( - GENERATOR_SOURCE_OUTPUT_PORT, - OutputPortType::Stateless, - )] - } - - fn build( - &self, - _output_schemas: HashMap, - _event_hub: EventHub, - _state: Option>, - ) -> Result, BoxedError> { - Ok(Box::new(ErrGeneratorSource { - count: self.count, - err_at: self.err_at, - })) - } -} - -#[derive(Debug)] -pub(crate) struct ErrGeneratorSource { - count: u64, - err_at: u64, -} - -#[async_trait] -impl Source for ErrGeneratorSource { - async fn serialize_state(&self) -> Result, BoxedError> { - Ok(vec![]) - } - - async fn start( - &mut self, - sender: Sender<(PortHandle, IngestionMessage)>, - _last_checkpoint: Option, - ) -> Result<(), BoxedError> { - for n in 1..(self.count + 1) { - if n == self.err_at { - return Err("Generated Error".to_string().into()); - } - - sender - .send(( - GENERATOR_SOURCE_OUTPUT_PORT, - IngestionMessage::OperationEvent { - table_index: 0, - op: Operation::Insert { - new: Record::new(vec![ - Field::String(format!("key_{n}")), - Field::String(format!("value_{n}")), - ]), - }, - id: Some(OpIdentifier::new(0, n)), - }, - )) - .await?; - sender - .send(( - GENERATOR_SOURCE_OUTPUT_PORT, - IngestionMessage::TransactionInfo(TransactionInfo::Commit { - id: Some(OpIdentifier::new(0, n)), - source_time: None, - }), - )) - .await?; - } - Ok(()) - } -} - -#[test] -#[should_panic] -fn test_run_dag_src_err() { - let count: u64 = 1_000_000; - - let mut dag = Dag::new(); - let latch = Arc::new(AtomicBool::new(true)); - - let source_handle = NodeHandle::new(None, 1.to_string()); - let proc_handle = NodeHandle::new(Some(1), 1.to_string()); - let sink_handle = NodeHandle::new(Some(1), 3.to_string()); - - dag.add_source( - source_handle.clone(), - Box::new(ErrGeneratorSourceFactory::new(count, 200_000)), - ); - dag.add_processor(proc_handle.clone(), Box::new(NoopProcessorFactory {})); - dag.add_sink( - sink_handle.clone(), - Box::new(CountingSinkFactory::new(count, latch)), - ); - - dag.connect( - Endpoint::new(source_handle, GENERATOR_SOURCE_OUTPUT_PORT), - Endpoint::new(proc_handle.clone(), DEFAULT_PORT_HANDLE), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, COUNTING_SINK_INPUT_PORT), - ) - .unwrap(); - - run_dag(dag).unwrap(); -} - -#[derive(Debug)] -pub(crate) struct ErrSinkFactory { - err_at: u64, - panic: bool, -} - -impl ErrSinkFactory { - pub fn new(err_at: u64, panic: bool) -> Self { - Self { err_at, panic } - } -} - -#[async_trait] -impl SinkFactory for ErrSinkFactory { - fn get_input_ports(&self) -> Vec { - vec![COUNTING_SINK_INPUT_PORT] - } - - fn get_input_port_name(&self, _port: &PortHandle) -> String { - "error".to_string() - } - - fn prepare(&self, _input_schemas: HashMap) -> Result<(), BoxedError> { - Ok(()) - } - - async fn build( - &self, - _input_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - Ok(Box::new(ErrSink { - err_at: self.err_at, - current: 0, - panic: self.panic, - })) - } - - fn type_name(&self) -> String { - "error".to_string() - } -} - -#[derive(Debug)] -pub(crate) struct ErrSink { - err_at: u64, - current: u64, - panic: bool, -} -impl Sink for ErrSink { - fn commit(&mut self, _epoch_details: &Epoch) -> Result<(), BoxedError> { - Ok(()) - } - - fn process(&mut self, _op: TableOperation) -> Result<(), BoxedError> { - self.current += 1; - if self.current == self.err_at { - if self.panic { - panic!("Generated error"); - } else { - return Err("Generated error".to_string().into()); - } - } - Ok(()) - } - - fn on_source_snapshotting_started( - &mut self, - _connection_name: String, - ) -> Result<(), BoxedError> { - Ok(()) - } - - fn on_source_snapshotting_done( - &mut self, - _connection_name: String, - _id: Option, - ) -> Result<(), BoxedError> { - Ok(()) - } - - fn set_source_state(&mut self, _source_state: &[u8]) -> Result<(), BoxedError> { - Ok(()) - } - - fn get_source_state(&mut self) -> Result>, BoxedError> { - Ok(None) - } - - fn get_latest_op_id(&mut self) -> Result, BoxedError> { - Ok(None) - } -} - -#[test] -#[should_panic] -fn test_run_dag_sink_err() { - let count: u64 = 1_000_000; - - let mut dag = Dag::new(); - let latch = Arc::new(AtomicBool::new(true)); - - let source_handle = NodeHandle::new(None, 1.to_string()); - let proc_handle = NodeHandle::new(Some(1), 1.to_string()); - let sink_handle = NodeHandle::new(Some(1), 3.to_string()); - - dag.add_source( - source_handle.clone(), - Box::new(GeneratorSourceFactory::new(count, latch, false)), - ); - dag.add_processor(proc_handle.clone(), Box::new(NoopProcessorFactory {})); - dag.add_sink( - sink_handle.clone(), - Box::new(ErrSinkFactory::new(200_000, false)), - ); - - dag.connect( - Endpoint::new(source_handle, GENERATOR_SOURCE_OUTPUT_PORT), - Endpoint::new(proc_handle.clone(), DEFAULT_PORT_HANDLE), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, COUNTING_SINK_INPUT_PORT), - ) - .unwrap(); - - run_dag(dag).unwrap(); -} - -#[test] -#[should_panic] -fn test_run_dag_sink_err_panic() { - let count: u64 = 1_000_000; - - let mut dag = Dag::new(); - let latch = Arc::new(AtomicBool::new(true)); - - let source_handle = NodeHandle::new(None, 1.to_string()); - let proc_handle = NodeHandle::new(Some(1), 1.to_string()); - let sink_handle = NodeHandle::new(Some(1), 3.to_string()); - - dag.add_source( - source_handle.clone(), - Box::new(GeneratorSourceFactory::new(count, latch, false)), - ); - dag.add_processor(proc_handle.clone(), Box::new(NoopProcessorFactory {})); - dag.add_sink( - sink_handle.clone(), - Box::new(ErrSinkFactory::new(200_000, true)), - ); - - dag.connect( - Endpoint::new(source_handle, GENERATOR_SOURCE_OUTPUT_PORT), - Endpoint::new(proc_handle.clone(), DEFAULT_PORT_HANDLE), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, COUNTING_SINK_INPUT_PORT), - ) - .unwrap(); - - run_dag(dag).unwrap(); -} diff --git a/dozer-core/src/tests/dag_base_run.rs b/dozer-core/src/tests/dag_base_run.rs deleted file mode 100644 index e427b0f03b..0000000000 --- a/dozer-core/src/tests/dag_base_run.rs +++ /dev/null @@ -1,376 +0,0 @@ -use crate::channels::ProcessorChannelForwarder; -use crate::epoch::Epoch; -use crate::event::EventHub; -use crate::executor::DagExecutor; -use crate::node::{PortHandle, Processor, ProcessorFactory}; -use crate::tests::sinks::{CountingSinkFactory, COUNTING_SINK_INPUT_PORT}; -use crate::tests::sources::{ - DualPortGeneratorSourceFactory, GeneratorSourceFactory, - DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_1, DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_2, - GENERATOR_SOURCE_OUTPUT_PORT, -}; -use crate::{Dag, Endpoint, DEFAULT_PORT_HANDLE}; -use dozer_types::errors::internal::BoxedError; -use dozer_types::node::NodeHandle; -use dozer_types::tonic::async_trait; -use dozer_types::types::{Schema, TableOperation}; -use tokio::sync::oneshot; - -use std::collections::HashMap; -use std::sync::atomic::AtomicBool; -use std::sync::Arc; -use std::thread; -use std::time::Duration; - -use super::{create_test_runtime, run_dag}; - -#[derive(Debug)] -pub(crate) struct NoopProcessorFactory {} - -#[async_trait] -impl ProcessorFactory for NoopProcessorFactory { - fn type_name(&self) -> String { - "Noop".to_owned() - } - - async fn get_output_schema( - &self, - _output_port: &PortHandle, - input_schemas: &HashMap, - ) -> Result { - Ok(input_schemas.get(&DEFAULT_PORT_HANDLE).unwrap().clone()) - } - - fn get_input_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - fn get_output_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - async fn build( - &self, - _input_schemas: HashMap, - _output_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - Ok(Box::new(NoopProcessor {})) - } - - fn id(&self) -> String { - "Noop".to_owned() - } -} - -#[derive(Debug)] -pub(crate) struct NoopProcessor {} - -impl Processor for NoopProcessor { - fn commit(&self, _epoch_details: &Epoch) -> Result<(), BoxedError> { - Ok(()) - } - - fn process( - &mut self, - mut op: TableOperation, - fw: &mut dyn ProcessorChannelForwarder, - ) -> Result<(), BoxedError> { - op.port = DEFAULT_PORT_HANDLE; - fw.send(op); - Ok(()) - } -} - -#[test] -fn test_run_dag() { - let count: u64 = 1_000; - - let mut dag = Dag::new(); - let latch = Arc::new(AtomicBool::new(true)); - - let source_handle = NodeHandle::new(Some(1), 1.to_string()); - let proc_handle = NodeHandle::new(Some(1), 2.to_string()); - let sink_handle = NodeHandle::new(Some(1), 3.to_string()); - - dag.add_source( - source_handle.clone(), - Box::new(GeneratorSourceFactory::new(count, latch.clone(), false)), - ); - dag.add_processor(proc_handle.clone(), Box::new(NoopProcessorFactory {})); - dag.add_sink( - sink_handle.clone(), - Box::new(CountingSinkFactory::new(count, latch)), - ); - - dag.connect( - Endpoint::new(source_handle, GENERATOR_SOURCE_OUTPUT_PORT), - Endpoint::new(proc_handle.clone(), DEFAULT_PORT_HANDLE), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, COUNTING_SINK_INPUT_PORT), - ) - .unwrap(); - - run_dag(dag).unwrap(); -} - -#[test] -fn test_run_dag_and_stop() { - let count: u64 = 1_000_000; - - let mut dag = Dag::new(); - let latch = Arc::new(AtomicBool::new(true)); - - let source_handle = NodeHandle::new(None, 1.to_string()); - let proc_handle = NodeHandle::new(Some(1), 2.to_string()); - let sink_handle = NodeHandle::new(Some(1), 3.to_string()); - - dag.add_source( - source_handle.clone(), - Box::new(GeneratorSourceFactory::new(count, latch.clone(), false)), - ); - dag.add_processor(proc_handle.clone(), Box::new(NoopProcessorFactory {})); - dag.add_sink( - sink_handle.clone(), - Box::new(CountingSinkFactory::new(count, latch)), - ); - - dag.connect( - Endpoint::new(source_handle, GENERATOR_SOURCE_OUTPUT_PORT), - Endpoint::new(proc_handle.clone(), DEFAULT_PORT_HANDLE), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, COUNTING_SINK_INPUT_PORT), - ) - .unwrap(); - - let runtime = create_test_runtime(); - let runtime_clone = runtime.clone(); - - let (sender, receiver) = oneshot::channel::<()>(); - let join_handle = runtime.block_on(async move { - DagExecutor::new(dag, Default::default()) - .await - .unwrap() - .start(receiver, Default::default(), runtime_clone) - .await - .unwrap() - }); - - thread::sleep(Duration::from_millis(1000)); - sender.send(()).unwrap(); - join_handle.join().unwrap(); -} - -#[derive(Debug)] -pub(crate) struct NoopJoinProcessorFactory {} - -pub const NOOP_JOIN_LEFT_INPUT_PORT: u16 = 1; -pub const NOOP_JOIN_RIGHT_INPUT_PORT: u16 = 2; - -#[async_trait] -impl ProcessorFactory for NoopJoinProcessorFactory { - fn type_name(&self) -> String { - "NoopJoin".to_owned() - } - - async fn get_output_schema( - &self, - _output_port: &PortHandle, - input_schemas: &HashMap, - ) -> Result { - Ok(input_schemas.get(&1).unwrap().clone()) - } - - fn get_input_ports(&self) -> Vec { - vec![1, 2] - } - - fn get_output_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - async fn build( - &self, - _input_schemas: HashMap, - _output_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - Ok(Box::new(NoopJoinProcessor {})) - } - - fn id(&self) -> String { - "NoopJoin".to_owned() - } -} - -#[derive(Debug)] -pub(crate) struct NoopJoinProcessor {} - -impl Processor for NoopJoinProcessor { - fn commit(&self, _epoch_details: &Epoch) -> Result<(), BoxedError> { - Ok(()) - } - - fn process( - &mut self, - mut op: TableOperation, - fw: &mut dyn ProcessorChannelForwarder, - ) -> Result<(), BoxedError> { - op.port = DEFAULT_PORT_HANDLE; - fw.send(op); - Ok(()) - } -} - -#[test] -fn test_run_dag_2_sources_stateless() { - let count: u64 = 50_000; - - let mut dag = Dag::new(); - let latch = Arc::new(AtomicBool::new(true)); - - let source1_handle = NodeHandle::new(None, 1.to_string()); - let source2_handle = NodeHandle::new(None, 2.to_string()); - - let proc_handle = NodeHandle::new(Some(1), 1.to_string()); - let sink_handle = NodeHandle::new(Some(1), 2.to_string()); - - dag.add_source( - source1_handle.clone(), - Box::new(GeneratorSourceFactory::new(count, latch.clone(), false)), - ); - dag.add_source( - source2_handle.clone(), - Box::new(GeneratorSourceFactory::new(count, latch.clone(), false)), - ); - dag.add_processor(proc_handle.clone(), Box::new(NoopJoinProcessorFactory {})); - dag.add_sink( - sink_handle.clone(), - Box::new(CountingSinkFactory::new(count * 2, latch)), - ); - - dag.connect( - Endpoint::new(source1_handle, GENERATOR_SOURCE_OUTPUT_PORT), - Endpoint::new(proc_handle.clone(), 1), - ) - .unwrap(); - - dag.connect( - Endpoint::new(source2_handle, GENERATOR_SOURCE_OUTPUT_PORT), - Endpoint::new(proc_handle.clone(), 2), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, COUNTING_SINK_INPUT_PORT), - ) - .unwrap(); - - run_dag(dag).unwrap(); -} - -#[test] -fn test_run_dag_2_sources_stateful() { - let count: u64 = 50_000; - - let mut dag = Dag::new(); - let latch = Arc::new(AtomicBool::new(true)); - - let source1_handle = NodeHandle::new(None, 1.to_string()); - let source2_handle = NodeHandle::new(None, 2.to_string()); - - let proc_handle = NodeHandle::new(Some(1), 1.to_string()); - let sink_handle = NodeHandle::new(Some(1), 2.to_string()); - - dag.add_source( - source1_handle.clone(), - Box::new(GeneratorSourceFactory::new(count, latch.clone(), true)), - ); - dag.add_source( - source2_handle.clone(), - Box::new(GeneratorSourceFactory::new(count, latch.clone(), true)), - ); - dag.add_processor(proc_handle.clone(), Box::new(NoopJoinProcessorFactory {})); - dag.add_sink( - sink_handle.clone(), - Box::new(CountingSinkFactory::new(count * 2, latch)), - ); - - dag.connect( - Endpoint::new(source1_handle, GENERATOR_SOURCE_OUTPUT_PORT), - Endpoint::new(proc_handle.clone(), 1), - ) - .unwrap(); - - dag.connect( - Endpoint::new(source2_handle, GENERATOR_SOURCE_OUTPUT_PORT), - Endpoint::new(proc_handle.clone(), 2), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, COUNTING_SINK_INPUT_PORT), - ) - .unwrap(); - - run_dag(dag).unwrap(); -} - -#[test] -fn test_run_dag_1_source_2_ports_stateless() { - let count: u64 = 50_000; - - let mut dag = Dag::new(); - let latch = Arc::new(AtomicBool::new(true)); - - let source_handle = NodeHandle::new(None, 1.to_string()); - let proc_handle = NodeHandle::new(Some(1), 1.to_string()); - let sink_handle = NodeHandle::new(Some(1), 2.to_string()); - - dag.add_source( - source_handle.clone(), - Box::new(DualPortGeneratorSourceFactory::new( - count, - latch.clone(), - false, - )), - ); - dag.add_processor(proc_handle.clone(), Box::new(NoopJoinProcessorFactory {})); - dag.add_sink( - sink_handle.clone(), - Box::new(CountingSinkFactory::new(count * 2, latch)), - ); - - dag.connect( - Endpoint::new( - source_handle.clone(), - DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_1, - ), - Endpoint::new(proc_handle.clone(), 1), - ) - .unwrap(); - - dag.connect( - Endpoint::new(source_handle, DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_2), - Endpoint::new(proc_handle.clone(), 2), - ) - .unwrap(); - - dag.connect( - Endpoint::new(proc_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, COUNTING_SINK_INPUT_PORT), - ) - .unwrap(); - - run_dag(dag).unwrap(); -} diff --git a/dozer-core/src/tests/dag_ports.rs b/dozer-core/src/tests/dag_ports.rs deleted file mode 100644 index 9222e458d3..0000000000 --- a/dozer-core/src/tests/dag_ports.rs +++ /dev/null @@ -1,159 +0,0 @@ -use crate::event::EventHub; -use crate::node::{ - OutputPortDef, OutputPortType, PortHandle, Processor, ProcessorFactory, Source, SourceFactory, -}; -use crate::{Dag, Endpoint, DEFAULT_PORT_HANDLE}; -use dozer_types::errors::internal::BoxedError; -use dozer_types::tonic::async_trait; -use dozer_types::{node::NodeHandle, types::Schema}; -use std::collections::HashMap; - -#[derive(Debug)] -pub struct DynPortsSourceFactory { - output_ports: Vec, -} - -impl DynPortsSourceFactory { - pub fn new(output_ports: Vec) -> Self { - Self { output_ports } - } -} - -impl SourceFactory for DynPortsSourceFactory { - fn get_output_schema(&self, _port: &PortHandle) -> Result { - todo!() - } - - fn get_output_port_name(&self, _port: &PortHandle) -> String { - todo!() - } - - fn get_output_ports(&self) -> Vec { - self.output_ports - .iter() - .map(|p| OutputPortDef::new(*p, OutputPortType::Stateless)) - .collect() - } - - fn build( - &self, - _input_schemas: HashMap, - _event_hub: EventHub, - _state: Option>, - ) -> Result, BoxedError> { - todo!() - } -} - -#[derive(Debug)] -pub struct DynPortsProcessorFactory { - input_ports: Vec, - output_ports: Vec, -} - -impl DynPortsProcessorFactory { - pub fn new(input_ports: Vec, output_ports: Vec) -> Self { - Self { - input_ports, - output_ports, - } - } -} - -#[async_trait] -impl ProcessorFactory for DynPortsProcessorFactory { - async fn get_output_schema( - &self, - _output_port: &PortHandle, - _input_schemas: &HashMap, - ) -> Result { - todo!() - } - - fn get_input_ports(&self) -> Vec { - self.input_ports.clone() - } - - fn get_output_ports(&self) -> Vec { - self.output_ports.clone() - } - - async fn build( - &self, - _input_schemas: HashMap, - _output_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - todo!() - } - - fn type_name(&self) -> String { - "DynPorts".to_owned() - } - - fn id(&self) -> String { - "DynPorts".to_owned() - } -} - -macro_rules! test_ports { - ($id:ident, $out_ports:expr, $in_ports:expr, $from_port:expr, $to_port:expr, $expect:expr) => { - #[test] - fn $id() { - let src = DynPortsSourceFactory::new($out_ports); - let proc = DynPortsProcessorFactory::new($in_ports, vec![DEFAULT_PORT_HANDLE]); - - let source_handle = NodeHandle::new(None, 1.to_string()); - let proc_handle = NodeHandle::new(Some(1), 1.to_string()); - - let mut dag = Dag::new(); - - dag.add_source(source_handle.clone(), Box::new(src)); - dag.add_processor(proc_handle.clone(), Box::new(proc)); - - let res = dag.connect( - Endpoint::new(source_handle, $from_port), - Endpoint::new(proc_handle, $to_port), - ); - - assert!(res.is_ok() == $expect) - } - }; -} - -test_ports!( - test_none_ports, - vec![DEFAULT_PORT_HANDLE], - vec![DEFAULT_PORT_HANDLE], - DEFAULT_PORT_HANDLE, - DEFAULT_PORT_HANDLE, - true -); - -test_ports!(test_matching_ports, vec![1], vec![2], 1, 2, true); -test_ports!(test_not_matching_ports, vec![2], vec![1], 1, 2, false); -test_ports!( - test_not_default_port, - vec![2], - vec![1], - DEFAULT_PORT_HANDLE, - 2, - false -); - -test_ports!( - test_not_default_port2, - vec![DEFAULT_PORT_HANDLE], - vec![1], - 1, - 2, - false -); -test_ports!( - test_not_default_port3, - vec![DEFAULT_PORT_HANDLE], - vec![DEFAULT_PORT_HANDLE], - DEFAULT_PORT_HANDLE, - 2, - false -); diff --git a/dozer-core/src/tests/dag_schemas.rs b/dozer-core/src/tests/dag_schemas.rs deleted file mode 100644 index 99555036ef..0000000000 --- a/dozer-core/src/tests/dag_schemas.rs +++ /dev/null @@ -1,261 +0,0 @@ -use crate::dag_schemas::{DagHaveSchemas, DagSchemas}; -use crate::event::EventHub; -use crate::node::{ - OutputPortDef, OutputPortType, PortHandle, Processor, ProcessorFactory, SinkFactory, Source, - SourceFactory, -}; -use crate::{Dag, Endpoint, DEFAULT_PORT_HANDLE}; - -use dozer_types::errors::internal::BoxedError; -use dozer_types::node::NodeHandle; -use dozer_types::tonic::async_trait; -use dozer_types::types::{FieldDefinition, FieldType, Schema, SourceDefinition}; -use std::collections::HashMap; - -#[derive(Debug)] -struct TestUsersSourceFactory {} - -impl SourceFactory for TestUsersSourceFactory { - fn get_output_schema(&self, _port: &PortHandle) -> Result { - Ok(Schema::default() - .field( - FieldDefinition::new( - "user_id".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - true, - ) - .field( - FieldDefinition::new( - "username".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - true, - ) - .field( - FieldDefinition::new( - "country_id".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - true, - ) - .clone()) - } - - fn get_output_port_name(&self, _port: &PortHandle) -> String { - "users".to_string() - } - - fn get_output_ports(&self) -> Vec { - vec![OutputPortDef::new( - DEFAULT_PORT_HANDLE, - OutputPortType::Stateless, - )] - } - - fn build( - &self, - _input_schemas: HashMap, - _event_hub: EventHub, - _state: Option>, - ) -> Result, BoxedError> { - todo!() - } -} - -#[derive(Debug)] -struct TestCountriesSourceFactory {} - -impl SourceFactory for TestCountriesSourceFactory { - fn get_output_schema(&self, _port: &PortHandle) -> Result { - Ok(Schema::default() - .field( - FieldDefinition::new( - "country_id".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - true, - ) - .field( - FieldDefinition::new( - "country_name".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - true, - ) - .clone()) - } - - fn get_output_port_name(&self, _port: &PortHandle) -> String { - "countries".to_string() - } - - fn get_output_ports(&self) -> Vec { - vec![OutputPortDef::new( - DEFAULT_PORT_HANDLE, - OutputPortType::Stateless, - )] - } - - fn build( - &self, - _input_schemas: HashMap, - _event_hub: EventHub, - _state: Option>, - ) -> Result, BoxedError> { - todo!() - } -} - -#[derive(Debug)] -struct TestJoinProcessorFactory {} - -#[async_trait] -impl ProcessorFactory for TestJoinProcessorFactory { - async fn get_output_schema( - &self, - _output_port: &PortHandle, - input_schemas: &HashMap, - ) -> Result { - let mut joined: Vec = Vec::new(); - joined.extend(input_schemas.get(&1).unwrap().fields.clone()); - joined.extend(input_schemas.get(&2).unwrap().fields.clone()); - Ok(Schema { - fields: joined, - primary_index: vec![], - }) - } - - fn get_input_ports(&self) -> Vec { - vec![1, 2] - } - - fn get_output_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - async fn build( - &self, - _input_schemas: HashMap, - _output_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - todo!() - } - - fn type_name(&self) -> String { - "TestJoin".to_owned() - } - - fn id(&self) -> String { - "TestJoin".to_owned() - } -} - -#[derive(Debug)] -struct TestSinkFactory {} - -#[async_trait] -impl SinkFactory for TestSinkFactory { - fn get_input_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - fn get_input_port_name(&self, _port: &PortHandle) -> String { - "test".to_owned() - } - - fn prepare(&self, _input_schemas: HashMap) -> Result<(), BoxedError> { - Ok(()) - } - - async fn build( - &self, - _input_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - todo!() - } - - fn type_name(&self) -> String { - "test".to_owned() - } -} - -#[tokio::test] -async fn test_extract_dag_schemas() { - let mut dag = Dag::new(); - - let users_handle = NodeHandle::new(Some(1), 1.to_string()); - let countries_handle = NodeHandle::new(Some(1), 2.to_string()); - let join_handle = NodeHandle::new(Some(1), 3.to_string()); - let sink_handle = NodeHandle::new(Some(1), 4.to_string()); - - let users_index = dag.add_source(users_handle.clone(), Box::new(TestUsersSourceFactory {})); - let countries_index = dag.add_source( - countries_handle.clone(), - Box::new(TestCountriesSourceFactory {}), - ); - let join_index = dag.add_processor(join_handle.clone(), Box::new(TestJoinProcessorFactory {})); - let sink_index = dag.add_sink(sink_handle.clone(), Box::new(TestSinkFactory {})); - - dag.connect( - Endpoint::new(users_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(join_handle.clone(), 1), - ) - .unwrap(); - dag.connect( - Endpoint::new(countries_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(join_handle.clone(), 2), - ) - .unwrap(); - dag.connect( - Endpoint::new(join_handle, DEFAULT_PORT_HANDLE), - Endpoint::new(sink_handle, DEFAULT_PORT_HANDLE), - ) - .unwrap(); - - let dag_schemas = DagSchemas::new(dag).await.unwrap(); - - let users_output = dag_schemas.get_node_output_schemas(users_index); - assert_eq!( - users_output.get(&DEFAULT_PORT_HANDLE).unwrap().fields.len(), - 3 - ); - - let countries_output = dag_schemas.get_node_output_schemas(countries_index); - assert_eq!( - countries_output - .get(&DEFAULT_PORT_HANDLE) - .unwrap() - .fields - .len(), - 2 - ); - - let join_input = dag_schemas.get_node_input_schemas(join_index); - assert_eq!(join_input.get(&1).unwrap().fields.len(), 3); - assert_eq!(join_input.get(&2).unwrap().fields.len(), 2); - - let join_output = dag_schemas.get_node_output_schemas(join_index); - assert_eq!( - join_output.get(&DEFAULT_PORT_HANDLE).unwrap().fields.len(), - 5 - ); - - let sink_input = dag_schemas.get_node_input_schemas(sink_index); - assert_eq!( - sink_input.get(&DEFAULT_PORT_HANDLE).unwrap().fields.len(), - 5 - ); -} diff --git a/dozer-core/src/tests/mod.rs b/dozer-core/src/tests/mod.rs deleted file mode 100644 index 64c78620ab..0000000000 --- a/dozer-core/src/tests/mod.rs +++ /dev/null @@ -1,38 +0,0 @@ -use std::sync::Arc; - -use futures::future::pending; -use tokio::runtime::{self, Runtime}; - -use crate::{errors::ExecutionError, executor::DagExecutor, Dag}; - -mod app; -mod checkpoint_ns; -mod dag_base_create_errors; -mod dag_base_errors; -mod dag_base_run; -mod dag_ports; -mod dag_schemas; -pub mod processors; -pub mod sinks; -pub mod sources; - -fn create_test_runtime() -> Arc { - Arc::new( - runtime::Builder::new_current_thread() - .enable_all() - .build() - .unwrap(), - ) -} - -fn run_dag(dag: Dag) -> Result<(), ExecutionError> { - let runtime = create_test_runtime(); - let runtime_clone = runtime.clone(); - let handle = runtime.block_on(async move { - DagExecutor::new(dag, Default::default()) - .await? - .start(pending::<()>(), Default::default(), runtime_clone) - .await - })?; - handle.join() -} diff --git a/dozer-core/src/tests/processors.rs b/dozer-core/src/tests/processors.rs deleted file mode 100644 index 57f41f8c14..0000000000 --- a/dozer-core/src/tests/processors.rs +++ /dev/null @@ -1,94 +0,0 @@ -use std::collections::HashMap; - -use dozer_types::{errors::internal::BoxedError, tonic::async_trait, types::Schema}; - -use crate::{ - event::EventHub, - node::{PortHandle, Processor, ProcessorFactory}, - DEFAULT_PORT_HANDLE, -}; - -#[derive(Debug)] -pub struct ConnectivityTestProcessorFactory; - -#[async_trait] -impl ProcessorFactory for ConnectivityTestProcessorFactory { - fn type_name(&self) -> String { - "ConnectivityTest".to_owned() - } - fn get_input_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - fn get_output_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - async fn get_output_schema( - &self, - _output_port: &PortHandle, - _input_schemas: &HashMap, - ) -> Result { - unimplemented!( - "This struct is for connectivity test, only input and output ports are defined" - ) - } - - async fn build( - &self, - _input_schemas: HashMap, - _output_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - unimplemented!( - "This struct is for connectivity test, only input and output ports are defined" - ) - } - - fn id(&self) -> String { - "ConnectivityTest".to_owned() - } -} - -#[derive(Debug)] -pub struct NoInputPortProcessorFactory; - -#[async_trait] -impl ProcessorFactory for NoInputPortProcessorFactory { - fn get_input_ports(&self) -> Vec { - vec![] - } - - fn get_output_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - async fn get_output_schema( - &self, - _output_port: &PortHandle, - _input_schemas: &HashMap, - ) -> Result { - unimplemented!( - "This struct is for connectivity test, only input and output ports are defined" - ) - } - - async fn build( - &self, - _input_schemas: HashMap, - _output_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - unimplemented!( - "This struct is for connectivity test, only input and output ports are defined" - ) - } - - fn type_name(&self) -> String { - "NoInput".to_owned() - } - - fn id(&self) -> String { - "NoInput".to_owned() - } -} diff --git a/dozer-core/src/tests/sinks.rs b/dozer-core/src/tests/sinks.rs deleted file mode 100644 index 6f67a10216..0000000000 --- a/dozer-core/src/tests/sinks.rs +++ /dev/null @@ -1,173 +0,0 @@ -use crate::epoch::Epoch; -use crate::event::EventHub; -use crate::node::{PortHandle, Sink, SinkFactory}; -use crate::DEFAULT_PORT_HANDLE; -use dozer_types::errors::internal::BoxedError; -use dozer_types::node::OpIdentifier; -use dozer_types::types::{Schema, TableOperation}; - -use dozer_types::log::debug; -use std::collections::HashMap; - -use dozer_types::tonic::async_trait; -use std::sync::atomic::{AtomicBool, Ordering}; -use std::sync::Arc; - -pub(crate) const COUNTING_SINK_INPUT_PORT: PortHandle = 90; - -#[derive(Debug)] -pub(crate) struct CountingSinkFactory { - expected: u64, - running: Arc, -} - -impl CountingSinkFactory { - pub fn new(expected: u64, barrier: Arc) -> Self { - Self { - expected, - running: barrier, - } - } -} - -#[async_trait] -impl SinkFactory for CountingSinkFactory { - fn get_input_ports(&self) -> Vec { - vec![COUNTING_SINK_INPUT_PORT] - } - - fn get_input_port_name(&self, _port: &PortHandle) -> String { - "counting".to_string() - } - - fn prepare(&self, _input_schemas: HashMap) -> Result<(), BoxedError> { - Ok(()) - } - - async fn build( - &self, - _input_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - Ok(Box::new(CountingSink { - expected: self.expected, - current: 0, - running: self.running.clone(), - })) - } - - fn type_name(&self) -> String { - "counting".to_string() - } -} - -#[derive(Debug)] -pub(crate) struct CountingSink { - expected: u64, - current: u64, - running: Arc, -} -impl Sink for CountingSink { - fn commit(&mut self, _epoch_details: &Epoch) -> Result<(), BoxedError> { - Ok(()) - } - - fn process(&mut self, _op: TableOperation) -> Result<(), BoxedError> { - self.current += 1; - if self.current == self.expected { - debug!( - "Received {} messages. Notifying sender to exit!", - self.current - ); - self.running.store(false, Ordering::Relaxed); - } - Ok(()) - } - - fn on_source_snapshotting_started( - &mut self, - _connection_name: String, - ) -> Result<(), BoxedError> { - Ok(()) - } - - fn on_source_snapshotting_done( - &mut self, - _connection_name: String, - _id: Option, - ) -> Result<(), BoxedError> { - Ok(()) - } - - fn set_source_state(&mut self, _source_state: &[u8]) -> Result<(), BoxedError> { - Ok(()) - } - - fn get_source_state(&mut self) -> Result>, BoxedError> { - Ok(None) - } - - fn get_latest_op_id(&mut self) -> Result, BoxedError> { - Ok(None) - } -} - -#[derive(Debug)] -pub struct ConnectivityTestSinkFactory; - -#[async_trait] -impl SinkFactory for ConnectivityTestSinkFactory { - fn get_input_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - fn get_input_port_name(&self, _port: &PortHandle) -> String { - "test".to_string() - } - - fn prepare(&self, _input_schemas: HashMap) -> Result<(), BoxedError> { - unimplemented!("This struct is for connectivity test, only input ports are defined") - } - - async fn build( - &self, - _input_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - unimplemented!("This struct is for connectivity test, only input ports are defined") - } - - fn type_name(&self) -> String { - "connectivity_test".to_string() - } -} - -#[derive(Debug)] -pub struct NoInputPortSinkFactory; - -#[async_trait] -impl SinkFactory for NoInputPortSinkFactory { - fn get_input_ports(&self) -> Vec { - vec![] - } - - fn get_input_port_name(&self, _port: &PortHandle) -> String { - panic!("No input ports defined") - } - - fn prepare(&self, _input_schemas: HashMap) -> Result<(), BoxedError> { - unimplemented!("This struct is for connectivity test, only input ports are defined") - } - - async fn build( - &self, - _input_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - unimplemented!("This struct is for connectivity test, only input ports are defined") - } - - fn type_name(&self) -> String { - "no_input_port".to_string() - } -} diff --git a/dozer-core/src/tests/sources.rs b/dozer-core/src/tests/sources.rs deleted file mode 100644 index 896b1cf782..0000000000 --- a/dozer-core/src/tests/sources.rs +++ /dev/null @@ -1,329 +0,0 @@ -use crate::event::EventHub; -use crate::node::{OutputPortDef, OutputPortType, PortHandle, Source, SourceFactory}; -use crate::DEFAULT_PORT_HANDLE; -use dozer_types::errors::internal::BoxedError; -use dozer_types::models::ingestion_types::{IngestionMessage, TransactionInfo}; -use dozer_types::node::OpIdentifier; -use dozer_types::tonic::async_trait; -use dozer_types::types::{ - Field, FieldDefinition, FieldType, Operation, Record, Schema, SourceDefinition, -}; -use tokio::{self, sync::mpsc::Sender}; - -use std::collections::HashMap; -use std::sync::atomic::{AtomicBool, Ordering}; -use std::sync::Arc; - -use std::time::Duration; - -pub(crate) const GENERATOR_SOURCE_OUTPUT_PORT: PortHandle = 100; - -#[derive(Debug)] -pub(crate) struct GeneratorSourceFactory { - count: u64, - running: Arc, - stateful: bool, -} - -impl GeneratorSourceFactory { - pub fn new(count: u64, barrier: Arc, stateful: bool) -> Self { - Self { - count, - running: barrier, - stateful, - } - } -} - -impl SourceFactory for GeneratorSourceFactory { - fn get_output_schema(&self, _port: &PortHandle) -> Result { - Ok(Schema::default() - .field( - FieldDefinition::new( - "id".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - true, - ) - .field( - FieldDefinition::new( - "value".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone()) - } - - fn get_output_port_name(&self, _port: &PortHandle) -> String { - "generator".to_string() - } - - fn get_output_ports(&self) -> Vec { - vec![OutputPortDef::new( - GENERATOR_SOURCE_OUTPUT_PORT, - if self.stateful { - OutputPortType::StatefulWithPrimaryKeyLookup - } else { - OutputPortType::Stateless - }, - )] - } - - fn build( - &self, - _input_schemas: HashMap, - _event_hub: EventHub, - _state: Option>, - ) -> Result, BoxedError> { - Ok(Box::new(GeneratorSource { - count: self.count, - running: self.running.clone(), - })) - } -} - -#[derive(Debug)] -pub(crate) struct GeneratorSource { - count: u64, - running: Arc, -} - -#[async_trait] -impl Source for GeneratorSource { - async fn serialize_state(&self) -> Result, BoxedError> { - Ok(vec![]) - } - - async fn start( - &mut self, - sender: Sender<(PortHandle, IngestionMessage)>, - last_checkpoint: Option, - ) -> Result<(), BoxedError> { - let start = last_checkpoint - .map(|checkpoint| checkpoint.seq_in_tx + 1) - .unwrap_or(0); - for n in start..(start + self.count) { - sender - .send(( - GENERATOR_SOURCE_OUTPUT_PORT, - IngestionMessage::OperationEvent { - table_index: 0, - op: Operation::Insert { - new: Record::new(vec![ - Field::String(format!("key_{n}")), - Field::String(format!("value_{n}")), - ]), - }, - id: Some(OpIdentifier::new(0, n)), - }, - )) - .await?; - sender - .send(( - GENERATOR_SOURCE_OUTPUT_PORT, - IngestionMessage::TransactionInfo(TransactionInfo::Commit { - id: Some(OpIdentifier::new(0, n)), - source_time: None, - }), - )) - .await?; - } - - loop { - if !self.running.load(Ordering::Relaxed) { - break; - } - tokio::time::sleep(Duration::from_millis(500)).await; - } - - Ok(()) - } -} - -pub(crate) const DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_1: PortHandle = 1000; -pub(crate) const DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_2: PortHandle = 2000; - -#[derive(Debug)] -pub(crate) struct DualPortGeneratorSourceFactory { - count: u64, - running: Arc, - stateful: bool, -} - -impl DualPortGeneratorSourceFactory { - pub fn new(count: u64, barrier: Arc, stateful: bool) -> Self { - Self { - count, - running: barrier, - stateful, - } - } -} - -impl SourceFactory for DualPortGeneratorSourceFactory { - fn get_output_schema(&self, _port: &PortHandle) -> Result { - Ok(Schema::default() - .field( - FieldDefinition::new( - "id".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - true, - ) - .field( - FieldDefinition::new( - "value".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone()) - } - - fn get_output_port_name(&self, port: &PortHandle) -> String { - match *port { - DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_1 => "generator1".to_string(), - DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_2 => "generator2".to_string(), - _ => panic!("Unknown port"), - } - } - - fn get_output_ports(&self) -> Vec { - vec![ - OutputPortDef::new( - DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_1, - if self.stateful { - OutputPortType::StatefulWithPrimaryKeyLookup - } else { - OutputPortType::Stateless - }, - ), - OutputPortDef::new( - DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_2, - if self.stateful { - OutputPortType::StatefulWithPrimaryKeyLookup - } else { - OutputPortType::Stateless - }, - ), - ] - } - - fn build( - &self, - _input_schemas: HashMap, - _event_hub: EventHub, - _state: Option>, - ) -> Result, BoxedError> { - Ok(Box::new(DualPortGeneratorSource { - count: self.count, - running: self.running.clone(), - })) - } -} - -#[derive(Debug)] -pub(crate) struct DualPortGeneratorSource { - count: u64, - running: Arc, -} - -#[async_trait] -impl Source for DualPortGeneratorSource { - async fn serialize_state(&self) -> Result, BoxedError> { - Ok(vec![]) - } - - async fn start( - &mut self, - sender: Sender<(PortHandle, IngestionMessage)>, - _last_checkpoint: Option, - ) -> Result<(), BoxedError> { - for n in 1..(self.count + 1) { - sender - .send(( - DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_1, - IngestionMessage::OperationEvent { - table_index: 0, - op: Operation::Insert { - new: Record::new(vec![ - Field::String(format!("key_{n}")), - Field::String(format!("value_{n}")), - ]), - }, - id: Some(OpIdentifier::new(0, n)), - }, - )) - .await?; - sender - .send(( - DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_2, - IngestionMessage::OperationEvent { - table_index: 0, - op: Operation::Insert { - new: Record::new(vec![ - Field::String(format!("key_{n}")), - Field::String(format!("value_{n}")), - ]), - }, - id: Some(OpIdentifier::new(0, n)), - }, - )) - .await?; - sender - .send(( - DUAL_PORT_GENERATOR_SOURCE_OUTPUT_PORT_1, - IngestionMessage::TransactionInfo(TransactionInfo::Commit { - id: Some(OpIdentifier::new(0, n)), - source_time: None, - }), - )) - .await?; - } - loop { - if !self.running.load(Ordering::Relaxed) { - break; - } - tokio::time::sleep(Duration::from_millis(500)).await; - } - Ok(()) - } -} - -#[derive(Debug)] -pub struct ConnectivityTestSourceFactory; - -impl SourceFactory for ConnectivityTestSourceFactory { - fn get_output_schema(&self, _port: &PortHandle) -> Result { - unimplemented!("This struct is for connectivity test, only output ports are defined") - } - - fn get_output_port_name(&self, _port: &PortHandle) -> String { - unimplemented!("This struct is for connectivity test, only output ports are defined") - } - - fn get_output_ports(&self) -> Vec { - vec![OutputPortDef::new( - DEFAULT_PORT_HANDLE, - OutputPortType::Stateless, - )] - } - - fn build( - &self, - _output_schemas: HashMap, - _event_hub: EventHub, - _state: Option>, - ) -> Result, BoxedError> { - unimplemented!("This struct is for connectivity test, only output ports are defined") - } -} diff --git a/dozer-deno/Cargo.toml b/dozer-deno/Cargo.toml deleted file mode 100644 index cd0aa018ef..0000000000 --- a/dozer-deno/Cargo.toml +++ /dev/null @@ -1,32 +0,0 @@ -[package] -name = "dozer-deno" -version = "0.4.0" -edition = "2021" -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-types = { path = "../dozer-types" } -tokio = "1.33.0" -deno_cache_dir = "0.7.1" -encoding_rs = "0.8.33" -once_cell = "1.18.0" -tempfile = "3.10.1" - -deno_ast = { version = "1.0", features = ["transpiling"] } -deno_core = { workspace = true } -deno_permissions = "0.2" -deno_terminal = "0.1.1" -deno_broadcast_channel = "0.136" -deno_cache = "0.74" -deno_console = "0.142" -deno_crypto = "0.156" -deno_fetch = "0.166" -deno_napi = "0.72" -deno_tls = "0.129" -deno_url = "0.142" -deno_web = "0.173" -deno_webidl = "0.142" -deno_websocket = "0.147" -deno_webstorage = "0.137" diff --git a/dozer-deno/js/06_util.js b/dozer-deno/js/06_util.js deleted file mode 100644 index bf71c371b9..0000000000 --- a/dozer-deno/js/06_util.js +++ /dev/null @@ -1,36 +0,0 @@ -// Copyright 2018-2024 the Deno authors. All rights reserved. MIT license. - -import { primordials } from "ext:core/mod.js"; -import { op_bootstrap_log_level } from "ext:core/ops"; -const { SafeArrayIterator } = primordials; - -// WARNING: Keep this in sync with Rust (search for LogLevel) -const LogLevel = { - Error: 1, - Warn: 2, - Info: 3, - Debug: 4, -}; - -const logSource = "JS"; - -let logLevel_ = null; -function logLevel() { - if (logLevel_ === null) { - logLevel_ = op_bootstrap_log_level() || 3; - } - return logLevel_; -} - -function log(...args) { - if (logLevel() >= LogLevel.Debug) { - // if we destructure `console` off `globalThis` too early, we don't bind to - // the right console, therefore we don't log anything out. - globalThis.console.error( - `DEBUG ${logSource} -`, - ...new SafeArrayIterator(args), - ); - } -} - -export { log }; diff --git a/dozer-deno/js/98_global_scope.js b/dozer-deno/js/98_global_scope.js deleted file mode 100644 index 1e8be0070e..0000000000 --- a/dozer-deno/js/98_global_scope.js +++ /dev/null @@ -1,288 +0,0 @@ -// Copyright 2018-2023 the Deno authors. All rights reserved. MIT license. - -const core = globalThis.Deno.core; -const primordials = globalThis.__bootstrap.primordials; -const { - ObjectDefineProperties, - SymbolFor, -} = primordials; - -import * as location from "ext:deno_web/12_location.js"; -import * as event from "ext:deno_web/02_event.js"; -import * as timers from "ext:deno_web/02_timers.js"; -import * as base64 from "ext:deno_web/05_base64.js"; -import * as encoding from "ext:deno_web/08_text_encoding.js"; -import * as console from "ext:deno_console/01_console.js"; -import * as caches from "ext:deno_cache/01_cache.js"; -import * as compression from "ext:deno_web/14_compression.js"; -import * as performance from "ext:deno_web/15_performance.js"; -import * as crypto from "ext:deno_crypto/00_crypto.js"; -import * as url from "ext:deno_url/00_url.js"; -import * as urlPattern from "ext:deno_url/01_urlpattern.js"; -import * as headers from "ext:deno_fetch/20_headers.js"; -import * as streams from "ext:deno_web/06_streams.js"; -import * as fileReader from "ext:deno_web/10_filereader.js"; -import * as webSocket from "ext:deno_websocket/01_websocket.js"; -import * as webSocketStream from "ext:deno_websocket/02_websocketstream.js"; -import * as broadcastChannel from "ext:deno_broadcast_channel/01_broadcast_channel.js"; -import * as file from "ext:deno_web/09_file.js"; -import * as formData from "ext:deno_fetch/21_formdata.js"; -import * as request from "ext:deno_fetch/23_request.js"; -import * as response from "ext:deno_fetch/23_response.js"; -import * as fetch from "ext:deno_fetch/26_fetch.js"; -import * as eventSource from "ext:deno_fetch/27_eventsource.js"; -import * as messagePort from "ext:deno_web/13_message_port.js"; -import * as webidl from "ext:deno_webidl/00_webidl.js"; -import { DOMException } from "ext:deno_web/01_dom_exception.js"; -import * as abortSignal from "ext:deno_web/03_abort_signal.js"; -import * as imageData from "ext:deno_web/16_image_data.js"; -import * as globalInterfaces from "ext:deno_web/04_global_interfaces.js"; -import * as webStorage from "ext:deno_webstorage/01_webstorage.js"; - -// https://developer.mozilla.org/en-US/docs/Web/API/WindowOrWorkerGlobalScope -const windowOrWorkerGlobalScope = { - AbortController: core.propNonEnumerable(abortSignal.AbortController), - AbortSignal: core.propNonEnumerable(abortSignal.AbortSignal), - Blob: core.propNonEnumerable(file.Blob), - ByteLengthQueuingStrategy: core.propNonEnumerable( - streams.ByteLengthQueuingStrategy, - ), - CloseEvent: core.propNonEnumerable(event.CloseEvent), - CompressionStream: core.propNonEnumerable(compression.CompressionStream), - CountQueuingStrategy: core.propNonEnumerable( - streams.CountQueuingStrategy, - ), - CryptoKey: core.propNonEnumerable(crypto.CryptoKey), - CustomEvent: core.propNonEnumerable(event.CustomEvent), - DecompressionStream: core.propNonEnumerable(compression.DecompressionStream), - DOMException: core.propNonEnumerable(DOMException), - ErrorEvent: core.propNonEnumerable(event.ErrorEvent), - Event: core.propNonEnumerable(event.Event), - EventTarget: core.propNonEnumerable(event.EventTarget), - File: core.propNonEnumerable(file.File), - FileReader: core.propNonEnumerable(fileReader.FileReader), - FormData: core.propNonEnumerable(formData.FormData), - Headers: core.propNonEnumerable(headers.Headers), - ImageData: core.propNonEnumerable(imageData.ImageData), - MessageEvent: core.propNonEnumerable(event.MessageEvent), - Performance: core.propNonEnumerable(performance.Performance), - PerformanceEntry: core.propNonEnumerable(performance.PerformanceEntry), - PerformanceMark: core.propNonEnumerable(performance.PerformanceMark), - PerformanceMeasure: core.propNonEnumerable(performance.PerformanceMeasure), - PromiseRejectionEvent: core.propNonEnumerable(event.PromiseRejectionEvent), - ProgressEvent: core.propNonEnumerable(event.ProgressEvent), - ReadableStream: core.propNonEnumerable(streams.ReadableStream), - ReadableStreamDefaultReader: core.propNonEnumerable( - streams.ReadableStreamDefaultReader, - ), - Request: core.propNonEnumerable(request.Request), - Response: core.propNonEnumerable(response.Response), - TextDecoder: core.propNonEnumerable(encoding.TextDecoder), - TextEncoder: core.propNonEnumerable(encoding.TextEncoder), - TextDecoderStream: core.propNonEnumerable(encoding.TextDecoderStream), - TextEncoderStream: core.propNonEnumerable(encoding.TextEncoderStream), - TransformStream: core.propNonEnumerable(streams.TransformStream), - URL: core.propNonEnumerable(url.URL), - URLPattern: core.propNonEnumerable(urlPattern.URLPattern), - URLSearchParams: core.propNonEnumerable(url.URLSearchParams), - WebSocket: core.propNonEnumerable(webSocket.WebSocket), - MessageChannel: core.propNonEnumerable(messagePort.MessageChannel), - MessagePort: core.propNonEnumerable(messagePort.MessagePort), - WritableStream: core.propNonEnumerable(streams.WritableStream), - WritableStreamDefaultWriter: core.propNonEnumerable( - streams.WritableStreamDefaultWriter, - ), - WritableStreamDefaultController: core.propNonEnumerable( - streams.WritableStreamDefaultController, - ), - ReadableByteStreamController: core.propNonEnumerable( - streams.ReadableByteStreamController, - ), - ReadableStreamBYOBReader: core.propNonEnumerable( - streams.ReadableStreamBYOBReader, - ), - ReadableStreamBYOBRequest: core.propNonEnumerable( - streams.ReadableStreamBYOBRequest, - ), - ReadableStreamDefaultController: core.propNonEnumerable( - streams.ReadableStreamDefaultController, - ), - TransformStreamDefaultController: core.propNonEnumerable( - streams.TransformStreamDefaultController, - ), - atob: core.propWritable(base64.atob), - btoa: core.propWritable(base64.btoa), - clearInterval: core.propWritable(timers.clearInterval), - clearTimeout: core.propWritable(timers.clearTimeout), - caches: { - enumerable: true, - configurable: true, - get: caches.cacheStorage, - }, - CacheStorage: core.propNonEnumerable(caches.CacheStorage), - Cache: core.propNonEnumerable(caches.Cache), - console: core.propNonEnumerable( - new console.Console((msg, level) => core.print(msg, level > 1)), - ), - crypto: core.propReadOnly(crypto.crypto), - Crypto: core.propNonEnumerable(crypto.Crypto), - SubtleCrypto: core.propNonEnumerable(crypto.SubtleCrypto), - fetch: core.propWritable(fetch.fetch), - EventSource: core.propWritable(eventSource.EventSource), - performance: core.propWritable(performance.performance), - reportError: core.propWritable(event.reportError), - setInterval: core.propWritable(timers.setInterval), - setTimeout: core.propWritable(timers.setTimeout), - structuredClone: core.propWritable(messagePort.structuredClone), - // Branding as a WebIDL object - [webidl.brand]: core.propNonEnumerable(webidl.brand), -}; - -const unstableWindowOrWorkerGlobalScope = { - BroadcastChannel: core.propNonEnumerable(broadcastChannel.BroadcastChannel), - WebSocketStream: core.propNonEnumerable(webSocketStream.WebSocketStream), -}; - -class Navigator { - constructor() { - webidl.illegalConstructor(); - } - - [SymbolFor("Deno.privateCustomInspect")](inspect) { - return `${this.constructor.name} ${inspect({})}`; - } -} - -const navigator = webidl.createBranded(Navigator); - -let numCpus, userAgent, language; - -function setNumCpus(val) { - numCpus = val; -} - -function setUserAgent(val) { - userAgent = val; -} - -function setLanguage(val) { - language = val; -} - -ObjectDefineProperties(Navigator.prototype, { - hardwareConcurrency: { - configurable: true, - enumerable: true, - get() { - webidl.assertBranded(this, NavigatorPrototype); - return numCpus; - }, - }, - userAgent: { - configurable: true, - enumerable: true, - get() { - webidl.assertBranded(this, NavigatorPrototype); - return userAgent; - }, - }, - language: { - configurable: true, - enumerable: true, - get() { - webidl.assertBranded(this, NavigatorPrototype); - return language; - }, - }, - languages: { - configurable: true, - enumerable: true, - get() { - webidl.assertBranded(this, NavigatorPrototype); - return [language]; - }, - }, -}); -const NavigatorPrototype = Navigator.prototype; - -class WorkerNavigator { - constructor() { - webidl.illegalConstructor(); - } - - [SymbolFor("Deno.privateCustomInspect")](inspect) { - return `${this.constructor.name} ${inspect({})}`; - } -} - -const workerNavigator = webidl.createBranded(WorkerNavigator); - -ObjectDefineProperties(WorkerNavigator.prototype, { - hardwareConcurrency: { - configurable: true, - enumerable: true, - get() { - webidl.assertBranded(this, WorkerNavigatorPrototype); - return numCpus; - }, - }, - userAgent: { - configurable: true, - enumerable: true, - get() { - webidl.assertBranded(this, WorkerNavigatorPrototype); - return userAgent; - }, - }, - language: { - configurable: true, - enumerable: true, - get() { - webidl.assertBranded(this, WorkerNavigatorPrototype); - return language; - }, - }, - languages: { - configurable: true, - enumerable: true, - get() { - webidl.assertBranded(this, WorkerNavigatorPrototype); - return [language]; - }, - }, -}); -const WorkerNavigatorPrototype = WorkerNavigator.prototype; - -const mainRuntimeGlobalProperties = { - Location: location.locationConstructorDescriptor, - location: location.locationDescriptor, - Window: globalInterfaces.windowConstructorDescriptor, - window: core.propGetterOnly(() => globalThis), - self: core.propGetterOnly(() => globalThis), - Navigator: core.propNonEnumerable(Navigator), - navigator: core.propGetterOnly(() => navigator), - localStorage: core.propGetterOnly(webStorage.localStorage), - sessionStorage: core.propGetterOnly(webStorage.sessionStorage), - Storage: core.propNonEnumerable(webStorage.Storage), -}; - -const workerRuntimeGlobalProperties = { - WorkerLocation: location.workerLocationConstructorDescriptor, - location: location.workerLocationDescriptor, - WorkerGlobalScope: globalInterfaces.workerGlobalScopeConstructorDescriptor, - DedicatedWorkerGlobalScope: - globalInterfaces.dedicatedWorkerGlobalScopeConstructorDescriptor, - WorkerNavigator: core.propNonEnumerable(WorkerNavigator), - navigator: core.propGetterOnly(() => workerNavigator), - self: core.propGetterOnly(() => globalThis), -}; - -export { - mainRuntimeGlobalProperties, - setLanguage, - setNumCpus, - setUserAgent, - unstableWindowOrWorkerGlobalScope, - windowOrWorkerGlobalScope, - workerRuntimeGlobalProperties, -}; diff --git a/dozer-deno/js/99_main.js b/dozer-deno/js/99_main.js deleted file mode 100644 index edd6a2d8de..0000000000 --- a/dozer-deno/js/99_main.js +++ /dev/null @@ -1,181 +0,0 @@ -const core = globalThis.Deno.core; -const ops = core.ops; -const internals = globalThis.__bootstrap.internals; -const primordials = globalThis.__bootstrap.primordials; -const { - ArrayPrototypeShift, - DateNow, - ErrorPrototype, - ObjectDefineProperties, - ObjectPrototypeIsPrototypeOf, - ObjectSetPrototypeOf, - WeakMapPrototypeDelete, - WeakMapPrototypeGet, -} = primordials; -import * as event from "ext:deno_web/02_event.js"; -import * as timers from "ext:deno_web/02_timers.js"; -import { - getDefaultInspectOptions, - getNoColor, - inspectArgs, - quoteString, -} from "ext:deno_console/01_console.js"; -import * as performance from "ext:deno_web/15_performance.js"; -import * as fetch from "ext:deno_fetch/26_fetch.js"; -import { - mainRuntimeGlobalProperties, - windowOrWorkerGlobalScope, -} from "ext:runtime/98_global_scope.js"; - -let globalThis_; - -function formatException(error) { - if (ObjectPrototypeIsPrototypeOf(ErrorPrototype, error)) { - return null; - } else if (typeof error == "string") { - return `Uncaught ${inspectArgs([quoteString(error, getDefaultInspectOptions())], { - colors: !getNoColor(), - }) - }`; - } else { - return `Uncaught ${inspectArgs([error], { colors: !getNoColor() })}`; - } -} - -function runtimeStart() { - //core.setMacrotaskCallback(timers.handleTimerMacrotask); - //core.setMacrotaskCallback(promiseRejectMacrotaskCallback); - core.setWasmStreamingCallback(fetch.handleWasmStreaming); - core.setReportExceptionCallback(event.reportException); - ops.op_set_format_exception_callback(formatException); -} - -core.setUnhandledPromiseRejectionHandler(processUnhandledPromiseRejection); -core.setHandledPromiseRejectionHandler(processRejectionHandled); - -// Notification that the core received an unhandled promise rejection that is about to -// terminate the runtime. If we can handle it, attempt to do so. -function processUnhandledPromiseRejection(promise, reason) { - const rejectionEvent = new event.PromiseRejectionEvent( - "unhandledrejection", - { - cancelable: true, - promise, - reason, - }, - ); - - // Note that the handler may throw, causing a recursive "error" event - globalThis_.dispatchEvent(rejectionEvent); - - // If event was not yet prevented, try handing it off to Node compat layer - // (if it was initialized) - if ( - !rejectionEvent.defaultPrevented && - typeof internals.nodeProcessUnhandledRejectionCallback !== "undefined" - ) { - internals.nodeProcessUnhandledRejectionCallback(rejectionEvent); - } - - // If event was not prevented (or "unhandledrejection" listeners didn't - // throw) we will let Rust side handle it. - if (rejectionEvent.defaultPrevented) { - return true; - } - - return false; -} - -function processRejectionHandled(promise, reason) { - const rejectionHandledEvent = new event.PromiseRejectionEvent( - "rejectionhandled", - { promise, reason }, - ); - - // Note that the handler may throw, causing a recursive "error" event - globalThis_.dispatchEvent(rejectionHandledEvent); - - if (typeof internals.nodeProcessRejectionHandledCallback !== "undefined") { - internals.nodeProcessRejectionHandledCallback(rejectionHandledEvent); - } -} - -function promiseRejectMacrotaskCallback() { - // We have no work to do, tell the runtime that we don't - // need to perform microtask checkpoint. - if (pendingRejections.length === 0) { - return undefined; - } - - while (pendingRejections.length > 0) { - const promise = ArrayPrototypeShift(pendingRejections); - const hasPendingException = ops.op_has_pending_promise_rejection( - promise, - ); - const reason = WeakMapPrototypeGet(pendingRejectionsReasons, promise); - WeakMapPrototypeDelete(pendingRejectionsReasons, promise); - - if (!hasPendingException) { - continue; - } - - const rejectionEvent = new event.PromiseRejectionEvent( - "unhandledrejection", - { - cancelable: true, - promise, - reason, - }, - ); - - const errorEventCb = (event) => { - if (event.error === reason) { - ops.op_remove_pending_promise_rejection(promise); - } - }; - // Add a callback for "error" event - it will be dispatched - // if error is thrown during dispatch of "unhandledrejection" - // event. - globalThis_.addEventListener("error", errorEventCb); - globalThis_.dispatchEvent(rejectionEvent); - globalThis_.removeEventListener("error", errorEventCb); - - // If event was not yet prevented, try handing it off to Node compat layer - // (if it was initialized) - if ( - !rejectionEvent.defaultPrevented && - typeof internals.nodeProcessUnhandledRejectionCallback !== "undefined" - ) { - internals.nodeProcessUnhandledRejectionCallback(rejectionEvent); - } - - // If event was not prevented (or "unhandledrejection" listeners didn't - // throw) we will let Rust side handle it. - if (rejectionEvent.defaultPrevented) { - ops.op_remove_pending_promise_rejection(promise); - } - } - return true; -} - -delete globalThis.console; - -ObjectDefineProperties(globalThis, windowOrWorkerGlobalScope); - -ObjectDefineProperties(globalThis, mainRuntimeGlobalProperties); - -performance.setTimeOrigin(DateNow()); -globalThis_ = globalThis; - -ObjectSetPrototypeOf(globalThis, Window.prototype); - -event.setEventTargetData(globalThis); -event.saveGlobalThisReference(globalThis); - -event.defineEventHandler(globalThis, "error"); -event.defineEventHandler(globalThis, "load"); -event.defineEventHandler(globalThis, "beforeunload"); -event.defineEventHandler(globalThis, "unload"); -event.defineEventHandler(globalThis, "unhandledrejection"); - -runtimeStart(); diff --git a/dozer-deno/js/README.md b/dozer-deno/js/README.md deleted file mode 100644 index 4055e6e0f5..0000000000 --- a/dozer-deno/js/README.md +++ /dev/null @@ -1,7 +0,0 @@ -# Bootstrap development notes - -Files in this directory are adapted from `deno_runtime/js/`. The best way to view the original files is to compile this project and jump to`deno_runtime` crate source. - -To expose additional APIs to the JavaScript runtime, `dozer-deno/src/runtime/js_runtime.rs` and these files should be updated together. - -When updating to a new `deno_runtime` version, these files should be updated accordingly. The functions in `dozer-deno/src/runtime/js_runtime.rs` should also be revised to keep in sync with original code in `deno_runtime`. diff --git a/dozer-deno/src/lib.rs b/dozer-deno/src/lib.rs deleted file mode 100644 index e4bee9f349..0000000000 --- a/dozer-deno/src/lib.rs +++ /dev/null @@ -1,10 +0,0 @@ -mod ts_module_loader; -pub use ts_module_loader::TypescriptModuleLoader; - -mod runtime; -pub use runtime::{Error as RuntimeError, JsWorker, Runtime}; - -fn user_agent() -> String { - let version: String = env!("CARGO_PKG_VERSION").into(); - format!("Dozer/{} {}", version, deno_core::v8_version()) -} diff --git a/dozer-deno/src/runtime/conversion.rs b/dozer-deno/src/runtime/conversion.rs deleted file mode 100644 index 0e9b6770b9..0000000000 --- a/dozer-deno/src/runtime/conversion.rs +++ /dev/null @@ -1,84 +0,0 @@ -use std::ops::Deref; - -use deno_core::{ - anyhow::{bail, Context as _}, - error::AnyError, -}; -use deno_napi::v8::{self, HandleScope, Local}; -use dozer_types::json_types::{DestructuredJson, JsonObject, JsonValue}; - -pub fn to_v8<'s>( - scope: &mut HandleScope<'s>, - value: JsonValue, -) -> Result, AnyError> { - match value.destructure() { - DestructuredJson::Null => Ok(v8::null(scope).into()), - DestructuredJson::Bool(value) => Ok(v8::Boolean::new(scope, value).into()), - DestructuredJson::Number(value) => { - let value = value.to_f64().context("number is not a f64")?; - Ok(v8::Number::new(scope, value).into()) - } - DestructuredJson::String(value) => Ok(v8::String::new(scope, &value) - .context(format!("failed to create string {}", value.deref()))? - .into()), - DestructuredJson::Array(values) => { - let array = v8::Array::new(scope, values.len() as i32); - for (index, value) in values.into_iter().enumerate() { - let value = to_v8(scope, value)?; - array.set_index(scope, index as u32, value); - } - Ok(array.into()) - } - DestructuredJson::Object(map) => { - let object = v8::Object::new(scope); - for (key, value) in map.into_iter() { - let key = v8::String::new(scope, &key) - .context(format!("failed to create key {}", key.deref()))?; - let value = to_v8(scope, value)?; - object.set(scope, key.into(), value); - } - Ok(object.into()) - } - } -} - -pub fn from_v8<'s>( - scope: &mut HandleScope<'s>, - value: Local<'s, v8::Value>, -) -> Result { - if value.is_null_or_undefined() { - Ok(JsonValue::NULL) - } else if value.is_boolean() { - Ok(value.boolean_value(scope).into()) - } else if value.is_number() { - Ok(value - .number_value(scope) - .context("number is not a f64")? - .into()) - } else if let Ok(value) = TryInto::>::try_into(value) { - Ok(value.to_rust_string_lossy(scope).into()) - } else if let Ok(value) = TryInto::>::try_into(value) { - let mut values = Vec::new(); - for index in 0..value.length() { - let value = value.get_index(scope, index).unwrap(); - let value = from_v8(scope, value)?; - values.push(value); - } - Ok(values.into()) - } else if let Ok(value) = TryInto::>::try_into(value) { - let mut map = JsonObject::new(); - let Some(keys) = value.get_own_property_names(scope, Default::default()) else { - return Ok(map.into()); - }; - for index in 0..keys.length() { - let key = keys.get_index(scope, index).unwrap(); - let value = value.get(scope, key).unwrap(); - let key = key.to_rust_string_lossy(scope); - let value = from_v8(scope, value)?; - map.insert(key, value); - } - Ok(map.into()) - } else { - bail!("cannot convert v8 value to JSON because its type is not supported") - } -} diff --git a/dozer-deno/src/runtime/exception.js b/dozer-deno/src/runtime/exception.js deleted file mode 100644 index 8f150c0f19..0000000000 --- a/dozer-deno/src/runtime/exception.js +++ /dev/null @@ -1,3 +0,0 @@ -export default function () { - throw new Error("exception from javascript"); -} diff --git a/dozer-deno/src/runtime/fetch.js b/dozer-deno/src/runtime/fetch.js deleted file mode 100644 index 8c500a69d5..0000000000 --- a/dozer-deno/src/runtime/fetch.js +++ /dev/null @@ -1,5 +0,0 @@ -export default async function () { - const response = await fetch('https://api.github.com/repos/getdozer/dozer/commits?per_page=1'); - const json = await response.json(); - return json; -} diff --git a/dozer-deno/src/runtime/fetch_exception.js b/dozer-deno/src/runtime/fetch_exception.js deleted file mode 100644 index 9166926550..0000000000 --- a/dozer-deno/src/runtime/fetch_exception.js +++ /dev/null @@ -1,5 +0,0 @@ -export default async function () { - const response = await fetch("https://github.com/getdozer/dozer"); - const json = await response.json(); - return json; -} diff --git a/dozer-deno/src/runtime/js_runtime.rs b/dozer-deno/src/runtime/js_runtime.rs deleted file mode 100644 index 70229f18be..0000000000 --- a/dozer-deno/src/runtime/js_runtime.rs +++ /dev/null @@ -1,144 +0,0 @@ -use std::rc::Rc; - -use crate::runtime::permissions::PermissionsContainer; -use deno_broadcast_channel::InMemoryBroadcastChannel; -use deno_cache::SqliteBackedCache; -use deno_console::deno_console; -use deno_core::{ - error::AnyError, extension, Extension, JsRuntime, ModuleId, ModuleSpecifier, RuntimeOptions, -}; -use deno_crypto::deno_crypto; -use deno_napi::deno_napi; -use deno_tls::deno_tls; -use deno_url::deno_url; -use deno_web::deno_web; -use deno_webidl::deno_webidl; -use deno_websocket::deno_websocket; -use deno_webstorage::deno_webstorage; -use tokio::select; - -use crate::{user_agent, TypescriptModuleLoader}; - -extension!( - dozer_permissions_worker, - options = { - permissions: PermissionsContainer, - }, - state = |state, options| { - state.put(options.permissions) - } -); - -extension!( - runtime, - deps = [ - deno_webidl, - deno_console, - deno_url, - deno_tls, - deno_web, - deno_fetch, - deno_cache, - deno_websocket, - deno_webstorage, - deno_crypto, - deno_broadcast_channel, - deno_napi - ], - esm_entry_point = "ext:runtime/99_main.js", - esm = [ - dir "js", - "98_global_scope.js", - "99_main.js", - ], -); - -pub struct JsWorker { - pub js_runtime: JsRuntime, -} - -impl JsWorker { - /// This is `MainWorker::from_options` with selected list of extensions. - pub fn new(extra_extensions: Vec) -> Result { - let mut extensions = { - vec![ - deno_webidl::init_ops_and_esm(), - deno_console::init_ops_and_esm(), - deno_url::init_ops_and_esm(), - deno_web::init_ops_and_esm::( - Default::default(), - Default::default(), - ), - deno_fetch::deno_fetch::init_ops_and_esm::( - deno_fetch::Options { - file_fetch_handler: Rc::new(deno_fetch::FsFetchHandler), - ..Default::default() - }, - ), - deno_cache::deno_cache::init_ops_and_esm::(Default::default()), - deno_websocket::init_ops_and_esm::( - user_agent(), - Default::default(), - Default::default(), - ), - deno_webstorage::init_ops_and_esm(Default::default()), - deno_crypto::init_ops_and_esm(Default::default()), - deno_broadcast_channel::deno_broadcast_channel::init_ops_and_esm::< - InMemoryBroadcastChannel, - >(Default::default()), - deno_tls::init_ops_and_esm(), - deno_napi::init_ops_and_esm::(), - //ops::bootstrap::deno_bootstrap::init_ops_and_esm(Some(Default::default())), - dozer_permissions_worker::init_ops_and_esm(PermissionsContainer::allow_all()), - runtime::init_ops_and_esm(), - ] - }; - extensions.extend(extra_extensions); - - Ok(JsWorker { - js_runtime: JsRuntime::new(RuntimeOptions { - module_loader: Some(Rc::new(TypescriptModuleLoader::new()?)), - extensions, - ..Default::default() - }), - }) - } - - /// `MainWorker::evaluate_module`. - pub async fn evaluate_module(&mut self, id: ModuleId) -> Result<(), AnyError> { - let mut receiver = self.js_runtime.mod_evaluate(id); - select! { - biased; - - maybe_result = &mut receiver => { - maybe_result - } - - event_loop_result = self.js_runtime.run_event_loop(Default::default()) => { - event_loop_result?; - receiver.await - } - } - } - - pub async fn execute_main_module( - &mut self, - module_specifier: &ModuleSpecifier, - ) -> Result<(), AnyError> { - let id = self.preload_main_module(module_specifier).await?; - self.evaluate_module(id).await - } - - pub async fn preload_main_module( - &mut self, - module_specifier: &ModuleSpecifier, - ) -> Result { - self.js_runtime.load_main_es_module(module_specifier).await - } - pub async fn preload_side_module( - &mut self, - module_specifier: &ModuleSpecifier, - ) -> Result { - self.js_runtime.load_side_es_module(module_specifier).await - } -} diff --git a/dozer-deno/src/runtime/mod.rs b/dozer-deno/src/runtime/mod.rs deleted file mode 100644 index 246c79190b..0000000000 --- a/dozer-deno/src/runtime/mod.rs +++ /dev/null @@ -1,256 +0,0 @@ -//! `JsRuntime` is `!Send + !Sync`, make it difficult to use. -//! Here we implement a `Runtime` struct that runs `JsRuntime` in a dedicated thread. -//! By sending work to the worker thread, `Runtime` is `Send + Sync`. - -use std::{collections::HashMap, fs::canonicalize, num::NonZeroI32, thread::JoinHandle}; - -use deno_core::{ - anyhow::{bail, Context as _}, - error::AnyError, - Extension, JsRuntime, ModuleSpecifier, -}; -use deno_napi::v8::{self, undefined, Function, Global, Local}; -use dozer_types::{ - json_types::JsonValue, - log::{error, info}, - thiserror, -}; -use tokio::sync::{mpsc, oneshot}; - -use self::conversion::{from_v8, to_v8}; - -mod conversion; -mod js_runtime; -pub use js_runtime::JsWorker; -pub(crate) mod permissions; -#[cfg(test)] -mod tests; - -#[derive(Debug)] -pub struct Runtime { - work_sender: mpsc::Sender, - handle: Option>, -} - -#[derive(Debug, thiserror::Error)] -pub enum Error { - #[error("failed to create JavaScript runtime: {0}")] - CreateJsRuntime(#[source] std::io::Error), - #[error("failed to canonicalize path {0}: {1}")] - CanonicalizePath(String, #[source] std::io::Error), - #[error("failed to load module {0}: {1}")] - LoadModule(String, #[source] AnyError), - #[error("failed to evaluate module {0}: {1}")] - EvaluateModule(String, #[source] AnyError), - #[error("failed to get namespace of module {0}: {1}")] - GetModuleNamespace(String, #[source] AnyError), - #[error("module {0} has no default export")] - ModuleNoDefaultExport(String), - #[error("module {0} default export is not a function: {1}")] - ModuleDefaultExportNotFunction(String, #[source] v8::DataError), -} - -impl Runtime { - /// Returns `Runtime` and the ids of the exported functions. - pub async fn new Extension) + Send + 'static>( - modules: Vec, - extension_generators: Vec, - ) -> Result<(Self, Vec), Error> { - let (init_sender, init_receiver) = oneshot::channel(); - let (work_sender, work_receiver) = mpsc::channel(10); - let handle = std::thread::spawn(move || { - let extensions = extension_generators - .into_iter() - .map(|generate| generate()) - .collect::>(); - let mut worker = match Worker::new(modules, extensions) { - Ok(worker) => worker, - Err(e) => { - let _ = init_sender.send(Err(e)); - return; - } - }; - if init_sender - .send(Ok(worker.functions.keys().cloned().collect())) - .is_err() - { - return; - } - worker.run(work_receiver) - }); - - let mut this = Self { - work_sender, - handle: Some(handle), - }; - let functions = match init_receiver.await { - Ok(Ok(functions)) => functions, - Ok(Err(e)) => return Err(e), - Err(_) => { - this.propagate_panic(); - } - }; - - Ok((this, functions)) - } - - pub async fn call_function( - &mut self, - id: NonZeroI32, - args: Vec, - ) -> Result { - let (return_sender, return_receiver) = oneshot::channel(); - if self - .work_sender - .send(Work::CallFunction { - id, - args, - return_sender, - }) - .await - .is_err() - { - self.propagate_panic(); - } - let Ok(result) = return_receiver.await else { - self.propagate_panic(); - }; - result - } - - fn propagate_panic(&mut self) -> ! { - self.handle - .take() - .expect("runtime panicked before and cannot be used again") - .join() - .unwrap(); - unreachable!("we should have panicked"); - } -} - -struct Worker { - tokio_runtime: tokio::runtime::Runtime, - js_runtime: JsWorker, - functions: HashMap>, -} - -impl Worker { - fn new(modules: Vec, extensions: Vec) -> Result { - let tokio_runtime = tokio::runtime::Builder::new_current_thread() - .enable_all() - .build() - .map_err(Error::CreateJsRuntime)?; - let mut js_worker = JsWorker::new(extensions).map_err(Error::CreateJsRuntime)?; - - let mut functions = HashMap::with_capacity(modules.len()); - for module in modules { - let (id, fun) = tokio_runtime.block_on(Self::load_function(&mut js_worker, module))?; - functions.insert(id, fun); - } - - Ok(Self { - tokio_runtime, - js_runtime: js_worker, - functions, - }) - } - - async fn call_function( - runtime: &mut JsRuntime, - function: NonZeroI32, - args: Vec, - functions: &HashMap>, - ) -> Result { - let function = functions - .get(&function) - .context(format!("function {} not found", function))?; - let mut scope = runtime.handle_scope(); - let recv = undefined(&mut scope); - let args = args - .into_iter() - .map(|arg| to_v8(&mut scope, arg)) - .collect::, _>>()?; - let Some(promise) = Local::new(&mut scope, function).call(&mut scope, recv.into(), &args) - else { - // Deno doesn't expose a way to get the exception. - bail!("uncaught javascript exception"); - }; - let promise = Global::new(&mut scope, promise); - drop(scope); - let result = runtime.resolve(promise); - runtime - .run_event_loop(deno_core::PollEventLoopOptions { - wait_for_inspector: false, - pump_v8_message_loop: true, - }) - .await?; - let result = result.await?; - let scope = &mut runtime.handle_scope(); - let result = Local::new(scope, result); - from_v8(scope, result) - } - - fn run(&mut self, mut work_receiver: mpsc::Receiver) { - let worker = &mut self.js_runtime; - let functions = &self.functions; - self.tokio_runtime.block_on(async { - while let Some(work) = work_receiver.recv().await { - match work { - Work::CallFunction { - id, - args, - return_sender, - } => { - // Ignore error if receiver is closed. - let _ = return_sender.send( - Self::call_function(&mut worker.js_runtime, id, args, functions).await, - ); - } - } - } - }) - } - async fn load_function( - worker: &mut JsWorker, - module: String, - ) -> Result<(NonZeroI32, Global), Error> { - let path = canonicalize(&module).map_err(|e| Error::CanonicalizePath(module.clone(), e))?; - let module_specifier = - ModuleSpecifier::from_file_path(path).expect("we just canonicalized it"); - info!("loading module {}", module_specifier); - let module_id = worker - .preload_side_module(&module_specifier) - .await - .map_err(|e| Error::LoadModule(module.clone(), e))?; - worker - .evaluate_module(module_id) - .await - .map_err(|e| Error::EvaluateModule(module.clone(), e))?; - let namespace = worker - .js_runtime - .get_module_namespace(module_id) - .map_err(|e| Error::GetModuleNamespace(module.clone(), e))?; - let scope = &mut worker.js_runtime.handle_scope(); - let namespace = v8::Local::new(scope, namespace); - let default_key = v8::String::new_external_onebyte_static(scope, b"default") - .unwrap() - .into(); - let default_export = namespace - .get(scope, default_key) - .ok_or_else(|| Error::ModuleNoDefaultExport(module.clone()))?; - let function: Local = default_export - .try_into() - .map_err(|e| Error::ModuleDefaultExportNotFunction(module.clone(), e))?; - let id = function.get_identity_hash(); - Ok((id, Global::new(scope, function))) - } -} - -#[derive(Debug)] -enum Work { - CallFunction { - id: NonZeroI32, - args: Vec, - return_sender: oneshot::Sender>, - }, -} diff --git a/dozer-deno/src/runtime/permissions.rs b/dozer-deno/src/runtime/permissions.rs deleted file mode 100644 index 2156746939..0000000000 --- a/dozer-deno/src/runtime/permissions.rs +++ /dev/null @@ -1,66 +0,0 @@ -// Copyright 2018-2024 the Deno authors. All rights reserved. MIT license. - -use std::path::Path; - -use deno_core::error::AnyError; -use deno_core::url::Url; - -// NOTE: Temporary permissions container to satisfy traits. We are migrating to the deno_permissions -// crate. -#[derive(Debug, Clone)] -pub struct PermissionsContainer(pub deno_permissions::PermissionsContainer); - -impl PermissionsContainer { - pub fn allow_all() -> Self { - Self(deno_permissions::PermissionsContainer::allow_all()) - } -} - -impl std::ops::Deref for PermissionsContainer { - type Target = deno_permissions::PermissionsContainer; - - fn deref(&self) -> &Self::Target { - &self.0 - } -} - -impl std::ops::DerefMut for PermissionsContainer { - fn deref_mut(&mut self) -> &mut Self::Target { - &mut self.0 - } -} - -impl deno_fetch::FetchPermissions for PermissionsContainer { - #[inline(always)] - fn check_net_url(&mut self, url: &Url, api_name: &str) -> Result<(), AnyError> { - self.0.check_net_url(url, api_name) - } - - #[inline(always)] - fn check_read(&mut self, path: &Path, api_name: &str) -> Result<(), AnyError> { - self.0.check_read(path, api_name) - } -} - -impl deno_web::TimersPermission for PermissionsContainer { - #[inline(always)] - fn allow_hrtime(&mut self) -> bool { - self.0.allow_hrtime() - } -} - -impl deno_websocket::WebSocketPermissions for PermissionsContainer { - #[inline(always)] - fn check_net_url(&mut self, url: &Url, api_name: &str) -> Result<(), AnyError> { - self.0.check_net_url(url, api_name) - } -} - -// NOTE(bartlomieju): for now, NAPI uses `--allow-ffi` flag, but that might -// change in the future. -impl deno_napi::NapiPermissions for PermissionsContainer { - #[inline(always)] - fn check(&mut self, path: Option<&Path>) -> Result<(), AnyError> { - self.0.check_ffi(path) - } -} diff --git a/dozer-deno/src/runtime/square.js b/dozer-deno/src/runtime/square.js deleted file mode 100644 index 8a73c42839..0000000000 --- a/dozer-deno/src/runtime/square.js +++ /dev/null @@ -1,3 +0,0 @@ -export default function (input) { - return input * input; -} diff --git a/dozer-deno/src/runtime/tests.rs b/dozer-deno/src/runtime/tests.rs deleted file mode 100644 index 00ab9449de..0000000000 --- a/dozer-deno/src/runtime/tests.rs +++ /dev/null @@ -1,43 +0,0 @@ -use dozer_types::json_types::json; - -use super::*; - -async fn call_function(module: &str, args: Vec) -> Result { - let (mut runtime, functions) = - Runtime::new:: Extension>(vec![format!("src/runtime/{module}")], vec![]).await?; - runtime.call_function(functions[0], args).await -} - -#[tokio::test] -async fn test_runtime() { - assert_eq!( - call_function("square.js", vec![json!(2.0)]).await.unwrap(), - json!(4.0) - ); -} - -#[tokio::test] -async fn test_function_call_exception() { - let error = call_function("exception.js", vec![]).await.unwrap_err(); - assert_eq!(error.to_string(), "uncaught javascript exception"); -} - -#[tokio::test] -async fn test_async_function_call() { - let Ok(result) = call_function("fetch.js", vec![]) - .await - .unwrap() - .into_array() - else { - panic!("expected array") - }; - assert!(!result.is_empty()); -} - -#[tokio::test] -async fn test_async_function_call_exception() { - let error = call_function("fetch_exception.js", vec![]) - .await - .unwrap_err(); - assert!(error.to_string().starts_with("SyntaxError: ")); -} diff --git a/dozer-deno/src/ts_module_loader/cache.rs b/dozer-deno/src/ts_module_loader/cache.rs deleted file mode 100644 index e01cfe054f..0000000000 --- a/dozer-deno/src/ts_module_loader/cache.rs +++ /dev/null @@ -1,43 +0,0 @@ -use std::{path::Path, time::SystemTime}; - -use super::fs::atomic_write_file; - -/// Permissions used to save a file in the disk caches. -pub const CACHE_PERM: u32 = 0o644; - -#[derive(Debug, Clone)] -pub struct RealDenoCacheEnv; - -impl deno_cache_dir::DenoCacheEnv for RealDenoCacheEnv { - fn read_file_bytes(&self, path: &Path) -> std::io::Result>> { - match std::fs::read(path) { - Ok(s) => Ok(Some(s)), - Err(err) if err.kind() == std::io::ErrorKind::NotFound => Ok(None), - Err(err) => Err(err), - } - } - - fn atomic_write_file(&self, path: &Path, bytes: &[u8]) -> std::io::Result<()> { - atomic_write_file(path, bytes, CACHE_PERM) - } - - fn modified(&self, path: &Path) -> std::io::Result> { - match std::fs::metadata(path) { - Ok(metadata) => Ok(Some( - metadata.modified().unwrap_or_else(|_| SystemTime::now()), - )), - Err(err) if err.kind() == std::io::ErrorKind::NotFound => Ok(None), - Err(err) => Err(err), - } - } - - fn is_file(&self, path: &Path) -> bool { - path.is_file() - } - - fn time_now(&self) -> SystemTime { - SystemTime::now() - } -} - -pub type GlobalHttpCache = deno_cache_dir::GlobalHttpCache; diff --git a/dozer-deno/src/ts_module_loader/file_fetcher.rs b/dozer-deno/src/ts_module_loader/file_fetcher.rs deleted file mode 100644 index de8b6a0ac1..0000000000 --- a/dozer-deno/src/ts_module_loader/file_fetcher.rs +++ /dev/null @@ -1,557 +0,0 @@ -use dozer_types::log::{self, debug}; -use std::{collections::HashMap, future::Future, pin::Pin, sync::Arc}; -use tokio::fs; - -use deno_ast::MediaType; -use deno_cache_dir::{GlobalToLocalCopy, HttpCache}; -use deno_core::{ - self, - error::{custom_error, generic_error, uri_error, AnyError}, - futures::{self, FutureExt as _}, - parking_lot::Mutex, - ModuleSpecifier, -}; -use deno_fetch::{ - data_url::DataUrl, - reqwest::{ - header::{HeaderValue, ACCEPT, IF_NONE_MATCH}, - StatusCode, Url, - }, -}; -use deno_terminal::colors; -use deno_web::BlobStore; - -use super::{ - http_util::{resolve_redirect_from_response, HeadersMap, HttpClient}, - text_encoding, -}; -use crate::runtime::permissions::PermissionsContainer; - -pub const SUPPORTED_SCHEMES: [&str; 5] = ["data", "blob", "file", "http", "https"]; - -/// A structure representing a source file. -#[derive(Debug, Clone, Eq, PartialEq)] -pub struct File { - /// For remote files, if there was an `X-TypeScript-Type` header, the parsed - /// out value of that header. - pub maybe_types: Option, - /// The resolved media type for the file. - pub media_type: MediaType, - /// The source of the file as a string. - pub source: Arc, - /// The _final_ specifier for the file. The requested specifier and the final - /// specifier maybe different for remote files that have been redirected. - pub specifier: ModuleSpecifier, - - pub maybe_headers: Option>, -} - -/// Simple struct implementing in-process caching to prevent multiple -/// fs reads/net fetches for same file. -#[derive(Debug, Clone, Default)] -struct FileCache(Arc>>); - -impl FileCache { - pub fn get(&self, specifier: &ModuleSpecifier) -> Option { - let cache = self.0.lock(); - cache.get(specifier).cloned() - } - - pub fn insert(&self, specifier: ModuleSpecifier, file: File) -> Option { - let mut cache = self.0.lock(); - cache.insert(specifier, file) - } -} - -/// Fetch a source file from the local file system. -async fn fetch_local(specifier: &ModuleSpecifier) -> Result { - let local = specifier - .to_file_path() - .map_err(|_| uri_error(format!("Invalid file path.\n Specifier: {specifier}")))?; - let bytes = fs::read(local).await?; - let charset = text_encoding::detect_charset(&bytes).to_string(); - let source = get_source_from_bytes(bytes, Some(charset))?; - let media_type = MediaType::from_specifier(specifier); - - Ok(File { - maybe_types: None, - media_type, - source: source.into(), - specifier: specifier.clone(), - maybe_headers: None, - }) -} - -/// Returns the decoded body and content-type of a provided -/// data URL. -pub fn get_source_from_data_url(specifier: &ModuleSpecifier) -> Result<(String, String), AnyError> { - let data_url = DataUrl::process(specifier.as_str()).map_err(|e| uri_error(format!("{e:?}")))?; - let mime = data_url.mime_type(); - let charset = mime.get_parameter("charset").map(|v| v.to_string()); - let (bytes, _) = data_url - .decode_to_vec() - .map_err(|e| uri_error(format!("{e:?}")))?; - Ok((get_source_from_bytes(bytes, charset)?, format!("{mime}"))) -} - -/// Given a vector of bytes and optionally a charset, decode the bytes to a -/// string. -pub fn get_source_from_bytes( - bytes: Vec, - maybe_charset: Option, -) -> Result { - let source = if let Some(charset) = maybe_charset { - text_encoding::convert_to_utf8(&bytes, &charset)?.to_string() - } else { - String::from_utf8(bytes)? - }; - - Ok(source) -} - -/// Return a validated scheme for a given module specifier. -fn get_validated_scheme(specifier: &ModuleSpecifier) -> Result { - let scheme = specifier.scheme(); - if !SUPPORTED_SCHEMES.contains(&scheme) { - Err(generic_error(format!( - "Unsupported scheme \"{scheme}\" for module \"{specifier}\". Supported schemes: {SUPPORTED_SCHEMES:#?}" - ))) - } else { - Ok(scheme.to_string()) - } -} - -/// Resolve a media type and optionally the charset from a module specifier and -/// the value of a content type header. -pub fn map_content_type( - specifier: &ModuleSpecifier, - maybe_content_type: Option<&String>, -) -> (MediaType, Option) { - if let Some(content_type) = maybe_content_type { - let mut content_types = content_type.split(';'); - let content_type = content_types.next().unwrap(); - let media_type = MediaType::from_content_type(specifier, content_type); - let charset = content_types - .map(str::trim) - .find_map(|s| s.strip_prefix("charset=")) - .map(String::from); - - (media_type, charset) - } else { - (MediaType::from_specifier(specifier), None) - } -} - -pub struct FetchOptions<'a> { - pub specifier: &'a ModuleSpecifier, - pub permissions: PermissionsContainer, - pub maybe_accept: Option<&'a str>, -} - -/// A structure for resolving, fetching and caching source files. -#[derive(Debug, Clone)] -pub struct FileFetcher { - allow_remote: bool, - cache: FileCache, - http_cache: Arc, - http_client: Arc, - blob_store: Arc, - download_log_level: log::Level, -} - -impl FileFetcher { - pub fn new( - http_cache: Arc, - allow_remote: bool, - http_client: Arc, - blob_store: Arc, - ) -> Self { - Self { - allow_remote, - cache: Default::default(), - http_cache, - http_client, - blob_store, - download_log_level: log::Level::Info, - } - } - - /// Creates a `File` structure for a remote file. - fn build_remote_file( - &self, - specifier: &ModuleSpecifier, - bytes: Vec, - headers: &HashMap, - ) -> Result { - let maybe_content_type = headers.get("content-type"); - let (media_type, maybe_charset) = map_content_type(specifier, maybe_content_type); - let source = get_source_from_bytes(bytes, maybe_charset)?; - let maybe_types = match media_type { - MediaType::JavaScript | MediaType::Cjs | MediaType::Mjs | MediaType::Jsx => { - headers.get("x-typescript-types").cloned() - } - _ => None, - }; - - Ok(File { - maybe_types, - media_type, - source: source.into(), - specifier: specifier.clone(), - maybe_headers: Some(headers.clone()), - }) - } - - /// Fetch cached remote file. - /// - /// This is a recursive operation if source file has redirections. - pub fn fetch_cached( - &self, - specifier: &ModuleSpecifier, - redirect_limit: i64, - ) -> Result, AnyError> { - debug!("FileFetcher::fetch_cached - specifier: {}", specifier); - if redirect_limit < 0 { - return Err(custom_error("Http", "Too many redirects.")); - } - - let cache_key = self.http_cache.cache_item_key(specifier)?; // compute this once - let Some(headers) = self.http_cache.read_headers(&cache_key)? else { - return Ok(None); - }; - if let Some(redirect_to) = headers.get("location") { - let redirect = deno_core::resolve_import(redirect_to, specifier.as_str())?; - return self.fetch_cached(&redirect, redirect_limit - 1); - } - let Some(bytes) = - self.http_cache - .read_file_bytes(&cache_key, None, GlobalToLocalCopy::Allow)? - else { - return Ok(None); - }; - let file = self.build_remote_file(specifier, bytes, &headers)?; - - Ok(Some(file)) - } - - /// Convert a data URL into a file, resulting in an error if the URL is - /// invalid. - fn fetch_data_url(&self, specifier: &ModuleSpecifier) -> Result { - debug!("FileFetcher::fetch_data_url() - specifier: {}", specifier); - let (source, content_type) = get_source_from_data_url(specifier)?; - let (media_type, _) = map_content_type(specifier, Some(&content_type)); - let mut headers = HashMap::new(); - headers.insert("content-type".to_string(), content_type); - Ok(File { - maybe_types: None, - media_type, - source: source.into(), - specifier: specifier.clone(), - maybe_headers: Some(headers), - }) - } - - /// Get a blob URL. - async fn fetch_blob_url(&self, specifier: &ModuleSpecifier) -> Result { - debug!("FileFetcher::fetch_blob_url() - specifier: {}", specifier); - let blob = self - .blob_store - .get_object_url(specifier.clone()) - .ok_or_else(|| { - custom_error("NotFound", format!("Blob URL not found: \"{specifier}\".")) - })?; - - let content_type = blob.media_type.clone(); - let bytes = blob.read_all().await?; - - let (media_type, maybe_charset) = map_content_type(specifier, Some(&content_type)); - let source = get_source_from_bytes(bytes, maybe_charset)?; - let mut headers = HashMap::new(); - headers.insert("content-type".to_string(), content_type); - - Ok(File { - maybe_types: None, - media_type, - source: source.into(), - specifier: specifier.clone(), - maybe_headers: Some(headers), - }) - } - - /// Asynchronously fetch remote source file specified by the URL following - /// redirects. - /// - /// **Note** this is a recursive method so it can't be "async", but needs to - /// return a `Pin>`. - fn fetch_remote( - &self, - specifier: &ModuleSpecifier, - permissions: PermissionsContainer, - redirect_limit: i64, - maybe_accept: Option, - ) -> Pin> + Send>> { - debug!("FileFetcher::fetch_remote() - specifier: {}", specifier); - if redirect_limit < 0 { - return futures::future::err(custom_error("Http", "Too many redirects.")).boxed(); - } - - if let Err(err) = permissions.check_specifier(specifier) { - return futures::future::err(err).boxed(); - } - - match self.fetch_cached(specifier, redirect_limit) { - Ok(Some(file)) => { - return futures::future::ok(file).boxed(); - } - Ok(None) => {} - Err(err) => { - return futures::future::err(err).boxed(); - } - } - - log::log!( - self.download_log_level, - "{} {}", - colors::green("Download"), - specifier - ); - - let maybe_etag = self - .http_cache - .cache_item_key(specifier) - .ok() - .and_then(|key| self.http_cache.read_headers(&key).ok().flatten()) - .and_then(|headers| headers.get("etag").cloned()); - let specifier = specifier.clone(); - let client = self.http_client.clone(); - let file_fetcher = self.clone(); - // A single pass of fetch either yields code or yields a redirect, server - // error causes a single retry to avoid crashing hard on intermittent failures. - - async fn handle_request_or_server_error( - retried: &mut bool, - specifier: &Url, - err_str: String, - ) -> Result<(), AnyError> { - // Retry once, and bail otherwise. - if !*retried { - *retried = true; - log::debug!("Import '{}' failed: {}. Retrying...", specifier, err_str); - tokio::time::sleep(std::time::Duration::from_millis(50)).await; - Ok(()) - } else { - Err(generic_error(format!( - "Import '{}' failed: {}", - specifier, err_str - ))) - } - } - - async move { - let mut retried = false; - loop { - let result = match fetch_once( - &client, - FetchOnceArgs { - url: specifier.clone(), - maybe_accept: maybe_accept.clone(), - maybe_etag: maybe_etag.clone(), - }, - ) - .await? - { - FetchOnceResult::NotModified => { - let file = file_fetcher.fetch_cached(&specifier, 10)?.unwrap(); - Ok(file) - } - FetchOnceResult::Redirect(redirect_url, headers) => { - file_fetcher.http_cache.set(&specifier, headers, &[])?; - file_fetcher - .fetch_remote( - &redirect_url, - permissions, - redirect_limit - 1, - maybe_accept, - ) - .await - } - FetchOnceResult::Code(bytes, headers) => { - file_fetcher - .http_cache - .set(&specifier, headers.clone(), &bytes)?; - let file = file_fetcher.build_remote_file(&specifier, bytes, &headers)?; - Ok(file) - } - FetchOnceResult::RequestError(err) => { - handle_request_or_server_error(&mut retried, &specifier, err).await?; - continue; - } - FetchOnceResult::ServerError(status) => { - handle_request_or_server_error( - &mut retried, - &specifier, - status.to_string(), - ) - .await?; - continue; - } - }; - break result; - } - } - .boxed() - } - - /// Fetch a source file and asynchronously return it. - pub async fn fetch( - &self, - specifier: &ModuleSpecifier, - permissions: PermissionsContainer, - ) -> Result { - self.fetch_with_options(FetchOptions { - specifier, - permissions, - maybe_accept: None, - }) - .await - } - - pub async fn fetch_with_options(&self, options: FetchOptions<'_>) -> Result { - let specifier = options.specifier; - debug!("FileFetcher::fetch() - specifier: {}", specifier); - let scheme = get_validated_scheme(specifier)?; - options.permissions.check_specifier(specifier)?; - if let Some(file) = self.cache.get(specifier) { - Ok(file) - } else if scheme == "file" { - // we do not in memory cache files, as this would prevent files on the - // disk changing effecting things like workers and dynamic imports. - fetch_local(specifier).await - } else if scheme == "data" { - self.fetch_data_url(specifier) - } else if scheme == "blob" { - self.fetch_blob_url(specifier).await - } else if !self.allow_remote { - Err(custom_error( - "NoRemote", - format!("A remote specifier was requested: \"{specifier}\", but --no-remote is specified."), - )) - } else { - let result = self - .fetch_remote( - specifier, - options.permissions, - 10, - options.maybe_accept.map(String::from), - ) - .await; - if let Ok(file) = &result { - self.cache.insert(specifier.clone(), file.clone()); - } - result - } - } -} - -#[derive(Debug, Eq, PartialEq)] -enum FetchOnceResult { - Code(Vec, HeadersMap), - NotModified, - Redirect(Url, HeadersMap), - RequestError(String), - ServerError(StatusCode), -} - -#[derive(Debug)] -struct FetchOnceArgs { - pub url: Url, - pub maybe_accept: Option, - pub maybe_etag: Option, -} - -/// Asynchronously fetches the given HTTP URL one pass only. -/// If no redirect is present and no error occurs, -/// yields Code(ResultPayload). -/// If redirect occurs, does not follow and -/// yields Redirect(url). -async fn fetch_once( - http_client: &HttpClient, - args: FetchOnceArgs, -) -> Result { - let mut request = http_client.get_no_redirect(args.url.clone())?; - - if let Some(etag) = args.maybe_etag { - let if_none_match_val = HeaderValue::from_str(&etag)?; - request = request.header(IF_NONE_MATCH, if_none_match_val); - } - if let Some(accept) = args.maybe_accept { - let accepts_val = HeaderValue::from_str(&accept)?; - request = request.header(ACCEPT, accepts_val); - } - let response = match request.send().await { - Ok(resp) => resp, - Err(err) => { - if err.is_connect() || err.is_timeout() { - return Ok(FetchOnceResult::RequestError(err.to_string())); - } - return Err(err.into()); - } - }; - - if response.status() == StatusCode::NOT_MODIFIED { - return Ok(FetchOnceResult::NotModified); - } - - let mut result_headers = HashMap::new(); - let response_headers = response.headers(); - - if let Some(warning) = response_headers.get("X-Deno-Warning") { - log::warn!( - "{} {}", - colors::yellow("Warning"), - warning.to_str().unwrap() - ); - } - - for key in response_headers.keys() { - let key_str = key.to_string(); - let values = response_headers.get_all(key); - let values_str = values - .iter() - .map(|e| e.to_str().unwrap().to_string()) - .collect::>() - .join(","); - result_headers.insert(key_str, values_str); - } - - if response.status().is_redirection() { - let new_url = resolve_redirect_from_response(&args.url, &response)?; - return Ok(FetchOnceResult::Redirect(new_url, result_headers)); - } - - let status = response.status(); - - if status.is_server_error() { - return Ok(FetchOnceResult::ServerError(status)); - } - - if status.is_client_error() { - let err = if response.status() == StatusCode::NOT_FOUND { - custom_error( - "NotFound", - format!("Import '{}' failed, not found.", args.url), - ) - } else { - generic_error(format!( - "Import '{}' failed: {}", - args.url, - response.status() - )) - }; - return Err(err); - } - - let body = response.bytes().await?.into(); - - Ok(FetchOnceResult::Code(body, result_headers)) -} diff --git a/dozer-deno/src/ts_module_loader/fs.rs b/dozer-deno/src/ts_module_loader/fs.rs deleted file mode 100644 index b2c9f5c4bf..0000000000 --- a/dozer-deno/src/ts_module_loader/fs.rs +++ /dev/null @@ -1,112 +0,0 @@ -use std::{ - fmt::Write as _, - fs::OpenOptions, - io::{Error, ErrorKind, Write as _}, - path::Path, -}; - -use deno_crypto::rand; - -/// Writes the file to the file system at a temporary path, then -/// renames it to the destination in a single sys call in order -/// to never leave the file system in a corrupted state. -/// -/// This also handles creating the directory if a NotFound error -/// occurs. -pub fn atomic_write_file>( - file_path: &Path, - data: T, - mode: u32, -) -> std::io::Result<()> { - fn atomic_write_file_raw( - temp_file_path: &Path, - file_path: &Path, - data: &[u8], - mode: u32, - ) -> std::io::Result<()> { - write_file(temp_file_path, data, mode)?; - std::fs::rename(temp_file_path, file_path)?; - Ok(()) - } - - fn inner(file_path: &Path, data: &[u8], mode: u32) -> std::io::Result<()> { - let temp_file_path = { - let rand: String = (0..4).fold(String::new(), |mut output, _| { - let _ = write!(output, "{:02x}", rand::random::()); - output - }); - let extension = format!("{rand}.tmp"); - file_path.with_extension(extension) - }; - - if let Err(write_err) = atomic_write_file_raw(&temp_file_path, file_path, data, mode) { - if write_err.kind() == ErrorKind::NotFound { - let parent_dir_path = file_path.parent().unwrap(); - match std::fs::create_dir_all(parent_dir_path) { - Ok(()) => { - return atomic_write_file_raw(&temp_file_path, file_path, data, mode) - .map_err(|err| add_file_context_to_err(file_path, err)); - } - Err(create_err) => { - if !parent_dir_path.exists() { - return Err(Error::new( - create_err.kind(), - format!( - "{:#} (for '{}')\nCheck the permission of the directory.", - create_err, - parent_dir_path.display() - ), - )); - } - } - } - } - return Err(add_file_context_to_err(file_path, write_err)); - } - Ok(()) - } - - inner(file_path, data.as_ref(), mode) -} - -fn add_file_context_to_err(file_path: &Path, err: Error) -> Error { - Error::new( - err.kind(), - format!("{:#} (for '{}')", err, file_path.display()), - ) -} - -pub fn write_file>(filename: &Path, data: T, mode: u32) -> std::io::Result<()> { - write_file_2(filename, data, true, mode, true, false) -} - -pub fn write_file_2>( - filename: &Path, - data: T, - update_mode: bool, - mode: u32, - is_create: bool, - is_append: bool, -) -> std::io::Result<()> { - let mut file = OpenOptions::new() - .read(false) - .write(true) - .append(is_append) - .truncate(!is_append) - .create(is_create) - .open(filename)?; - - if update_mode { - #[cfg(unix)] - { - use std::os::unix::fs::PermissionsExt; - let mode = mode & 0o777; - let permissions = PermissionsExt::from_mode(mode); - file.set_permissions(permissions)?; - } - #[cfg(not(unix))] - let _ = mode; - } - - file.write_all(data.as_ref()) -} diff --git a/dozer-deno/src/ts_module_loader/http_util.rs b/dozer-deno/src/ts_module_loader/http_util.rs deleted file mode 100644 index 071ea21a93..0000000000 --- a/dozer-deno/src/ts_module_loader/http_util.rs +++ /dev/null @@ -1,114 +0,0 @@ -use std::{collections::HashMap, sync::Arc}; - -use deno_core::error::{generic_error, AnyError}; -use deno_fetch::{ - create_http_client, - reqwest::{self, header::LOCATION, Response, Url}, - CreateHttpClientOptions, -}; -use deno_tls::RootCertStoreProvider; -use dozer_types::log; - -use crate::user_agent; - -/// Construct the next uri based on base uri and location header fragment -/// See -fn resolve_url_from_location(base_url: &Url, location: &str) -> Url { - if location.starts_with("http://") || location.starts_with("https://") { - // absolute uri - Url::parse(location).expect("provided redirect url should be a valid url") - } else if location.starts_with("//") { - // "//" authority path-abempty - Url::parse(&format!("{}:{}", base_url.scheme(), location)) - .expect("provided redirect url should be a valid url") - } else if location.starts_with('/') { - // path-absolute - base_url - .join(location) - .expect("provided redirect url should be a valid url") - } else { - // assuming path-noscheme | path-empty - let base_url_path_str = base_url.path().to_owned(); - // Pop last part or url (after last slash) - let segs: Vec<&str> = base_url_path_str.rsplitn(2, '/').collect(); - let new_path = format!("{}/{}", segs.last().unwrap_or(&""), location); - base_url - .join(&new_path) - .expect("provided redirect url should be a valid url") - } -} - -pub fn resolve_redirect_from_response( - request_url: &Url, - response: &Response, -) -> Result { - debug_assert!(response.status().is_redirection()); - if let Some(location) = response.headers().get(LOCATION) { - let location_string = location.to_str()?; - log::debug!("Redirecting to {:?}...", &location_string); - let new_url = resolve_url_from_location(request_url, location_string); - Ok(new_url) - } else { - Err(generic_error(format!( - "Redirection from '{request_url}' did not provide location header" - ))) - } -} - -// TODO(ry) HTTP headers are not unique key, value pairs. There may be more than -// one header line with the same key. This should be changed to something like -// Vec<(String, String)> -pub type HeadersMap = HashMap; - -pub struct HttpClient { - options: CreateHttpClientOptions, - root_cert_store_provider: Option>, - cell: once_cell::sync::OnceCell, -} - -impl std::fmt::Debug for HttpClient { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.debug_struct("HttpClient") - .field("options", &self.options) - .finish() - } -} - -impl HttpClient { - pub fn new( - root_cert_store_provider: Option>, - unsafely_ignore_certificate_errors: Option>, - ) -> Self { - Self { - options: CreateHttpClientOptions { - unsafely_ignore_certificate_errors, - ..Default::default() - }, - root_cert_store_provider, - cell: Default::default(), - } - } - - fn client(&self) -> Result<&reqwest::Client, AnyError> { - self.cell.get_or_try_init(|| { - create_http_client( - &user_agent(), - CreateHttpClientOptions { - root_cert_store: match &self.root_cert_store_provider { - Some(provider) => Some(provider.get_or_try_init()?.clone()), - None => None, - }, - ..self.options.clone() - }, - ) - }) - } - - /// Do a GET request without following redirects. - pub fn get_no_redirect( - &self, - url: U, - ) -> Result { - Ok(self.client()?.get(url)) - } -} diff --git a/dozer-deno/src/ts_module_loader/mod.rs b/dozer-deno/src/ts_module_loader/mod.rs deleted file mode 100644 index 21630ab0f6..0000000000 --- a/dozer-deno/src/ts_module_loader/mod.rs +++ /dev/null @@ -1,169 +0,0 @@ -// Copyright 2018-2023 the Deno authors. All rights reserved. MIT license. -//! This example shows how to use swc to transpile TypeScript and JSX/TSX -//! modules. -//! -//! It will only transpile, not typecheck (like Deno's `--no-check` flag). - -use std::cell::RefCell; -use std::collections::HashMap; -use std::rc::Rc; -use std::sync::Arc; - -use deno_core::{self, anyhow, futures}; -use deno_core::{ModuleLoadResponse, RequestedModuleType}; - -use crate::runtime::permissions::PermissionsContainer; -use anyhow::bail; -use anyhow::Error; -use deno_ast::MediaType; -use deno_ast::ParseParams; -use deno_ast::SourceTextInfo; -use deno_core::error::AnyError; -use deno_core::resolve_import; -use deno_core::ModuleLoader; -use deno_core::ModuleSource; -use deno_core::ModuleSpecifier; -use deno_core::ModuleType; -use deno_core::ResolutionKind; -use deno_core::SourceMapGetter; -use futures::FutureExt; - -use self::cache::GlobalHttpCache; -use self::cache::RealDenoCacheEnv; -use self::file_fetcher::FileFetcher; -use self::http_util::HttpClient; - -use tempfile::TempDir; - -#[derive(Clone)] -struct SourceMapStore(Rc>>>); - -impl SourceMapGetter for SourceMapStore { - fn get_source_map(&self, specifier: &str) -> Option> { - self.0.borrow().get(specifier).cloned() - } - - fn get_source_line(&self, _file_name: &str, _line_number: usize) -> Option { - None - } -} - -pub struct TypescriptModuleLoader { - source_maps: SourceMapStore, - file_fetcher: FileFetcher, - _temp_dir: TempDir, -} - -impl TypescriptModuleLoader { - pub fn new() -> Result { - let temp_dir = tempfile::Builder::new() - .prefix("dozer-deno-cache") - .tempdir()?; - Ok(Self { - source_maps: SourceMapStore(Rc::new(RefCell::new(HashMap::new()))), - file_fetcher: FileFetcher::new( - Arc::new(GlobalHttpCache::new( - temp_dir.path().to_path_buf(), - RealDenoCacheEnv, - )), - true, - Arc::new(HttpClient::new(None, None)), - Default::default(), - ), - _temp_dir: temp_dir, - }) - } -} - -impl ModuleLoader for TypescriptModuleLoader { - fn resolve( - &self, - specifier: &str, - referrer: &str, - _kind: ResolutionKind, - ) -> Result { - Ok(resolve_import(specifier, referrer)?) - } - - fn load( - &self, - module_specifier: &ModuleSpecifier, - _maybe_referrer: Option<&ModuleSpecifier>, - _is_dyn_import: bool, - _requested_module_type: RequestedModuleType, - ) -> ModuleLoadResponse { - let source_maps = self.source_maps.clone(); - async fn load( - source_maps: SourceMapStore, - file_fetcher: FileFetcher, - module_specifier: ModuleSpecifier, - ) -> Result { - let media_type = MediaType::from_specifier(&module_specifier); - let (module_type, should_transpile) = match media_type { - MediaType::JavaScript | MediaType::Mjs | MediaType::Cjs => { - (ModuleType::JavaScript, false) - } - MediaType::Jsx => (ModuleType::JavaScript, true), - MediaType::TypeScript - | MediaType::Mts - | MediaType::Cts - | MediaType::Dts - | MediaType::Dmts - | MediaType::Dcts - | MediaType::Tsx => (ModuleType::JavaScript, true), - MediaType::Json => (ModuleType::Json, false), - _ => bail!("Unknown media type: {:?}", module_specifier), - }; - - let code = file_fetcher - .fetch(&module_specifier, PermissionsContainer::allow_all()) - .await? - .source - .to_string(); - let code = if should_transpile { - let parsed = deno_ast::parse_module(ParseParams { - specifier: module_specifier.to_string(), - text_info: SourceTextInfo::from_string(code), - media_type, - capture_tokens: false, - scope_analysis: false, - maybe_syntax: None, - })?; - let res = parsed.transpile(&deno_ast::EmitOptions { - inline_source_map: false, - source_map: true, - inline_sources: true, - ..Default::default() - })?; - let source_map = res.source_map.unwrap(); - source_maps - .0 - .borrow_mut() - .insert(module_specifier.to_string(), source_map.into_bytes()); - res.text - } else { - code - }; - Ok(ModuleSource::new( - module_type, - deno_core::ModuleSourceCode::String(code.into()), - &module_specifier, - )) - } - - ModuleLoadResponse::Async( - load( - source_maps, - self.file_fetcher.clone(), - module_specifier.clone(), - ) - .boxed_local(), - ) - } -} - -mod cache; -mod file_fetcher; -mod fs; -mod http_util; -mod text_encoding; diff --git a/dozer-deno/src/ts_module_loader/text_encoding.rs b/dozer-deno/src/ts_module_loader/text_encoding.rs deleted file mode 100644 index 3bc3439be5..0000000000 --- a/dozer-deno/src/ts_module_loader/text_encoding.rs +++ /dev/null @@ -1,40 +0,0 @@ -use std::{ - borrow::Cow, - io::{Error, ErrorKind}, -}; - -use encoding_rs::Encoding; - -/// Attempts to detect the character encoding of the provided bytes. -/// -/// Supports UTF-8, UTF-16 Little Endian and UTF-16 Big Endian. -pub fn detect_charset(bytes: &'_ [u8]) -> &'static str { - const UTF16_LE_BOM: &[u8] = b"\xFF\xFE"; - const UTF16_BE_BOM: &[u8] = b"\xFE\xFF"; - - if bytes.starts_with(UTF16_LE_BOM) { - "utf-16le" - } else if bytes.starts_with(UTF16_BE_BOM) { - "utf-16be" - } else { - // Assume everything else is utf-8 - "utf-8" - } -} - -/// Attempts to convert the provided bytes to a UTF-8 string. -/// -/// Supports all encodings supported by the encoding_rs crate, which includes -/// all encodings specified in the WHATWG Encoding Standard, and only those -/// encodings (see: ). -pub fn convert_to_utf8<'a>(bytes: &'a [u8], charset: &'_ str) -> Result, Error> { - match Encoding::for_label(charset.as_bytes()) { - Some(encoding) => encoding - .decode_without_bom_handling_and_without_replacement(bytes) - .ok_or_else(|| ErrorKind::InvalidData.into()), - None => Err(Error::new( - ErrorKind::InvalidInput, - format!("Unsupported charset: {charset}"), - )), - } -} diff --git a/dozer-ingestion/Cargo.toml b/dozer-ingestion/Cargo.toml deleted file mode 100644 index 1525d8cc5f..0000000000 --- a/dozer-ingestion/Cargo.toml +++ /dev/null @@ -1,60 +0,0 @@ -[package] -name = "dozer-ingestion" -version = "0.4.0" -edition = "2021" -authors = ["getdozer/dozer-dev"] -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-ingestion-connector = { path = "./connector" } -dozer-ingestion-deltalake = { path = "./deltalake", optional = true } -dozer-ingestion-ethereum = { path = "./ethereum", optional = true } -dozer-ingestion-grpc = { path = "./grpc" } -dozer-ingestion-javascript = { path = "./javascript", optional = true } -dozer-ingestion-kafka = { path = "./kafka", optional = true } -dozer-ingestion-mongodb = { path = "./mongodb", optional = true } -dozer-ingestion-mysql = { path = "./mysql" } -dozer-ingestion-object-store = { path = "./object-store", optional = true } -dozer-ingestion-postgres = { path = "./postgres" } -dozer-ingestion-snowflake = { path = "./snowflake", optional = true } -dozer-ingestion-webhook = { path = "./webhook" } - -tokio = { version = "1", features = ["full"] } -futures = "0.3.28" -prost-reflect = { version = "0.12.0", features = ["serde", "text-format"] } -rand = "0.8.5" -url = "2.4.1" -chrono = "0.4.26" -bytes = "1.4.0" - -[dev-dependencies] -criterion = { version = "0.5.1", features = ["html_reports"] } -serial_test = "2.0.0" -dozer-tracing = { path = "../dozer-tracing" } -tempfile = "3.10.1" -parquet = "50.0.0" -env_logger = "0.10.0" -hex = "0.4.3" -dozer-utils = { path = "../dozer-utils" } - -[features] -snowflake = ["dep:dozer-ingestion-snowflake"] -ethereum = ["dep:dozer-ingestion-ethereum"] -kafka = ["dep:dozer-ingestion-kafka"] -mongodb = ["dep:dozer-ingestion-mongodb"] -datafusion = [ - "dep:dozer-ingestion-deltalake", - "dep:dozer-ingestion-object-store", -] -javascript = ["dep:dozer-ingestion-javascript"] - - -[[bench]] -name = "connectors" -harness = false - -[[bench]] -name = "grpc" -harness = false diff --git a/dozer-ingestion/README.md b/dozer-ingestion/README.md deleted file mode 100644 index 6b5d6decb2..0000000000 --- a/dozer-ingestion/README.md +++ /dev/null @@ -1,89 +0,0 @@ -# Dozer Ingestion - -This module implements several connectors that can act as a source in either real-time or batch fashion. -Each of the connectors implements the [`Connector` trait](https://github.com/getdozer/dozer/blob/main/dozer-ingestion/src/connectors/mod.rs) to support being a source to the data pipeline. - -## New connector implementation - -### Trait - -Every connector to external database needs to implement the `Connector` trait [/dozer-ingestion/src/connectors/mod.rs](https://github.com/getdozer/dozer/blob/main/dozer-ingestion/src/connectors/mod.rs) - -```rust -pub trait Connector: Send + Sync + Debug { - /// Returns all the external types and their corresponding Dozer types. - /// If the external type is not supported, None should be returned. - fn types_mapping() -> Vec<(String, Option)> - where - Self: Sized; - - /// Validates the connector's connection level properties. - async fn validate_connection(&self) -> Result<(), ConnectorError>; - - /// Lists all the table names in the connector. - async fn list_tables(&self) -> Result, ConnectorError>; - - /// Validates the connector's table level properties for each table. - async fn validate_tables(&self, tables: &[TableIdentifier]) -> Result<(), ConnectorError>; - - /// Lists all the column names for each table. - async fn list_columns(&self, tables: Vec) -> Result, ConnectorError>; - - /// Gets the schema for each table. Only requested columns need to be mapped. - /// - /// If this function fails at the connector level, such as a network error, it should return a outer level `Err`. - /// Otherwise the outer level `Ok` should always contain the same number of elements as `table_infos`. - /// - /// If it fails at the table or column level, such as a unsupported data type, one of the elements should be `Err`. - async fn get_schemas( - &self, - table_infos: &[TableInfo], - ) -> Result, ConnectorError>; - - /// Starts outputting data from `tables` to `ingestor`. This method should never return unless there is an unrecoverable error. - async fn start(&self, ingestor: &Ingestor, tables: Vec) -> Result<(), ConnectorError>; -} -``` - -Detailed explanation of the data structures and method contracts can be found in [the specification](./SPEC.md). - -### Connector functions usage in dozer commands - -Dozer uses connector methods in 3 different commands. During `connector ls` execution, dozer just fetches schemas. To get schemas we use `list_tables`, `list_columns` and `get_schemas` methods. - -The other two dozer commands which use connectors are `build` and `run app`. During both command execution, first, we validate the connection and schema using `validate_connection`, `validate_tables` and `get_schemas` methods. After that `run app` command also calls `start` method and the connector starts data ingestion. - -## Source configuration - -The tables and columns that are used in ingestion is defined in `sources` configuration. - -That part of configuration looks like this: - -```yaml - name: users - connection: pg_data_connection - table_name: userdata - columns: - - gender - - email -``` - -From this configuration `table_name` is the table name in the external database and `name` is used in dozer transformations. `connection` is the reference to a connection, which should already be defined in `connections` configuration. The `columns` property is used to filter the columns from the external database. If this value is an empty array, ingestion will fetch all columns of that table. - -### Tables and columns selection - -Every external schema should be mapped to dozer types. The latest types definitions can be found at [https://getdozer.io/docs/reference/data_types](https://getdozer.io/docs/reference/data_types). - -If an external type is not supported, connector should return an error during `get_schemas`. During ingestion data should be mapped the same way as in `get_schemas`. - -### Unit tests - -It is important to have unit tests for schema mapping and data mapping to dozer types. - -More complex tests require connection a real database. Such test cases are expected to have the following: - -* Database infrastructure, preferably created in docker container(s) -* Connection configuration (with placeholders) -* It should be possible to run test cases without any manual modification to the database. - -[Short description of how to run existing tests.](src/tests/README.md) diff --git a/dozer-ingestion/SPEC.md b/dozer-ingestion/SPEC.md deleted file mode 100644 index 1d1c918e45..0000000000 --- a/dozer-ingestion/SPEC.md +++ /dev/null @@ -1,232 +0,0 @@ -# `Connector` specification - -## Methods - -### `types_mapping` - -Returns all the external types and their corresponding Dozer types. - -Returns: - -- `Vec<(String, Option)>` - Vector of tuples of external type name and corresponding Dozer type. If the external type is not supported, `None` should be returned. - -### `validate_connection` - -Validates the connector's connection level properties. - -Returns: - -- `Result<(), ConnectorError>` - `Ok` if the connection can be made and can be used for ingestion, `Err` otherwise. - -### `list_tables` - -Lists all the table names in the connector. - -Returns: - -- `Result, ConnectorError>` - Vector of table identifiers. See [TableIdentifier](#tableidentifier) for more details. - -### `validate_tables` - -Validates the connector's table level properties for each table. - -Arguments: - -- `tables: &[TableIdentifier]` - Vector of tables to validate. - -Returns: - -- `Result<(), ConnectorError>` - `Ok` if all the tables can be used for ingestion, `Err` otherwise. - -This validation does not validate if columns can be mapped to Dozer types. It only validates table level properties like if the table exists. - -### `list_columns` - -Lists all the column names for each table. - -Arguments: - -- `tables: Vec` - Vector of tables to list columns for. - -Returns: - -- `Result, ConnectorError>` - Vector of table info. See [TableInfo](#tableinfo) for more details. - -If the method succeeds, the returned vector must have the same number of elements as the `tables` argument, and the order must be the same. - -### `get_schemas` - -Gets the schema for each table. Only requested columns need to be mapped. - -Arguments: - -- `table_infos: &[TableInfo]` - Vector of table info. See [TableInfo](#tableinfo) for more details. - -Returns: - -- `Result, ConnectorError>` - Vector of source schema results. See [SourceSchemaResult](#sourceschemaresult) for more details. - -The outer level error should be returned upon connection failure. If connection is successful, the inner level error should be returned for each table. - -The returned `Vec` must have the same number of elements as the `table_infos` argument, and the order must be the same. - -### `start` - -Starts outputting data from `tables` to `ingestor`. - -Arguments: - -- `ingestor: &Ingestor` - Ingestor to output data to. -- `tables: Vec` - Vector of table info. See [TableInfo](#tableinfo) for more details. - -Returns: - -- `Result<(), ConnectorError>` - `Ok` if the all data is successfully output, `Err` otherwise. - -See [Ingestor](#ingestor) for contract of the data that Dozer expects. - -## Ingestor - -`Ingestor` is the sender side of a spsc channel. The message is defined in `IngestionMessage`. - -### `IngestionMessage` - -`identifier` must be monotonically increasing for each connector, ordering is defined by `Rust`'s `PartialOrd`. - -```rust -/// Messages that connectors send to Dozer. -pub struct IngestionMessage { - /// The message's identifier, must be unique in a connector. - pub identifier: OpIdentifier, - /// The message kind. - pub kind: IngestionMessageKind, -} -``` - -### `OpIdentifier` - -```rust -/// A identifier made of two `u64`s. -pub struct OpIdentifier { - /// High 64 bits of the identifier. - pub txid: u64, - /// Low 64 bits of the identifier. - pub seq_in_tx: u64, -} -``` - -### `IngestionMessageKind` - -`SnapshottingDone` should be sent from connectors that have distinct snapshot phase and streaming phase, such as postgres. - -For connectors that only streams, such as kafka, `SnapshottingDone` should not be sent. - -```rust -/// All possible kinds of `IngestionMessage`. -pub enum IngestionMessageKind { - /// A CDC event. - OperationEvent(Operation), - /// A connector uses this message kind to notify Dozer that a initial snapshot of the source table is done, - /// and the data is up-to-date until next CDC event. - SnapshottingDone, -} -``` - -### `Operation` - -```rust -/// A CDC event. -pub enum Operation { - Delete { old: Record }, - Insert { new: Record }, - Update { old: Record, new: Record }, -} -``` - -### `Record` - -`schema_id` is how Dozer knows which table this record belongs to. It must be `Some` and the same as the returned `SchemaIdentifier` from `get_schemas` for the table it originates from. - -`values` must be of the same length, order, and type as the `fields` in corresponding `Schema`, expect for `old` in `Delete` and `Update` operation, see [Omitting fields](#omitting-fields-in-delete-and-update-operations). - -```rust -pub struct Record { - /// Schema implemented by this Record - pub schema_id: Option, - /// List of values, following the definitions of `fields` of the associated schema - pub values: Vec, -} -``` - -### Omitting fields in `Delete` and `Update` operations - -If a connector declares a table's `CdcType` to be `OnlyPK`, the connector can omit the fields which is not part of the primary key of the `old` record of `Delete` and `Update` operations. - -However, the `values` must still be of the same length and order as the `fields` in corresponding `Schema`, omitted fields can be filled with `Field::Null`. - -## Data Structures - -### `TableIdentifier` - -```rust -/// Unique identifier of a source table. A source table must have a `name`, optionally under a `schema` scope. -pub struct TableIdentifier { - /// The `schema` scope of the table. - /// - /// Connector that supports schema scope must decide on a default schema, that doesn't must assert that `schema.is_none()`. - pub schema: Option, - /// The table name, must be unique under the `schema` scope, or global scope if `schema` is `None`. - pub name: String, -} -``` - -### `TableInfo` - -```rust -/// `TableIdentifier` with column names. -pub struct TableInfo { - /// The `schema` scope of the table. - pub schema: Option, - /// The table name, must be unique under the `schema` scope, or global scope if `schema` is `None`. - pub name: String, - /// The column names to be mapped. - pub column_names: Vec, -} -``` - -### `SourceSchemaResult` - -```rust -/// Result of mapping one source table schema to Dozer schema. -pub type SourceSchemaResult = Result; -``` - -### `SourceSchema` - -```rust -/// A source table's schema and CDC type. -pub struct SourceSchema { - /// Dozer schema mapped from the source table. Columns are already filtered based on `TableInfo.column_names`. - pub schema: Schema, - /// The source table's CDC type. - pub cdc_type: CdcType, -} -``` - -### `Schema` - -TODO. - -### `CdcType` - -```rust -/// A source table's CDC event type. -pub enum CdcType { - /// Connector gets old record on delete/update operations. - FullChanges, - /// Connector only gets PK of old record on delete/update operations. - OnlyPK, - /// Connector cannot get any info about old records. In other words, the table is append-only. - Nothing, -} -``` diff --git a/dozer-ingestion/benches/connectors.rs b/dozer-ingestion/benches/connectors.rs deleted file mode 100644 index 356965fcbf..0000000000 --- a/dozer-ingestion/benches/connectors.rs +++ /dev/null @@ -1,36 +0,0 @@ -use criterion::{criterion_group, criterion_main, BenchmarkId, Criterion}; -use dozer_ingestion::test_util::create_test_runtime; -use dozer_ingestion_connector::dozer_types::serde_yaml; -use helper::TestConfig; -mod helper; -fn connectors(criter: &mut Criterion) { - let runtime = create_test_runtime(); - let configs = load_test_config(); - - for config in configs { - let mut iterator = helper::get_connection_iterator(runtime.clone(), config.clone()); - let pb = helper::get_progress(); - let mut count = 0; - criter.bench_with_input( - BenchmarkId::new(config.connection.name, config.size), - &config.size, - |b, _| { - b.iter(|| { - iterator.next(); - count += 1; - if count % 100 == 0 { - pb.set_position(count as u64); - } - }) - }, - ); - } -} - -pub fn load_test_config() -> Vec { - let test_config = include_str!("./connectors.sample.yaml"); - serde_yaml::from_str::>(test_config).unwrap() -} - -criterion_group!(benches, connectors); -criterion_main!(benches); diff --git a/dozer-ingestion/benches/connectors.sample.yaml b/dozer-ingestion/benches/connectors.sample.yaml deleted file mode 100644 index a6ca25f486..0000000000 --- a/dozer-ingestion/benches/connectors.sample.yaml +++ /dev/null @@ -1,37 +0,0 @@ -# Runs bench on all connectors in this file -# Copy relevant connectors to dozer-ingestion/benches/connectors.yaml and run bench - -- connection: - config: !Postgres - user: postgres - password: postgres - host: 0.0.0.0 - port: 5437 - database: flights - name: bookings_conn - tables: [bookings] - -- connection: - config: !Grpc - schemas: !Path "workspaces/trips.json" - name: ingest - name: trips_conn - -- connection: - config: !Grpc - schemas: !Path "workspaces/trips.json" - adapter: "arrow" - name: ingest - name: trips_arrow_conn - -- connection: - config: !LocalStorage - details: - path: ./taxi-objectstore - tables: - - !Table - name: trips - config: !Parquet - path: data - extension: .parquet - name: ny_taxi diff --git a/dozer-ingestion/benches/grpc.rs b/dozer-ingestion/benches/grpc.rs deleted file mode 100644 index 04a73f0ce4..0000000000 --- a/dozer-ingestion/benches/grpc.rs +++ /dev/null @@ -1,120 +0,0 @@ -use std::{sync::Arc, thread}; - -use criterion::{criterion_group, criterion_main, BenchmarkId, Criterion}; -use dozer_ingestion::test_util::create_test_runtime; -use dozer_ingestion_connector::dozer_types::{ - arrow::array::{Int32Array, StringArray}, - arrow::{datatypes as arrow_types, record_batch::RecordBatch}, - arrow_types::from_arrow::serialize_record_batch, - grpc_types::ingest::{ingest_service_client::IngestServiceClient, IngestArrowRequest}, - indicatif::{MultiProgress, ProgressBar}, - serde_yaml, - tonic::transport::Channel, -}; -mod helper; -use crate::helper::TestConfig; - -const ARROW_PORT: u32 = 60056; -const BATCH_SIZE: usize = 100; - -fn grpc(criter: &mut Criterion) { - let runtime = create_test_runtime(); - let configs = load_test_config(); - - let multi_pb = MultiProgress::new(); - for config in configs { - let mut iterator = helper::get_connection_iterator(runtime.clone(), config.clone()); - let pb = helper::get_progress(); - pb.set_message("consumer"); - - let pb2 = helper::get_progress(); - pb2.set_message("producer"); - - multi_pb.add(pb.clone()); - multi_pb.add(pb2.clone()); - let mut count = 0; - - // Start ingesting using arrow df - if config.connection.name == "users_arrow" { - runtime.spawn(ingest_arrow(BATCH_SIZE, config.size, pb2)); - } - let mut n_count = 0; - criter.bench_with_input( - BenchmarkId::new(config.connection.name, config.size), - &config.size, - |b, _| { - b.iter(|| { - let r = iterator.next(); - if r.is_none() { - n_count += 1; - } - count += 1; - if count % 100 == 0 { - pb.set_position(count as u64); - } - }) - }, - ); - } -} - -async fn ingest_arrow(batch_size: usize, total: usize, pb: ProgressBar) { - let mut idx = 0; - let schema = arrow_types::Schema::new(vec![ - arrow_types::Field::new("id", arrow_types::DataType::Int32, false), - arrow_types::Field::new("name", arrow_types::DataType::Utf8, false), - ]); - let mut ingest_client = get_grpc_client(ARROW_PORT).await; - while idx < total - batch_size { - let o = idx + batch_size; - - let ids = (idx..o).map(|i| i as i32).collect::>(); - let names = (idx..o) - .map(|i| format!("dario_{i}")) - .collect::>(); - let a = Int32Array::from_iter(ids); - let b = StringArray::from_iter_values(names); - - let record_batch = - RecordBatch::try_new(Arc::new(schema.clone()), vec![Arc::new(a), Arc::new(b)]).unwrap(); - let res = ingest_client - .ingest_arrow(IngestArrowRequest { - schema_name: "users".to_string(), - records: serialize_record_batch(&record_batch), - seq_no: idx as u32, - ..Default::default() - }) - .await; - - if res.is_err() { - break; - } - pb.set_position(idx as u64); - idx += batch_size; - } -} - -async fn get_grpc_client(port: u32) -> IngestServiceClient { - let retries = 10; - let url = format!("http://0.0.0.0:{port}"); - let mut res = IngestServiceClient::connect(url.clone()).await; - for r in 0..retries { - if res.is_ok() { - break; - } - if r == retries - 1 { - panic!("failed to connect after {r} times"); - } - thread::sleep(std::time::Duration::from_millis(300)); - res = IngestServiceClient::connect(url.clone()).await; - } - res.unwrap() -} - -pub fn load_test_config() -> Vec { - let test_config = include_str!("./grpc.yaml"); - serde_yaml::from_str::>(test_config).unwrap() -} - -criterion_group!(benches, grpc); -criterion_main!(benches); diff --git a/dozer-ingestion/benches/grpc.yaml b/dozer-ingestion/benches/grpc.yaml deleted file mode 100644 index 4cc21ec77c..0000000000 --- a/dozer-ingestion/benches/grpc.yaml +++ /dev/null @@ -1,31 +0,0 @@ -- connection: - config: !Grpc - adapter: "arrow" - port: 60056 - schemas: !Inline | - [{ - "name": "users", - "schema": { - "fields": [ - { - "name": "id", - "data_type": "Int32", - "nullable": false, - "dict_id": 0, - "dict_is_ordered": false, - "metadata": {} - }, - { - "name": "name", - "data_type": "Utf8", - "nullable": true, - "dict_id": 0, - "dict_is_ordered": false, - "metadata": {} - } - ], - "metadata": {} - } - }] - name: users_arrow - size: 10000000 diff --git a/dozer-ingestion/benches/helper.rs b/dozer-ingestion/benches/helper.rs deleted file mode 100644 index fdcd3b5b92..0000000000 --- a/dozer-ingestion/benches/helper.rs +++ /dev/null @@ -1,65 +0,0 @@ -use std::sync::Arc; - -use dozer_ingestion::dozer_types::event::EventHub; -use dozer_ingestion_connector::{ - dozer_types::{ - indicatif::{ProgressBar, ProgressStyle}, - log::error, - models::connection::Connection, - serde::{Deserialize, Serialize}, - }, - Connector, IngestionIterator, Ingestor, TableInfo, -}; -use tokio::runtime::Runtime; - -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(crate = "dozer_ingestion_connector::dozer_types::serde")] -pub struct TestConfig { - pub connection: Connection, - pub tables_filter: Option>, - #[serde(default = "default_size")] - pub size: usize, -} -fn default_size() -> usize { - 1000 -} - -pub fn get_progress() -> ProgressBar { - let pb = ProgressBar::new_spinner(); - - pb.set_style( - ProgressStyle::with_template("{spinner:.blue} {msg}: {pos}: {per_sec}") - .unwrap() - // For more spinners check out the cli-spinners project: - // https://github.com/sindresorhus/cli-spinners/blob/master/spinners.json - .tick_strings(&[ - "▹▹▹▹▹", - "▸▹▹▹▹", - "▹▸▹▹▹", - "▹▹▸▹▹", - "▹▹▹▸▹", - "▹▹▹▹▸", - "▪▪▪▪▪", - ]), - ); - pb -} - -pub fn get_connection_iterator(runtime: Arc, config: TestConfig) -> IngestionIterator { - let mut connector = - dozer_ingestion::get_connector(runtime.clone(), EventHub::new(1), config.connection, None) - .unwrap(); - let tables = runtime.block_on(list_tables(&mut *connector)); - let (ingestor, iterator) = Ingestor::initialize_channel(Default::default()); - runtime.clone().spawn_blocking(move || async move { - if let Err(e) = runtime.block_on(connector.start(&ingestor, tables, None)) { - error!("Error starting connector: {:?}", e); - } - }); - iterator -} - -async fn list_tables(connector: &mut dyn Connector) -> Vec { - let tables = connector.list_tables().await.unwrap(); - connector.list_columns(tables).await.unwrap() -} diff --git a/dozer-ingestion/connector/Cargo.toml b/dozer-ingestion/connector/Cargo.toml deleted file mode 100644 index 6f003316c7..0000000000 --- a/dozer-ingestion/connector/Cargo.toml +++ /dev/null @@ -1,12 +0,0 @@ -[package] -name = "dozer-ingestion-connector" -version = "0.4.0" -edition = "2021" -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-types = { path = "../../dozer-types" } -futures = "0.3.28" -tokio = "1.32.0" diff --git a/dozer-ingestion/connector/src/ingestor.rs b/dozer-ingestion/connector/src/ingestor.rs deleted file mode 100644 index 170ee9601e..0000000000 --- a/dozer-ingestion/connector/src/ingestor.rs +++ /dev/null @@ -1,139 +0,0 @@ -use dozer_types::models::ingestion_types::IngestionMessage; -use std::{error::Error, fmt::Display, time::Duration}; -use tokio::{ - sync::mpsc::{channel, Receiver, Sender}, - time::timeout, -}; - -#[derive(Debug, Clone)] -pub struct IngestionConfig { - forwarder_channel_cap: usize, -} - -impl Default for IngestionConfig { - fn default() -> Self { - Self { - forwarder_channel_cap: 100000, - } - } -} - -#[derive(Debug)] -/// `IngestionIterator` is the receiver side of a spsc channel. The sender side is `Ingestor`. -pub struct IngestionIterator { - pub receiver: Receiver, -} - -impl Iterator for IngestionIterator { - type Item = IngestionMessage; - fn next(&mut self) -> Option { - self.receiver.blocking_recv() - } -} - -impl IngestionIterator { - pub async fn next_timeout(&mut self, duration: Duration) -> Option { - timeout(duration, self.receiver.recv()).await.ok().flatten() - } -} - -#[derive(Debug, Clone)] -/// `Ingestor` is the sender side of a spsc channel. The receiver side is `IngestionIterator`. -/// -/// `IngestionMessage` is the message type that is sent over the channel. -pub struct Ingestor { - sender: Sender, -} - -#[derive(Debug, Clone, Copy)] -pub struct SendError; - -impl Display for SendError { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - write!(f, "ingestor receiver dropped") - } -} - -impl Error for SendError {} - -impl Ingestor { - pub fn initialize_channel(config: IngestionConfig) -> (Ingestor, IngestionIterator) { - let (sender, receiver) = channel(config.forwarder_channel_cap); - let ingestor = Self { sender }; - - let iterator = IngestionIterator { receiver }; - (ingestor, iterator) - } - - pub async fn handle_message(&self, message: IngestionMessage) -> Result<(), SendError> { - self.sender.send(message).await.map_err(|_| SendError) - } - - pub fn blocking_handle_message(&self, message: IngestionMessage) -> Result<(), SendError> { - self.sender.blocking_send(message).map_err(|_| SendError) - } - - pub fn is_closed(&self) -> bool { - self.sender.is_closed() - } -} - -#[cfg(test)] -mod tests { - use super::Ingestor; - use dozer_types::models::ingestion_types::{IngestionMessage, TransactionInfo}; - use dozer_types::types::{Operation, Record}; - - #[tokio::test] - async fn test_message_handle() { - let (sender, mut rx) = tokio::sync::mpsc::channel(10); - let ingestor = Ingestor { sender }; - - // Expected seq no - 2 - let operation = Operation::Insert { - new: Record::new(vec![]), - }; - - // Expected seq no - 3 - let operation2 = Operation::Insert { - new: Record::new(vec![]), - }; - - ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index: 0, - op: operation.clone(), - id: None, - }) - .await - .unwrap(); - ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index: 0, - op: operation2.clone(), - id: None, - }) - .await - .unwrap(); - ingestor - .handle_message(IngestionMessage::TransactionInfo( - TransactionInfo::SnapshottingDone { id: None }, - )) - .await - .unwrap(); - - let expected_op_event_message = vec![operation, operation2].into_iter(); - - for op in expected_op_event_message { - let msg = rx.recv().await.unwrap(); - assert_eq!( - IngestionMessage::OperationEvent { - table_index: 0, - op, - id: None - }, - msg - ); - } - } -} diff --git a/dozer-ingestion/connector/src/lib.rs b/dozer-ingestion/connector/src/lib.rs deleted file mode 100644 index fdea56e861..0000000000 --- a/dozer-ingestion/connector/src/lib.rs +++ /dev/null @@ -1,145 +0,0 @@ -use std::fmt::Debug; - -use dozer_types::errors::internal::BoxedError; -use dozer_types::node::OpIdentifier; -use dozer_types::serde; -use dozer_types::serde::{Deserialize, Serialize}; -pub use dozer_types::tonic::async_trait; -use dozer_types::types::{FieldType, Schema}; - -mod ingestor; -pub mod schema_parser; -pub mod test_util; -pub mod utils; - -pub use ingestor::{IngestionConfig, IngestionIterator, Ingestor}; - -pub use dozer_types; -pub use futures; -pub use tokio; - -#[derive(Clone, Copy, Serialize, Deserialize, Debug, Eq, PartialEq, Default)] -#[serde(crate = "dozer_types::serde")] -/// A source table's CDC event type. -pub enum CdcType { - /// Connector gets old record on delete/update operations. - FullChanges, - /// Connector only gets PK of old record on delete/update operations. - OnlyPK, - #[default] - /// Connector cannot get any info about old records. In other words, the table is append-only. - Nothing, -} - -#[derive(Clone, Serialize, Deserialize, Debug, Eq, PartialEq)] -#[serde(crate = "dozer_types::serde")] -/// A source table's schema and CDC type. -pub struct SourceSchema { - /// Dozer schema mapped from the source table. Columns are already filtered based on `TableInfo.column_names`. - pub schema: Schema, - #[serde(default)] - /// The source table's CDC type. - pub cdc_type: CdcType, -} - -impl SourceSchema { - pub fn new(schema: Schema, cdc_type: CdcType) -> Self { - Self { schema, cdc_type } - } -} - -/// Result of mapping one source table schema to Dozer schema. -pub type SourceSchemaResult = Result; - -#[async_trait] -pub trait Connector: Send + Sync + Debug { - /// Returns all the external types and their corresponding Dozer types. - /// If the external type is not supported, None should be returned. - fn types_mapping() -> Vec<(String, Option)> - where - Self: Sized; - - /// Validates the connector's connection level properties. - async fn validate_connection(&mut self) -> Result<(), BoxedError>; - - /// Lists all the table names in the connector. - async fn list_tables(&mut self) -> Result, BoxedError>; - - /// Validates the connector's table level properties for each table. - async fn validate_tables(&mut self, tables: &[TableIdentifier]) -> Result<(), BoxedError>; - - /// Lists all the column names for each table. - async fn list_columns( - &mut self, - tables: Vec, - ) -> Result, BoxedError>; - - /// Gets the schema for each table. Only requested columns need to be mapped. - /// - /// If this function fails at the connector level, such as a network error, it should return a outer level `Err`. - /// Otherwise the outer level `Ok` should always contain the same number of elements as `table_infos`. - /// - /// If it fails at the table or column level, such as a unsupported data type, one of the elements should be `Err`. - async fn get_schemas( - &mut self, - table_infos: &[TableInfo], - ) -> Result, BoxedError>; - - /// Lists all tables and columns and gets the schema for each table. - async fn list_all_schemas( - &mut self, - ) -> Result<(Vec, Vec), BoxedError> { - let tables = self.list_tables().await?; - let table_infos = self.list_columns(tables).await?; - let schemas = self - .get_schemas(&table_infos) - .await? - .into_iter() - .collect::, _>>()?; - Ok((table_infos, schemas)) - } - - /// Serializes any state that's required to re-instantiate this connector. Should not be confused with `last_checkpoint`. - async fn serialize_state(&self) -> Result, BoxedError>; - - /// Starts outputting data from `tables` to `ingestor`. This method should never return unless there is an unrecoverable error. - async fn start( - &mut self, - ingestor: &Ingestor, - tables: Vec, - last_checkpoint: Option, - ) -> Result<(), BoxedError>; -} - -#[derive(Debug, Clone, PartialEq, Eq, Hash)] -/// Unique identifier of a source table. A source table must have a `name`, optionally under a `schema` scope. -pub struct TableIdentifier { - /// The `schema` scope of the table. - /// - /// Connector that supports schema scope must decide on a default schema, that doesn't must assert that `schema.is_none()`. - pub schema: Option, - /// The table name, must be unique under the `schema` scope, or global scope if `schema` is `None`. - pub name: String, -} - -impl TableIdentifier { - pub fn new(schema: Option, name: String) -> Self { - Self { schema, name } - } - - pub fn from_table_name(name: String) -> Self { - Self { schema: None, name } - } -} - -#[derive(Serialize, Deserialize, Clone, Debug, Eq, PartialEq)] -#[serde(crate = "self::serde")] -/// `TableIdentifier` with column names. -pub struct TableInfo { - /// The `schema` scope of the table. - pub schema: Option, - /// The table name, must be unique under the `schema` scope, or global scope if `schema` is `None`. - pub name: String, - /// The column names to be mapped. - pub column_names: Vec, -} diff --git a/dozer-ingestion/connector/src/schema_parser.rs b/dozer-ingestion/connector/src/schema_parser.rs deleted file mode 100644 index b0d801fef2..0000000000 --- a/dozer-ingestion/connector/src/schema_parser.rs +++ /dev/null @@ -1,24 +0,0 @@ -use dozer_types::models::ingestion_types::ConfigSchemas; -use dozer_types::thiserror::{self, Error}; -use std::path::{Path, PathBuf}; - -#[derive(Debug, Error)] -pub enum SchemaParserError { - #[error("cannot read file {0:?}: {1}")] - CannotReadFile(PathBuf, #[source] std::io::Error), -} - -pub struct SchemaParser; - -impl SchemaParser { - pub fn parse_config(schemas: &ConfigSchemas) -> Result { - match schemas { - ConfigSchemas::Inline(schemas_str) => Ok(schemas_str.clone()), - ConfigSchemas::Path(path) => { - let path = Path::new(path); - std::fs::read_to_string(path) - .map_err(|e| SchemaParserError::CannotReadFile(path.to_path_buf(), e)) - } - } - } -} diff --git a/dozer-ingestion/connector/src/test_util.rs b/dozer-ingestion/connector/src/test_util.rs deleted file mode 100644 index 50703ff266..0000000000 --- a/dozer-ingestion/connector/src/test_util.rs +++ /dev/null @@ -1,67 +0,0 @@ -use std::sync::Arc; - -use dozer_types::{ - constants::DEFAULT_CONFIG_PATH, - log::error, - models::{config::Config, connection::ConnectionConfig}, -}; -use futures::stream::{AbortHandle, Abortable}; -use tokio::runtime::Runtime; - -use crate::{Connector, IngestionIterator, Ingestor, TableInfo}; - -pub fn create_test_runtime() -> Arc { - Arc::new( - tokio::runtime::Builder::new_current_thread() - .enable_all() - .build() - .unwrap(), - ) -} - -pub fn spawn_connector( - runtime: Arc, - mut connector: impl Connector + 'static, - tables: Vec, -) -> (IngestionIterator, AbortHandle) { - let (ingestor, iterator) = Ingestor::initialize_channel(Default::default()); - let (abort_handle, abort_registration) = AbortHandle::new_pair(); - runtime.clone().spawn_blocking(move || { - runtime.block_on(async move { - if let Ok(Err(e)) = - Abortable::new(connector.start(&ingestor, tables, None), abort_registration).await - { - error!("Connector `start` returned error: {e}") - } - }) - }); - (iterator, abort_handle) -} - -pub fn spawn_connector_all_tables( - runtime: Arc, - mut connector: impl Connector + 'static, -) -> (IngestionIterator, AbortHandle) { - let tables = runtime.block_on(list_all_table(&mut connector)); - spawn_connector(runtime, connector, tables) -} - -pub fn create_runtime_and_spawn_connector_all_tables( - connector: impl Connector + 'static, -) -> (IngestionIterator, AbortHandle) { - let runtime = create_test_runtime(); - spawn_connector_all_tables(runtime.clone(), connector) -} - -async fn list_all_table(connector: &mut impl Connector) -> Vec { - let tables = connector.list_tables().await.unwrap(); - connector.list_columns(tables).await.unwrap() -} - -pub fn load_test_connection_config() -> ConnectionConfig { - let config_path = std::path::PathBuf::from(format!("src/tests/{DEFAULT_CONFIG_PATH}")); - - let dozer_config = std::fs::read_to_string(config_path).unwrap(); - let mut dozer_config = dozer_types::serde_yaml::from_str::(&dozer_config).unwrap(); - dozer_config.connections.remove(0).config -} diff --git a/dozer-ingestion/connector/src/utils.rs b/dozer-ingestion/connector/src/utils.rs deleted file mode 100644 index ca1f19c9c0..0000000000 --- a/dozer-ingestion/connector/src/utils.rs +++ /dev/null @@ -1,83 +0,0 @@ -use dozer_types::thiserror::Error; - -#[derive(Debug, Clone)] -pub struct ListOrFilterColumns { - pub schema: Option, - pub name: String, - pub columns: Option>, -} - -#[derive(Debug, Error)] -#[error("table not found: {}", table_name(schema.as_deref(), name))] -pub struct TableNotFound { - pub schema: Option, - pub name: String, -} - -fn table_name(schema: Option<&str>, name: &str) -> String { - match schema { - Some(schema) => format!("{}.{}", schema, name), - None => name.to_string(), - } -} - -pub fn warn_dropped_primary_index(table_name: &str) { - dozer_types::log::warn!( - "One or more primary index columns from the source table are \ - not part of the defined schema for table: '{0}'. \ - The primary index will therefore not be present in the Dozer table", - table_name - ); -} - -#[macro_export] -macro_rules! retry_on_network_failure { - ($description:expr, $operation:expr, $network_error_predicate:expr $(, $reconnect:expr)? $(,)?) => - { - $crate::retry_on_network_failure_impl!( - $description, - $operation, - $network_error_predicate, - tokio::time::sleep(RETRY_INTERVAL).await - $(, $reconnect)? - ) - } -} - -#[macro_export] -macro_rules! blocking_retry_on_network_failure { - ($description:expr, $operation:expr, $network_error_predicate:expr $(, $reconnect:expr)? $(,)?) => - { - $crate::retry_on_network_failure_impl!( - $description, - $operation, - $network_error_predicate, - std::thread::sleep(RETRY_INTERVAL) - $(, $reconnect)? - ) - } -} - -#[macro_export] -macro_rules! retry_on_network_failure_impl { - ($description:expr, $operation:expr, $network_error_predicate:expr, $sleep:expr $(, $reconnect:expr)? $(,)?) => { - loop { - match $operation { - ok @ Ok(_) => break ok, - Err(err) => { - if ($network_error_predicate)(&err) { - const RETRY_INTERVAL: std::time::Duration = std::time::Duration::from_secs(5); - dozer_types::log::error!( - "network error during {}: {err:?}. retrying in {RETRY_INTERVAL:?}...", - $description - ); - $sleep; - $($reconnect)? - } else { - break Err(err); - } - } - } - } - }; -} diff --git a/dozer-ingestion/deltalake/Cargo.toml b/dozer-ingestion/deltalake/Cargo.toml deleted file mode 100644 index bef786066a..0000000000 --- a/dozer-ingestion/deltalake/Cargo.toml +++ /dev/null @@ -1,16 +0,0 @@ -[package] -name = "dozer-ingestion-deltalake" -version = "0.4.0" -edition = "2021" -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-ingestion-connector = { path = "../connector" } -dozer-ingestion-object-store = { path = "../object-store" } - -[dependencies.deltalake] -version = "0.17.1" -default-features = false -features = ["datafusion"] diff --git a/dozer-ingestion/deltalake/src/connector.rs b/dozer-ingestion/deltalake/src/connector.rs deleted file mode 100644 index 958af46360..0000000000 --- a/dozer-ingestion/deltalake/src/connector.rs +++ /dev/null @@ -1,126 +0,0 @@ -use crate::reader::DeltaLakeReader; -use crate::schema_helper::SchemaHelper; -use dozer_ingestion_connector::{ - async_trait, - dozer_types::{ - errors::internal::BoxedError, models::ingestion_types::DeltaLakeConfig, node::OpIdentifier, - types::FieldType, - }, - utils::{ListOrFilterColumns, TableNotFound}, - Connector, Ingestor, SourceSchemaResult, TableIdentifier, TableInfo, -}; - -#[derive(Debug)] -pub struct DeltaLakeConnector { - config: DeltaLakeConfig, -} - -impl DeltaLakeConnector { - pub fn new(config: DeltaLakeConfig) -> Self { - Self { config } - } -} - -#[async_trait] -impl Connector for DeltaLakeConnector { - fn types_mapping() -> Vec<(String, Option)> - where - Self: Sized, - { - todo!() - } - - async fn validate_connection(&mut self) -> Result<(), BoxedError> { - Ok(()) - } - - async fn list_tables(&mut self) -> Result, BoxedError> { - Ok(self - .config - .tables - .iter() - .map(|table| TableIdentifier::from_table_name(table.name.clone())) - .collect()) - } - - async fn validate_tables(&mut self, tables: &[TableIdentifier]) -> Result<(), BoxedError> { - let mut delta_table_names = vec![]; - // Collect delta table names in config, the validate table info - for delta_table in self.config.tables.iter() { - delta_table_names.push(delta_table.name.as_str()); - } - for table in tables.iter() { - if !delta_table_names.contains(&table.name.as_str()) || table.schema.is_some() { - return Err(TableNotFound { - schema: table.schema.clone(), - name: table.name.clone(), - } - .into()); - } - } - Ok(()) - } - - async fn list_columns( - &mut self, - tables: Vec, - ) -> Result, BoxedError> { - let table_infos = tables - .into_iter() - .map(|table| ListOrFilterColumns { - schema: table.schema, - name: table.name, - columns: None, - }) - .collect::>(); - let schema_helper = SchemaHelper::new(self.config.clone()); - let source_schemas = schema_helper.get_schemas(&table_infos).await?; - - let mut result = vec![]; - for (source_schema, table_info) in source_schemas.into_iter().zip(table_infos) { - let column_names = source_schema? - .schema - .fields - .into_iter() - .map(|field| field.name) - .collect(); - result.push(TableInfo { - schema: None, - name: table_info.name, - column_names, - }) - } - Ok(result) - } - - async fn get_schemas( - &mut self, - table_infos: &[TableInfo], - ) -> Result, BoxedError> { - let table_infos = table_infos - .iter() - .map(|table_info| ListOrFilterColumns { - schema: None, - name: table_info.name.clone(), - columns: Some(table_info.column_names.clone()), - }) - .collect::>(); - let schema_helper = SchemaHelper::new(self.config.clone()); - schema_helper.get_schemas(&table_infos).await - } - - async fn serialize_state(&self) -> Result, BoxedError> { - Ok(vec![]) - } - - async fn start( - &mut self, - ingestor: &Ingestor, - tables: Vec, - last_checkpoint: Option, - ) -> Result<(), BoxedError> { - assert!(last_checkpoint.is_none()); - let reader = DeltaLakeReader::new(self.config.clone()); - reader.read(&tables, ingestor).await - } -} diff --git a/dozer-ingestion/deltalake/src/lib.rs b/dozer-ingestion/deltalake/src/lib.rs deleted file mode 100644 index a03a7a8997..0000000000 --- a/dozer-ingestion/deltalake/src/lib.rs +++ /dev/null @@ -1,6 +0,0 @@ -mod connector; -mod reader; -mod schema_helper; -mod test; - -pub use connector::DeltaLakeConnector; diff --git a/dozer-ingestion/deltalake/src/reader.rs b/dozer-ingestion/deltalake/src/reader.rs deleted file mode 100644 index d55e066a52..0000000000 --- a/dozer-ingestion/deltalake/src/reader.rs +++ /dev/null @@ -1,92 +0,0 @@ -use std::sync::Arc; - -use deltalake::datafusion::prelude::SessionContext; -use dozer_ingestion_connector::{ - dozer_types::{ - arrow_types::from_arrow::{map_schema_to_dozer, map_value_to_dozer_field}, - errors::internal::BoxedError, - models::ingestion_types::{DeltaLakeConfig, IngestionMessage}, - types::{Operation, Record}, - }, - futures::StreamExt, - tokio, - utils::TableNotFound, - Ingestor, TableInfo, -}; - -pub struct DeltaLakeReader { - config: DeltaLakeConfig, -} - -impl DeltaLakeReader { - pub fn new(config: DeltaLakeConfig) -> Self { - Self { config } - } - - pub async fn read(&self, table: &[TableInfo], ingestor: &Ingestor) -> Result<(), BoxedError> { - for (table_index, table) in table.iter().enumerate() { - self.read_impl(table_index, table, ingestor).await?; - } - Ok(()) - } - - async fn read_impl( - &self, - table_index: usize, - table: &TableInfo, - ingestor: &Ingestor, - ) -> Result<(), BoxedError> { - let table_path = table_path(&self.config, &table.name)?; - let ctx = SessionContext::new(); - let delta_table = deltalake::open_table(table_path).await?; - let cols: Vec<&str> = table.column_names.iter().map(|c| c.as_str()).collect(); - let data = ctx - .read_table(Arc::new(delta_table))? - .select_columns(&cols)? - .execute_stream() - .await?; - - tokio::pin!(data); - while let Some(Ok(batch)) = data.next().await { - let batch_schema = batch.schema(); - let dozer_schema = map_schema_to_dozer(&batch_schema)?; - for row in 0..batch.num_rows() { - let fields = batch - .columns() - .iter() - .enumerate() - .map(|(col, column)| { - map_value_to_dozer_field(column, row, cols[col], &dozer_schema).unwrap() - }) - .collect::>(); - - ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index, - op: Operation::Insert { - new: Record { - values: fields, - lifetime: None, - }, - }, - id: None, - }) - .await - .unwrap(); - } - } - Ok(()) - } -} - -pub fn table_path(config: &DeltaLakeConfig, table_name: &str) -> Result { - for delta_table in config.tables.iter() { - if delta_table.name == table_name { - return Ok(delta_table.path.clone()); - } - } - Err(TableNotFound { - schema: None, - name: table_name.to_string(), - }) -} diff --git a/dozer-ingestion/deltalake/src/schema_helper.rs b/dozer-ingestion/deltalake/src/schema_helper.rs deleted file mode 100644 index cc20d3904a..0000000000 --- a/dozer-ingestion/deltalake/src/schema_helper.rs +++ /dev/null @@ -1,45 +0,0 @@ -use deltalake::{arrow::datatypes::SchemaRef, datafusion::prelude::SessionContext}; -use dozer_ingestion_connector::{ - dozer_types::{errors::internal::BoxedError, models::ingestion_types::DeltaLakeConfig}, - utils::ListOrFilterColumns, - CdcType, SourceSchema, SourceSchemaResult, -}; -use dozer_ingestion_object_store::schema_mapper::map_schema; - -use crate::reader::table_path; -use std::sync::Arc; - -pub struct SchemaHelper { - config: DeltaLakeConfig, -} - -impl SchemaHelper { - pub fn new(config: DeltaLakeConfig) -> Self { - Self { config } - } - - pub async fn get_schemas( - &self, - tables: &[ListOrFilterColumns], - ) -> Result, BoxedError> { - let mut schemas = vec![]; - for table in tables.iter() { - schemas.push(self.get_schemas_impl(table).await); - } - Ok(schemas) - } - - async fn get_schemas_impl( - &self, - table: &ListOrFilterColumns, - ) -> Result { - let table_path = table_path(&self.config, &table.name)?; - let ctx = SessionContext::new(); - let delta_table = deltalake::open_table(table_path).await?; - let arrow_schema: SchemaRef = (*ctx.read_table(Arc::new(delta_table))?.schema()) - .clone() - .into(); - let schema = map_schema(arrow_schema, table)?; - Ok(SourceSchema::new(schema, CdcType::Nothing)) - } -} diff --git a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/.part-00000-04ec9591-0b73-459e-8d18-ba5711d6cbe1-c000.snappy.parquet.crc b/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/.part-00000-04ec9591-0b73-459e-8d18-ba5711d6cbe1-c000.snappy.parquet.crc deleted file mode 100644 index 87694ce3ae..0000000000 Binary files a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/.part-00000-04ec9591-0b73-459e-8d18-ba5711d6cbe1-c000.snappy.parquet.crc and /dev/null differ diff --git a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/.part-00000-c9b90f86-73e6-46c8-93ba-ff6bfaf892a1-c000.snappy.parquet.crc b/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/.part-00000-c9b90f86-73e6-46c8-93ba-ff6bfaf892a1-c000.snappy.parquet.crc deleted file mode 100644 index 35d245353a..0000000000 Binary files a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/.part-00000-c9b90f86-73e6-46c8-93ba-ff6bfaf892a1-c000.snappy.parquet.crc and /dev/null differ diff --git a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/.part-00001-911a94a2-43f6-4acb-8620-5e68c2654989-c000.snappy.parquet.crc b/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/.part-00001-911a94a2-43f6-4acb-8620-5e68c2654989-c000.snappy.parquet.crc deleted file mode 100644 index ec945d35b4..0000000000 Binary files a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/.part-00001-911a94a2-43f6-4acb-8620-5e68c2654989-c000.snappy.parquet.crc and /dev/null differ diff --git a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/_change_data/.gitkeep b/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/_change_data/.gitkeep deleted file mode 100644 index e69de29bb2..0000000000 diff --git a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/_delta_index/.gitkeep b/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/_delta_index/.gitkeep deleted file mode 100644 index e69de29bb2..0000000000 diff --git a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/_delta_log/00000000000000000000.json b/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/_delta_log/00000000000000000000.json deleted file mode 100644 index c52444eae2..0000000000 --- a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/_delta_log/00000000000000000000.json +++ /dev/null @@ -1,5 +0,0 @@ -{"commitInfo":{"timestamp":1615043767813,"operation":"WRITE","operationParameters":{"mode":"Overwrite","partitionBy":"[]"},"isBlindAppend":false,"operationMetrics":{"numFiles":"2","numOutputBytes":"885","numOutputRows":"5"}}} -{"protocol":{"minReaderVersion":1,"minWriterVersion":2}} -{"metaData":{"id":"c48a3abf-ea47-498b-b173-52ce534e8dab","format":{"provider":"parquet","options":{}},"schemaString":"{\"type\":\"struct\",\"fields\":[{\"name\":\"value\",\"type\":\"integer\",\"nullable\":true,\"metadata\":{}}]}","partitionColumns":[],"configuration":{},"createdTime":1615043767476}} -{"add":{"path":"part-00000-c9b90f86-73e6-46c8-93ba-ff6bfaf892a1-c000.snappy.parquet","partitionValues":{},"size":440,"modificationTime":1615043767000,"dataChange":true, "stats":"{\"numRecords\":2,\"nullCount\":{\"value\":0},\"minValues\":{\"value\": 0},\"maxValues\":{\"value\":2}}"}} -{"add":{"path":"part-00001-911a94a2-43f6-4acb-8620-5e68c2654989-c000.snappy.parquet","partitionValues":{},"size":445,"modificationTime":1615043767000,"dataChange":true, "stats":"{\"numRecords\":3,\"nullCount\":{\"value\":0},\"minValues\":{\"value\": 2},\"maxValues\":{\"value\":4}}"}} diff --git a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/_delta_log/00000000000000000001.json b/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/_delta_log/00000000000000000001.json deleted file mode 100644 index a4fb7af570..0000000000 --- a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/_delta_log/00000000000000000001.json +++ /dev/null @@ -1,3 +0,0 @@ -{"commitInfo":{"timestamp":1615043776199,"operation":"DELETE","operationParameters":{"predicate":"[\"(`value` = 3)\"]"},"readVersion":0,"isBlindAppend":false,"operationMetrics":{"numRemovedFiles":"1","numDeletedRows":"1","numAddedFiles":"1","numCopiedRows":"2"}}} -{"remove":{"path":"part-00001-911a94a2-43f6-4acb-8620-5e68c2654989-c000.snappy.parquet","deletionTimestamp":1615043776198,"dataChange":true,"extendedFileMetadata":true,"partitionValues":{},"size":445}} -{"add":{"path":"part-00000-04ec9591-0b73-459e-8d18-ba5711d6cbe1-c000.snappy.parquet","partitionValues":{},"size":440,"modificationTime":1615043776000,"dataChange":true,"stats":"{\"numRecords\":2,\"nullCount\":{\"value\":0},\"minValues\":{\"value\": 2},\"maxValues\":{\"value\":4}}"}} diff --git a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/part-00000-04ec9591-0b73-459e-8d18-ba5711d6cbe1-c000.snappy.parquet b/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/part-00000-04ec9591-0b73-459e-8d18-ba5711d6cbe1-c000.snappy.parquet deleted file mode 100644 index 67d45baa2f..0000000000 Binary files a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/part-00000-04ec9591-0b73-459e-8d18-ba5711d6cbe1-c000.snappy.parquet and /dev/null differ diff --git a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/part-00000-c9b90f86-73e6-46c8-93ba-ff6bfaf892a1-c000.snappy.parquet b/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/part-00000-c9b90f86-73e6-46c8-93ba-ff6bfaf892a1-c000.snappy.parquet deleted file mode 100644 index a50604bfef..0000000000 Binary files a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/part-00000-c9b90f86-73e6-46c8-93ba-ff6bfaf892a1-c000.snappy.parquet and /dev/null differ diff --git a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/part-00001-911a94a2-43f6-4acb-8620-5e68c2654989-c000.snappy.parquet b/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/part-00001-911a94a2-43f6-4acb-8620-5e68c2654989-c000.snappy.parquet deleted file mode 100644 index fa54bd1848..0000000000 Binary files a/dozer-ingestion/deltalake/src/test/data/delta-0.8.0/part-00001-911a94a2-43f6-4acb-8620-5e68c2654989-c000.snappy.parquet and /dev/null differ diff --git a/dozer-ingestion/deltalake/src/test/deltalake_test.rs b/dozer-ingestion/deltalake/src/test/deltalake_test.rs deleted file mode 100644 index 164015809e..0000000000 --- a/dozer-ingestion/deltalake/src/test/deltalake_test.rs +++ /dev/null @@ -1,61 +0,0 @@ -use crate::DeltaLakeConnector; -use dozer_ingestion_connector::{ - dozer_types::{ - models::ingestion_types::{DeltaLakeConfig, DeltaTable, IngestionMessage}, - types::{Field, FieldType, Operation, SourceDefinition}, - }, - test_util::create_runtime_and_spawn_connector_all_tables, - tokio, Connector, -}; - -#[tokio::test] -async fn get_schema_from_deltalake() { - let path = "src/test/data/delta-0.8.0"; - let table_name = "test_table"; - let delta_table = DeltaTable { - path: path.to_string(), - name: table_name.to_string(), - }; - let config = DeltaLakeConfig { - tables: vec![delta_table], - }; - - let mut connector = DeltaLakeConnector::new(config); - let (_, schemas) = connector.list_all_schemas().await.unwrap(); - let field = schemas[0].schema.fields[0].clone(); - assert_eq!(&field.name, "value"); - assert_eq!(field.typ, FieldType::Int); - assert!(field.nullable); - assert_eq!(field.source, SourceDefinition::Dynamic); -} - -#[test] -fn read_deltalake() { - let path = "src/test/data/delta-0.8.0"; - let table_name = "test_table"; - let delta_table = DeltaTable { - path: path.to_string(), - name: table_name.to_string(), - }; - let config = DeltaLakeConfig { - tables: vec![delta_table], - }; - - let connector = DeltaLakeConnector::new(config); - - let (iterator, _) = create_runtime_and_spawn_connector_all_tables(connector); - - let fields = vec![Field::Int(0), Field::Int(1), Field::Int(2), Field::Int(4)]; - let mut values = vec![]; - for message in iterator { - if let IngestionMessage::OperationEvent { - op: Operation::Insert { new }, - .. - } = message - { - values.extend(new.values); - } - } - values.sort(); - assert_eq!(fields, values); -} diff --git a/dozer-ingestion/deltalake/src/test/mod.rs b/dozer-ingestion/deltalake/src/test/mod.rs deleted file mode 100644 index ad170fd2a9..0000000000 --- a/dozer-ingestion/deltalake/src/test/mod.rs +++ /dev/null @@ -1,2 +0,0 @@ -#[cfg(test)] -mod deltalake_test; diff --git a/dozer-ingestion/ethereum/Cargo.toml b/dozer-ingestion/ethereum/Cargo.toml deleted file mode 100644 index dd42dca0f0..0000000000 --- a/dozer-ingestion/ethereum/Cargo.toml +++ /dev/null @@ -1,13 +0,0 @@ -[package] -name = "dozer-ingestion-ethereum" -version = "0.4.0" -edition = "2021" -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-ingestion-connector = { path = "../connector" } -web3 = "0.19.0" -hex-literal = "0.4.1" -dozer-tracing = { path = "../../dozer-tracing" } diff --git a/dozer-ingestion/ethereum/src/README.md b/dozer-ingestion/ethereum/src/README.md deleted file mode 100644 index 77efbb7644..0000000000 --- a/dozer-ingestion/ethereum/src/README.md +++ /dev/null @@ -1,12 +0,0 @@ -``` -docker run -d --name ethereum-node -v "$PWD":/root \ - -p 8545:8545 -p 30303:30303 \ - ethereum/client-go --http.addr 0.0.0.0 --gcmode archive --syncmode full --txlookuplimit 0 - - -docker run -v "$PWD":/root \ - -p 8545:8545 -p 30303:30303 -p 3334:3334 -p 8551:8551 \ - ethereum/client-go --http.addr 0.0.0.0 --gcmode archive --syncmode full --txlookuplimit 0 --ws --ws.port 3334 --ws.api eth,net,web3 --ws.origins '*' - - -``` \ No newline at end of file diff --git a/dozer-ingestion/ethereum/src/helper.rs b/dozer-ingestion/ethereum/src/helper.rs deleted file mode 100644 index c71fc6b330..0000000000 --- a/dozer-ingestion/ethereum/src/helper.rs +++ /dev/null @@ -1,27 +0,0 @@ -use web3::transports::{Batch, Http, WebSocket}; - -pub async fn get_wss_client(url: &str) -> Result, web3::Error> { - Ok(web3::Web3::new( - web3::transports::WebSocket::new(url).await?, - )) -} - -pub async fn get_batch_wss_client( - url: &str, -) -> Result<(web3::Web3>, WebSocket), web3::Error> { - let transport = web3::transports::WebSocket::new(url).await?; - Ok(( - web3::Web3::new(web3::transports::Batch::new(transport.clone())), - transport, - )) -} - -pub async fn get_batch_http_client( - url: &str, -) -> Result<(web3::Web3>, Http), web3::Error> { - let transport = web3::transports::Http::new(url)?; - Ok(( - web3::Web3::new(web3::transports::Batch::new(transport.clone())), - transport, - )) -} diff --git a/dozer-ingestion/ethereum/src/lib.rs b/dozer-ingestion/ethereum/src/lib.rs deleted file mode 100644 index 99d39eea43..0000000000 --- a/dozer-ingestion/ethereum/src/lib.rs +++ /dev/null @@ -1,5 +0,0 @@ -pub mod helper; -mod log; -mod trace; -pub use log::EthLogConnector; -pub use trace::EthTraceConnector; diff --git a/dozer-ingestion/ethereum/src/log/connector.rs b/dozer-ingestion/ethereum/src/log/connector.rs deleted file mode 100644 index e41661f82d..0000000000 --- a/dozer-ingestion/ethereum/src/log/connector.rs +++ /dev/null @@ -1,262 +0,0 @@ -use std::collections::HashMap; -use std::{str::FromStr, sync::Arc}; - -use super::helper; -use super::sender::{run, EthDetails}; -use dozer_ingestion_connector::dozer_types::node::OpIdentifier; -use dozer_ingestion_connector::utils::TableNotFound; -use dozer_ingestion_connector::{ - async_trait, - dozer_types::{ - errors::internal::BoxedError, - log::warn, - models::ingestion_types::{EthFilter, EthLogConfig}, - serde_json, - types::FieldType, - }, - CdcType, Connector, Ingestor, SourceSchema, SourceSchemaResult, TableIdentifier, TableInfo, -}; -use web3::ethabi::{Contract, Event}; -use web3::types::{Address, BlockNumber, Filter, FilterBuilder, H256, U64}; - -#[derive(Debug)] -pub struct EthLogConnector { - config: EthLogConfig, - // Address -> (contract, contract_name) - contracts: HashMap, - // contract_signacture -> SchemaID - schema_map: HashMap, - conn_name: String, -} - -#[derive(Debug, Clone)] -// (Contract, Name) -pub struct ContractTuple(pub Contract, pub String); - -pub const ETH_LOGS_TABLE: &str = "eth_logs"; -impl EthLogConnector { - pub fn build_filter(filter: &EthFilter) -> Filter { - let builder = FilterBuilder::default(); - - // Optionally add a from_block filter - let builder = match filter.from_block { - Some(block_no) => builder.from_block(BlockNumber::Number(U64::from(block_no))), - None => builder, - }; - // Optionally add a to_block filter - let builder = match filter.to_block { - Some(block_no) => builder.to_block(BlockNumber::Number(U64::from(block_no))), - None => builder, - }; - - // Optionally Add Address filter - let builder = match filter.addresses.is_empty() { - false => { - let addresses = filter - .addresses - .iter() - .map(|a| Address::from_str(a).unwrap()) - .collect(); - builder.address(addresses) - } - true => builder, - }; - - // Optionally add topics - let builder = match filter.topics.is_empty() { - false => { - let topics: Vec> = filter - .topics - .iter() - .map(|t| vec![H256::from_str(t).unwrap()]) - .collect(); - builder.topics( - topics.first().cloned(), - topics.get(1).cloned(), - topics.get(2).cloned(), - topics.get(3).cloned(), - ) - } - true => builder, - }; - - builder.build() - } - - pub fn new(config: EthLogConfig, conn_name: String) -> Self { - let mut contracts = HashMap::new(); - - for c in &config.contracts { - let contract = serde_json::from_str(&c.abi).expect("unable to parse contract from abi"); - contracts.insert( - c.address.to_string().to_lowercase(), - ContractTuple(contract, c.name.to_string()), - ); - } - - let schema_map = Self::build_schema_map(&contracts); - Self { - config, - contracts, - schema_map, - conn_name, - } - } - - fn build_schema_map(contracts: &HashMap) -> HashMap { - let mut schema_map = HashMap::new(); - - let mut signatures = vec![]; - for contract_tuple in contracts.values() { - let contract = contract_tuple.0.clone(); - let events: Vec<&Event> = contract.events.values().flatten().collect(); - for evt in events { - signatures.push(evt.signature()); - } - } - signatures.sort(); - - for (idx, signature) in signatures.iter().enumerate() { - schema_map.insert(signature.to_owned(), 2 + idx); - } - schema_map - } -} - -#[async_trait] -impl Connector for EthLogConnector { - fn types_mapping() -> Vec<(String, Option)> - where - Self: Sized, - { - todo!() - } - - async fn validate_connection(&mut self) -> Result<(), BoxedError> { - // Return contract parsing error - for contract in &self.config.contracts { - serde_json::from_str(&contract.abi)?; - } - Ok(()) - } - - async fn list_tables(&mut self) -> Result, BoxedError> { - let event_schema_names = helper::get_contract_event_schemas(&self.contracts) - .into_iter() - .map(|(name, _)| TableIdentifier::from_table_name(name)); - let mut result = vec![TableIdentifier::from_table_name(ETH_LOGS_TABLE.to_string())]; - result.extend(event_schema_names); - Ok(result) - } - - async fn validate_tables(&mut self, tables: &[TableIdentifier]) -> Result<(), BoxedError> { - let existing_tables = self.list_tables().await?; - for table in tables { - if !existing_tables.contains(table) || table.schema.is_some() { - return Err(TableNotFound { - schema: table.schema.clone(), - name: table.name.clone(), - } - .into()); - } - } - Ok(()) - } - - async fn list_columns( - &mut self, - tables: Vec, - ) -> Result, BoxedError> { - let event_schemas = helper::get_contract_event_schemas(&self.contracts); - let mut result = vec![]; - for table in tables { - let column_names = if table.name == ETH_LOGS_TABLE && table.schema.is_none() { - helper::get_eth_schema() - .fields - .into_iter() - .map(|field| field.name) - .collect() - } else if let Some((_, schema)) = event_schemas - .iter() - .find(|(name, _)| name == &table.name && table.schema.is_none()) - { - schema - .schema - .fields - .iter() - .map(|field| field.name.clone()) - .collect() - } else { - return Err(TableNotFound { - schema: table.schema.clone(), - name: table.name.clone(), - } - .into()); - }; - result.push(TableInfo { - schema: table.schema, - name: table.name, - column_names, - }) - } - Ok(result) - } - - async fn get_schemas( - &mut self, - table_infos: &[TableInfo], - ) -> Result, BoxedError> { - let mut schemas = vec![( - ETH_LOGS_TABLE.to_string(), - SourceSchema::new(helper::get_eth_schema(), CdcType::Nothing), - )]; - - let event_schemas = helper::get_contract_event_schemas(&self.contracts); - schemas.extend(event_schemas); - - let mut result = vec![]; - for table in table_infos { - if let Some((_, schema)) = schemas - .iter() - .find(|(name, _)| name == &table.name && table.schema.is_none()) - { - warn!("TODO: filter columns"); - result.push(Ok(schema.clone())); - } else { - result.push(Err(TableNotFound { - schema: table.schema.clone(), - name: table.name.clone(), - } - .into())); - } - } - - Ok(result) - } - - async fn serialize_state(&self) -> Result, BoxedError> { - Ok(vec![]) - } - - async fn start( - &mut self, - ingestor: &Ingestor, - tables: Vec, - _last_checkpoint: Option, - ) -> Result<(), BoxedError> { - // Start a new thread that interfaces with ETH node - let wss_url = self.config.wss_url.to_owned(); - let filter = self.config.filter.to_owned().unwrap_or_default(); - - let details = Arc::new(EthDetails::new( - wss_url, - filter, - ingestor, - self.contracts.to_owned(), - tables, - self.schema_map.to_owned(), - self.conn_name.clone(), - )); - run(details).await - } -} diff --git a/dozer-ingestion/ethereum/src/log/helper.rs b/dozer-ingestion/ethereum/src/log/helper.rs deleted file mode 100644 index 609d49df6c..0000000000 --- a/dozer-ingestion/ethereum/src/log/helper.rs +++ /dev/null @@ -1,318 +0,0 @@ -use std::collections::HashMap; -use std::sync::Arc; - -use dozer_ingestion_connector::dozer_types::log::error; -use dozer_ingestion_connector::dozer_types::types::{ - Field, FieldDefinition, FieldType, Operation, Record, Schema, SourceDefinition, -}; -use dozer_ingestion_connector::{CdcType, SourceSchema, TableInfo}; -use web3::ethabi::RawLog; -use web3::types::Log; - -use super::connector::{ContractTuple, ETH_LOGS_TABLE}; -use super::sender::EthDetails; - -pub fn get_contract_event_schemas( - contracts: &HashMap, -) -> Vec<(String, SourceSchema)> { - let mut schemas = vec![]; - - for contract_tuple in contracts.values() { - for event in contract_tuple.0.events.values().flatten() { - let mut fields = vec![]; - for input in event.inputs.iter().cloned() { - fields.push(FieldDefinition { - name: input.name, - typ: match input.kind { - web3::ethabi::ParamType::Address => FieldType::String, - web3::ethabi::ParamType::Bytes => FieldType::Binary, - web3::ethabi::ParamType::FixedBytes(_) => FieldType::Binary, - web3::ethabi::ParamType::Int(_) => FieldType::UInt, - web3::ethabi::ParamType::Uint(_) => FieldType::UInt, - web3::ethabi::ParamType::Bool => FieldType::Boolean, - web3::ethabi::ParamType::String => FieldType::String, - // TODO: These are to be mapped to appropriate types - web3::ethabi::ParamType::Array(_) - | web3::ethabi::ParamType::FixedArray(_, _) - | web3::ethabi::ParamType::Tuple(_) => FieldType::Text, - }, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }); - } - - schemas.push(( - get_table_name(contract_tuple, &event.name), - SourceSchema::new( - Schema { - fields, - primary_index: vec![], - }, - CdcType::Nothing, - ), - )); - } - } - - schemas -} - -pub fn decode_event( - log: Log, - contracts: HashMap, - tables: Vec, -) -> Option<(usize, Operation)> { - let address = format!("{:?}", log.address); - - let mut c = contracts.get(&address); - - if c.is_none() { - // match on wildcard - let wild_card_contract = contracts.iter().find(|(k, _)| k.to_string() == *"*"); - c = wild_card_contract.map(|c| c.1); - } - if let Some(contract_tuple) = c { - // Topics 0, 1, 2 should be name, buyer, seller in most cases - let name = log - .topics - .first() - .expect("name is expected") - .to_owned() - .to_string(); - let opt_event = contract_tuple - .0 - .events - .values() - .flatten() - .find(|evt| evt.signature().to_string() == name); - - if let Some(event) = opt_event { - let table_name = get_table_name(contract_tuple, &event.name); - let table_index = tables.iter().position(|t| t.name == table_name); - if let Some(table_index) = table_index { - let parsed_event = event.parse_log(RawLog { - topics: log.topics, - data: log.data.0, - }); - - match parsed_event { - Ok(parsed_event) => { - let values = parsed_event - .params - .into_iter() - .map(|p| map_abitype_to_field(p.value)) - .collect(); - return Some(( - table_index, - Operation::Insert { - new: Record { - values, - lifetime: None, - }, - }, - )); - } - Err(_) => { - error!( - "parsing event failed: block_no: {}, txn_hash: {}. Have you included the right abi to address mapping ?", - log.block_number.unwrap(), - log.transaction_hash.unwrap() - ); - return None; - } - } - } - } - } - - None -} - -pub fn get_table_name(contract_tuple: &ContractTuple, event_name: &str) -> String { - format!("{}_{}", contract_tuple.1, event_name) -} - -pub fn map_abitype_to_field(f: web3::ethabi::Token) -> Field { - match f { - web3::ethabi::Token::Address(f) => Field::String(format!("{f:?}")), - web3::ethabi::Token::FixedBytes(f) => Field::Binary(f), - web3::ethabi::Token::Bytes(f) => Field::Binary(f), - // TODO: Convert i64 appropriately - web3::ethabi::Token::Int(f) => Field::UInt(f.low_u64()), - web3::ethabi::Token::Uint(f) => Field::UInt(f.low_u64()), - web3::ethabi::Token::Bool(f) => Field::Boolean(f), - web3::ethabi::Token::String(f) => Field::String(f), - web3::ethabi::Token::FixedArray(f) - | web3::ethabi::Token::Array(f) - | web3::ethabi::Token::Tuple(f) => Field::Text( - f.iter() - .map(|f| f.to_string()) - .collect::>() - .join(","), - ), - } -} -pub fn map_log_to_event(log: Log, details: Arc) -> Option<(usize, Operation)> { - // Check if table is requested - let table_index = details.tables.iter().position(|t| t.name == ETH_LOGS_TABLE); - - if let Some(table_index) = table_index { - if log.log_index.is_some() { - let values = map_log_to_values(log); - Some(( - table_index, - Operation::Insert { - new: Record { - values, - lifetime: None, - }, - }, - )) - } else { - None - } - } else { - None - } -} - -pub fn get_id(log: &Log) -> u64 { - let block_no = log - .block_number - .expect("expected for non pendning") - .as_u64(); - - let log_idx = log.log_index.expect("expected for non pendning").as_u64(); - - block_no * 100_000 + log_idx * 2 -} -pub fn map_log_to_values(log: Log) -> Vec { - let block_no = log.block_number.expect("expected for non pending").as_u64(); - let txn_idx = log - .transaction_index - .expect("expected for non pending") - .as_u64(); - let log_idx = log.log_index.expect("expected for non pending").as_u64(); - - let idx = get_id(&log); - - let values = vec![ - Field::UInt(idx), - Field::String(format!("{:?}", log.address)), - Field::Text( - log.topics - .iter() - .map(|t| t.to_string()) - .collect::>() - .join(" "), - ), - Field::Binary(log.data.0), - log.block_hash - .map_or(Field::Null, |f| Field::String(f.to_string())), - Field::UInt(block_no), - log.transaction_hash - .map_or(Field::Null, |f| Field::String(f.to_string())), - Field::UInt(txn_idx), - Field::UInt(log_idx), - log.transaction_log_index - .map_or(Field::Null, |f| Field::Int(f.try_into().unwrap())), - log.log_type.map_or(Field::Null, Field::String), - log.removed.map_or(Field::Null, Field::Boolean), - ]; - - values -} - -pub fn get_eth_schema() -> Schema { - Schema { - fields: vec![ - FieldDefinition { - name: "id".to_string(), - typ: FieldType::UInt, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "address".to_string(), - typ: FieldType::String, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "topics".to_string(), - typ: FieldType::String, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "data".to_string(), - typ: FieldType::Binary, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "block_hash".to_string(), - typ: FieldType::String, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "block_number".to_string(), - typ: FieldType::UInt, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "transaction_hash".to_string(), - typ: FieldType::String, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "transaction_index".to_string(), - typ: FieldType::Int, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "log_index".to_string(), - typ: FieldType::Int, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "transaction_log_index".to_string(), - typ: FieldType::Int, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "log_type".to_string(), - typ: FieldType::String, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "removed".to_string(), - typ: FieldType::Boolean, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }, - ], - - primary_index: vec![0], - } -} diff --git a/dozer-ingestion/ethereum/src/log/mod.rs b/dozer-ingestion/ethereum/src/log/mod.rs deleted file mode 100644 index 84ee673377..0000000000 --- a/dozer-ingestion/ethereum/src/log/mod.rs +++ /dev/null @@ -1,17 +0,0 @@ -mod connector; -mod helper; -mod sender; -pub use connector::EthLogConnector; -use dozer_ingestion_connector::dozer_types::thiserror::{self, Error}; - -#[cfg(test)] -mod tests; - -#[derive(Debug, Error)] -enum Error { - #[error("Failed fetching after {0} recursions")] - EthTooManyRecurisions(usize), - - #[error("Received empty message in connector")] - EmptyMessage, -} diff --git a/dozer-ingestion/ethereum/src/log/sender.rs b/dozer-ingestion/ethereum/src/log/sender.rs deleted file mode 100644 index bb922acb7d..0000000000 --- a/dozer-ingestion/ethereum/src/log/sender.rs +++ /dev/null @@ -1,255 +0,0 @@ -use core::time; -use std::collections::HashMap; -use std::sync::Arc; - -use dozer_ingestion_connector::dozer_types::errors::internal::BoxedError; -use dozer_ingestion_connector::dozer_types::log::{debug, info, trace, warn}; -use dozer_ingestion_connector::dozer_types::models::ingestion_types::{ - EthFilter, IngestionMessage, -}; -use dozer_ingestion_connector::futures::future::BoxFuture; -use dozer_ingestion_connector::futures::{FutureExt, StreamExt}; -use dozer_ingestion_connector::{tokio, Ingestor, TableInfo}; -use web3::transports::WebSocket; -use web3::types::{Log, H256}; -use web3::Web3; - -use crate::log::Error; -use crate::{helper as conn_helper, EthLogConnector}; - -use super::connector::ContractTuple; -use super::helper; - -const MAX_RETRIES: usize = 3; - -pub struct EthDetails<'a> { - wss_url: String, - filter: EthFilter, - ingestor: &'a Ingestor, - contracts: HashMap, - pub tables: Vec, - pub schema_map: HashMap, - pub conn_name: String, -} - -impl<'a> EthDetails<'a> { - #[allow(clippy::too_many_arguments)] - pub fn new( - wss_url: String, - filter: EthFilter, - ingestor: &'a Ingestor, - contracts: HashMap, - tables: Vec, - schema_map: HashMap, - conn_name: String, - ) -> Self { - EthDetails { - wss_url, - filter, - ingestor, - contracts, - tables, - schema_map, - conn_name, - } - } -} - -pub async fn run(details: Arc>) -> Result<(), BoxedError> { - let client = conn_helper::get_wss_client(&details.wss_url).await?; - - // Get current block no. - let latest_block_no = client.eth().block_number().await?.as_u64(); - - let block_end = match details.filter.to_block { - None => latest_block_no, - Some(block_no) => block_no, - }; - - // Default to current block if from_block is not specified - let block_start = details.filter.from_block.unwrap_or(block_end); - - fetch_logs( - details.clone(), - client.clone(), - block_start, - block_end, - 0, - MAX_RETRIES, - ) - .await?; - - let changes_handler_filter = match details.filter.to_block { - None => { - // Create a filter from the last block to check for changes - let mut filter = details.filter.clone(); - filter.from_block = Some(latest_block_no); - Some(filter) - } - Some(block_end) => { - if block_end > latest_block_no { - // Create a filter from the last block to defined end of blocks. - // It can be used to fetch future records with limiting it by `block_to` - - let mut filter = details.filter.clone(); - filter.from_block = Some(latest_block_no); - filter.to_block = Some(block_end); - Some(filter) - } else { - None - } - } - }; - - if let Some(filter) = changes_handler_filter { - debug!( - "[{}] Fetching from block ..: {}", - details.conn_name, block_end - ); - - let filter = client - .eth_filter() - .create_logs_filter(EthLogConnector::build_filter(&filter)) - .await?; - - let stream = filter.stream(time::Duration::from_secs(1)); - - tokio::pin!(stream); - - loop { - let msg = stream.next().await; - - let msg = msg.ok_or(Error::EmptyMessage)??; - - process_log(details.clone(), msg).await; - } - } else { - info!("[{}] Reading reached block_to limit", details.conn_name); - } - Ok(()) -} - -pub fn fetch_logs( - details: Arc, - client: Web3, - block_start: u64, - block_end: u64, - depth: usize, - retries_left: usize, -) -> BoxFuture<'_, Result<(), BoxedError>> { - let filter = details.filter.clone(); - let depth_str = (0..depth) - .map(|_| " ".to_string()) - .collect::>() - .join(""); - async move { - let mut applied_filter = filter.clone(); - applied_filter.from_block = Some(block_start); - applied_filter.to_block = Some(block_end); - let res = client.eth().logs(EthLogConnector::build_filter(&applied_filter)).await; - - match res { - Ok(logs) => { - debug!("[{}] {} Fetched: {} , block_start: {},block_end: {}, depth: {}", details.conn_name, depth_str, logs.len(), block_start, block_end, depth); - for msg in logs { - process_log( - details.clone(), - msg, - ).await; - } - Ok(()) - }, - Err(e) => match e { - web3::Error::Rpc(rpc_error) => { - // Infura returns a RpcError if the no of records are more than 10000 - // { code: ServerError(-32005), message: "query returned more than 10000 results", data: None } - // break it down into half on each error and exit after 10 errors in a specific branch - if rpc_error.code.code() == -32005 { - debug!("[{}] {} More than 10000 records, block_start: {},block_end: {}, depth: {}", details.conn_name, depth_str, block_start, block_end, depth); - if depth > 100 { - Err(Error::EthTooManyRecurisions(depth).into()) - } else { - let middle = (block_start + block_end) / 2; - debug!("[{}] {} Splitting in two calls block_start: {}, middle: {}, block_end: {}", details.conn_name, depth_str,block_start, block_end, middle); - fetch_logs( - details.clone(), - client.clone(), - block_start, - middle, - depth + 1, - MAX_RETRIES - ) - .await?; - - fetch_logs( - details, - client.clone(), - middle + 1, - block_end, - depth + 1, - MAX_RETRIES - ) - .await?; - Ok(()) - } - } else { - Err(rpc_error.into()) - } - } - e => { - if retries_left == 0 { - Err(e.into()) - } else { - warn!("[{}] Retrying to fetch logs", details.conn_name); - fetch_logs(details, client, block_start, block_end, depth, retries_left - 1).await?; - Ok(()) - } - }, - }, - } - } - .boxed() -} - -async fn process_log(details: Arc>, msg: Log) { - // Filter pending logs. log.log_index is None for pending State - if msg.log_index.is_some() { - if let Some((table_index, op)) = helper::map_log_to_event(msg.to_owned(), details.clone()) { - trace!("Writing log : {:?}", op); - // Write eth_log record - if details - .ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index, - op, - id: None, - }) - .await - .is_err() - { - // If receiving end is closed, exit - return; - } - } else { - trace!("Ignoring log : {:?}", msg); - } - - // write event record optionally - - let op = helper::decode_event(msg, details.contracts.to_owned(), details.tables.clone()); - if let Some((table_index, op)) = op { - trace!("Writing event : {:?}", op); - // if receiving end is closed, ignore - let _ = details - .ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index, - op, - id: None, - }) - .await; - } else { - trace!("Writing event : {:?}", op); - } - } -} diff --git a/dozer-ingestion/ethereum/src/log/tests/Dockerfile b/dozer-ingestion/ethereum/src/log/tests/Dockerfile deleted file mode 100644 index 2dd177721d..0000000000 --- a/dozer-ingestion/ethereum/src/log/tests/Dockerfile +++ /dev/null @@ -1,13 +0,0 @@ -# node:alpine will be our base image to create this image -FROM node:alpine -# Set the /app directory as working directory -WORKDIR /app - -RUN apk --no-cache add curl - -# Install ganache-cli globally -RUN npm install -g ganache -# Running as an archive node from the block specifiec in the dozer-config.yaml -# https://docs.infura.io/infura/tutorials/ethereum/fork-ethereum-with-ganache - -CMD ["ganache", "-h", "0.0.0.0", "-m", "marriage tiny prepare canyon grape half kingdom guide desert surge three nurse"] diff --git a/dozer-ingestion/ethereum/src/log/tests/connector.rs b/dozer-ingestion/ethereum/src/log/tests/connector.rs deleted file mode 100644 index 60951b6d17..0000000000 --- a/dozer-ingestion/ethereum/src/log/tests/connector.rs +++ /dev/null @@ -1,37 +0,0 @@ -use dozer_ingestion_connector::{ - dozer_types::types::{Field, Operation}, - test_util::create_test_runtime, -}; -use hex_literal::hex; - -use super::helper::run_eth_sample; - -#[test] -#[ignore] -fn test_eth_iterator() { - let runtime = create_test_runtime(); - let wss_url = "ws://localhost:8545".to_string(); - let my_account = hex!("b49B3BEE604eF76410E84C7C98bC20335FdA0f75").into(); - - let validate = |op: &Operation, idx: usize, field: Option| { - if let Operation::Insert { new } = op { - assert_eq!(new.values.get(idx), field.as_ref()); - } else { - panic!("expected insert"); - } - }; - - let (contract, msgs) = - runtime - .clone() - .block_on(run_eth_sample(runtime.clone(), wss_url, my_account)); - - let address = format!("{:?}", contract.address()); - validate(&msgs[0], 1, Some(Field::String(address))); - - validate(&msgs[1], 0, Some(Field::String(format!("{my_account:?}")))); - - validate(&msgs[1], 1, Some(Field::String("Hello World!".to_string()))); - - validate(&msgs[2], 0, None); -} diff --git a/dozer-ingestion/ethereum/src/log/tests/contracts/CustomEvent.code b/dozer-ingestion/ethereum/src/log/tests/contracts/CustomEvent.code deleted file mode 100644 index 1265e48979..0000000000 --- a/dozer-ingestion/ethereum/src/log/tests/contracts/CustomEvent.code +++ /dev/null @@ -1 +0,0 @@ -608060405234801561001057600080fd5b50610167806100206000396000f3fe608060405234801561001057600080fd5b506004361061002b5760003560e01c8063f8a8fd6d14610030575b600080fd5b61003861003a565b005b3373ffffffffffffffffffffffffffffffffffffffff167f0738f4da267a110d810e6e89fc59e46be6de0c37b1d5cd559b267dc3688e74e060405161007e90610111565b60405180910390a27f3c85f81a244ac32f77f1322f42aad33171eafc3648aecc9ff0140c8993c3795760405160405180910390a1565b600082825260208201905092915050565b7f48656c6c6f20576f726c64210000000000000000000000000000000000000000600082015250565b60006100fb600c836100b4565b9150610106826100c5565b602082019050919050565b6000602082019050818103600083015261012a816100ee565b905091905056fea2646970667358221220565587b323a87ce05813eeacd18f881d447eb97daeb6019b810bd2bfcb3f4eae64736f6c63430008110033 \ No newline at end of file diff --git a/dozer-ingestion/ethereum/src/log/tests/contracts/CustomEvent.json b/dozer-ingestion/ethereum/src/log/tests/contracts/CustomEvent.json deleted file mode 100644 index 0013a98f67..0000000000 --- a/dozer-ingestion/ethereum/src/log/tests/contracts/CustomEvent.json +++ /dev/null @@ -1,34 +0,0 @@ -[ - { - "anonymous": false, - "inputs": [], - "name": "CustomEvent", - "type": "event" - }, - { - "anonymous": false, - "inputs": [ - { - "indexed": true, - "internalType": "address", - "name": "sender", - "type": "address" - }, - { - "indexed": false, - "internalType": "string", - "name": "message", - "type": "string" - } - ], - "name": "Log", - "type": "event" - }, - { - "inputs": [], - "name": "test", - "outputs": [], - "stateMutability": "nonpayable", - "type": "function" - } -] \ No newline at end of file diff --git a/dozer-ingestion/ethereum/src/log/tests/contracts/CustomEvent.sol b/dozer-ingestion/ethereum/src/log/tests/contracts/CustomEvent.sol deleted file mode 100644 index 6ef72d80a5..0000000000 --- a/dozer-ingestion/ethereum/src/log/tests/contracts/CustomEvent.sol +++ /dev/null @@ -1,12 +0,0 @@ -// SPDX-License-Identifier: GPL-3.0 -pragma solidity >=0.5.0 <0.9.0; - -contract Event { - event Log(address indexed sender, string message); - event CustomEvent(); - - function test() public { - emit Log(msg.sender, "Hello World!"); - emit CustomEvent(); - } -} diff --git a/dozer-ingestion/ethereum/src/log/tests/docker-compose.yml b/dozer-ingestion/ethereum/src/log/tests/docker-compose.yml deleted file mode 100644 index ce6c1219eb..0000000000 --- a/dozer-ingestion/ethereum/src/log/tests/docker-compose.yml +++ /dev/null @@ -1,12 +0,0 @@ -version: '3.9' -services: - ethereum: - build: ./ - ports: - - "8545:8545" - healthcheck: - test: curl -sf -X POST --data '{"jsonrpc":"2.0","method":"eth_blockNumber","params":[],"id":1}' http:// 0.0.0.0:8545 - interval: 5s - timeout: 5s - retries: 10 - \ No newline at end of file diff --git a/dozer-ingestion/ethereum/src/log/tests/helper.rs b/dozer-ingestion/ethereum/src/log/tests/helper.rs deleted file mode 100644 index 3a49f35ef1..0000000000 --- a/dozer-ingestion/ethereum/src/log/tests/helper.rs +++ /dev/null @@ -1,102 +0,0 @@ -use std::{sync::Arc, time::Duration}; - -use dozer_ingestion_connector::{ - dozer_types::{ - errors::internal::BoxedError, - log::info, - models::ingestion_types::{EthContract, EthFilter, EthLogConfig, IngestionMessage}, - types::Operation, - }, - test_util::spawn_connector, - tokio::runtime::Runtime, - Connector, TableInfo, -}; -use web3::{ - contract::{Contract, Options}, - transports::WebSocket, - types::H160, -}; - -use crate::{helper, EthLogConnector}; - -pub async fn deploy_contract(wss_url: String, my_account: H160) -> Contract { - let web3 = helper::get_wss_client(&wss_url).await.unwrap(); - // Get the contract bytecode for instance from Solidity compiler - let bytecode = include_str!("./contracts/CustomEvent.code").trim_end(); - let abi = include_bytes!("./contracts/CustomEvent.json"); - // Deploying a contract - let builder = Contract::deploy(web3.eth(), abi).unwrap(); - let contract = builder - .confirmations(0) - .options(Options::with(|opt| { - opt.gas = Some(3_000_000.into()); - })) - .execute(bytecode, (), my_account) - .await - .unwrap(); - - contract - .call("test", (), my_account, Options::default()) - .await - .unwrap(); - contract -} - -pub async fn get_eth_tables( - wss_url: String, - contract: Contract, -) -> Result<(EthLogConnector, Vec), BoxedError> { - let address = format!("{:?}", contract.address()); - let mut eth_connector = EthLogConnector::new( - EthLogConfig { - wss_url, - filter: Some(EthFilter { - from_block: Some(0), - to_block: None, - addresses: vec![address.clone()], - topics: vec![], - }), - contracts: vec![EthContract { - name: "custom_event".to_string(), - address, - abi: include_str!("./contracts/CustomEvent.json") - .trim_end() - .to_string(), - }], - }, - "eth_test".to_string(), - ); - - let (table_infos, _) = eth_connector.list_all_schemas().await?; - for table_info in table_infos.iter() { - info!("Schema: {}", table_info.name); - } - Ok((eth_connector, table_infos)) -} - -pub async fn run_eth_sample( - runtime: Arc, - wss_url: String, - my_account: H160, -) -> (Contract, Vec) { - dozer_tracing::init_telemetry(None, &Default::default()); - let orig_hook = std::panic::take_hook(); - std::panic::set_hook(Box::new(move |panic_info| { - // invoke the default handler and exit the process - orig_hook(panic_info); - })); - - let contract = deploy_contract(wss_url.clone(), my_account).await; - - let (connector, tables) = get_eth_tables(wss_url, contract.clone()).await.unwrap(); - let (mut iterator, _) = spawn_connector(runtime, connector, tables); - - let mut msgs = vec![]; - while let Some(IngestionMessage::OperationEvent { - table_index: 0, op, .. - }) = iterator.next_timeout(Duration::from_millis(400)).await - { - msgs.push(op); - } - (contract, msgs) -} diff --git a/dozer-ingestion/ethereum/src/log/tests/mod.rs b/dozer-ingestion/ethereum/src/log/tests/mod.rs deleted file mode 100644 index 7fba8c6200..0000000000 --- a/dozer-ingestion/ethereum/src/log/tests/mod.rs +++ /dev/null @@ -1,2 +0,0 @@ -mod connector; -pub mod helper; diff --git a/dozer-ingestion/ethereum/src/trace/connector.rs b/dozer-ingestion/ethereum/src/trace/connector.rs deleted file mode 100644 index 215aa5717a..0000000000 --- a/dozer-ingestion/ethereum/src/trace/connector.rs +++ /dev/null @@ -1,224 +0,0 @@ -use dozer_ingestion_connector::{ - async_trait, - dozer_types::{ - errors::internal::BoxedError, - log::{error, info, warn}, - models::ingestion_types::{default_batch_size, EthTraceConfig, IngestionMessage}, - node::OpIdentifier, - types::FieldType, - }, - utils::TableNotFound, - CdcType, Connector, Ingestor, SourceSchema, SourceSchemaResult, TableIdentifier, TableInfo, -}; - -use super::super::helper as conn_helper; -use super::helper::{self, get_block_traces, map_trace_to_ops}; - -#[derive(Debug)] -pub struct EthTraceConnector { - pub config: EthTraceConfig, - pub conn_name: String, -} - -pub const ETH_TRACE_TABLE: &str = "eth_traces"; -pub const RETRIES: u16 = 10; -impl EthTraceConnector { - pub fn new(config: EthTraceConfig, conn_name: String) -> Self { - Self { config, conn_name } - } -} - -#[async_trait] -impl Connector for EthTraceConnector { - fn types_mapping() -> Vec<(String, Option)> - where - Self: Sized, - { - todo!() - } - - async fn validate_connection(&mut self) -> Result<(), BoxedError> { - validate(&self.config).await - } - - async fn list_tables(&mut self) -> Result, BoxedError> { - Ok(vec![TableIdentifier::from_table_name( - ETH_TRACE_TABLE.to_string(), - )]) - } - - async fn validate_tables(&mut self, tables: &[TableIdentifier]) -> Result<(), BoxedError> { - for table in tables { - if table.name != ETH_TRACE_TABLE || table.schema.is_some() { - return Err(TableNotFound { - schema: table.schema.clone(), - name: table.name.clone(), - } - .into()); - } - } - Ok(()) - } - - async fn list_columns( - &mut self, - tables: Vec, - ) -> Result, BoxedError> { - let mut result = Vec::new(); - for table in tables { - if table.name != ETH_TRACE_TABLE || table.schema.is_some() { - return Err(TableNotFound { - schema: table.schema.clone(), - name: table.name.clone(), - } - .into()); - } - let column_names = helper::get_trace_schema() - .fields - .into_iter() - .map(|field| field.name) - .collect(); - result.push(TableInfo { - schema: table.schema, - name: table.name, - column_names, - }) - } - Ok(result) - } - - async fn get_schemas( - &mut self, - _table_infos: &[TableInfo], - ) -> Result, BoxedError> { - warn!("TODO: respect table_infos"); - Ok(vec![Ok(SourceSchema::new( - helper::get_trace_schema(), - CdcType::Nothing, - ))]) - } - - async fn serialize_state(&self) -> Result, BoxedError> { - Ok(vec![]) - } - - async fn start( - &mut self, - ingestor: &Ingestor, - _tables: Vec, - _last_checkpoint: Option, - ) -> Result<(), BoxedError> { - let config = self.config.clone(); - let conn_name = self.conn_name.clone(); - run(ingestor, config, conn_name).await - } -} - -pub async fn validate(config: &EthTraceConfig) -> Result<(), BoxedError> { - // Check if transport can be initialized - let tuple = conn_helper::get_batch_http_client(&config.https_url).await?; - - // Check if debug API is available - get_block_traces(tuple, (1000000, 1000005)).await?; - - Ok(()) -} - -pub async fn run( - ingestor: &Ingestor, - config: EthTraceConfig, - conn_name: String, -) -> Result<(), BoxedError> { - let client_tuple = conn_helper::get_batch_http_client(&config.https_url).await?; - - info!( - "Starting Eth Trace connector: {} from block {}", - conn_name, config.from_block - ); - let batch_iter = BatchIterator::new( - config.from_block, - config.to_block, - config.batch_size.unwrap_or_else(default_batch_size), - ); - - let mut errors: Vec = vec![]; - for batch in batch_iter { - for retry in 0..RETRIES { - if retry >= RETRIES - 1 { - error!("Eth Trace connector failed more than {RETRIES} times"); - return Err(errors.pop().unwrap()); - } - - let res = get_block_traces(client_tuple.clone(), batch).await; - match res { - Ok(arr) => { - for result in arr { - let ops = map_trace_to_ops(&result.result); - - for op in ops { - if ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index: 0, // We have only one table - op, - id: None, - }) - .await - .is_err() - { - // If receiving end is closed, exit - return Ok(()); - } - } - } - - break; - } - Err(e) => { - errors.push(e); - error!( - "Failed to get traces for block {}.. Attempt {}", - batch.0, - retry + 1 - ); - } - } - } - } - - Ok(()) -} - -pub struct BatchIterator { - current_block: u64, - to_block: Option, - batch_size: u64, -} - -impl BatchIterator { - pub fn new(current_block: u64, to_block: Option, batch_size: u64) -> Self { - Self { - current_block, - to_block, - batch_size, - } - } -} -impl Iterator for BatchIterator { - type Item = (u64, u64); - - fn next(&mut self) -> Option { - let mut end_block = self.current_block + self.batch_size; - if let Some(to_block) = self.to_block { - if self.current_block > to_block { - return None; - } - - if end_block > to_block + 1 { - end_block = to_block + 1; - } - } - let current_batch = (self.current_block, end_block); - self.current_block = end_block; - Some(current_batch) - } -} diff --git a/dozer-ingestion/ethereum/src/trace/helper.rs b/dozer-ingestion/ethereum/src/trace/helper.rs deleted file mode 100644 index b69fbf67bd..0000000000 --- a/dozer-ingestion/ethereum/src/trace/helper.rs +++ /dev/null @@ -1,176 +0,0 @@ -use dozer_ingestion_connector::dozer_types::{ - errors::internal::BoxedError, - log::{debug, error}, - serde::{Deserialize, Serialize}, - serde_json::{self, json}, - types::{Field, FieldDefinition, FieldType, Operation, Record, Schema, SourceDefinition}, -}; -use web3::transports::{Batch, Http}; -use web3::types::{H160, U256}; -use web3::{BatchTransport, Transport, Web3}; - -#[derive(Default, Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde( - crate = "dozer_ingestion_connector::dozer_types::serde", - rename_all = "camelCase" -)] -pub struct TraceResult { - pub result: Trace, -} - -#[derive(Default, Debug, Clone, PartialEq, Serialize, Deserialize)] -#[serde( - crate = "dozer_ingestion_connector::dozer_types::serde", - rename_all = "camelCase" -)] -pub struct Trace { - #[serde(rename = "type")] - pub type_field: String, - pub from: H160, - pub to: H160, - pub value: Option, - pub gas: U256, - pub gas_used: U256, - pub input: Option, - pub output: Option, - pub calls: Option>, -} - -pub async fn get_block_traces( - tuple: (Web3>, Http), - batch: (u64, u64), -) -> Result, BoxedError> { - debug_assert!(batch.0 < batch.1, "Batch start must be less than batch end"); - let (client, transport) = tuple; - let mut requests = vec![]; - let mut results = vec![]; - debug!("Getting eth traces for block range: {:?}", batch); - let mut request_count = 0; - let (from, to) = batch; - for block_no in from..to { - let request = client.transport().prepare( - "debug_traceBlockByNumber", - vec![ - format!("0x{block_no:x}").into(), - json!({ - "tracer": "callTracer" - } - ), - ], - ); - requests.push(request); - request_count += 1; - } - - let batch_results = transport.send_batch(requests).await?; - - debug!( - "Requests: {:?}, Results: {:?}", - request_count, - batch_results.len(), - ); - - for (idx, res) in batch_results.iter().enumerate() { - let res = res.clone().map_err(|e| { - error!("Error getting trace: {:?}", e); - e - })?; - - let r: Vec = serde_json::from_value(res)?; - - debug!("Idx: {} : Response: {:?}", idx, r); - - results.extend(r); - } - Ok(results) -} - -pub fn get_trace_schema() -> Schema { - Schema { - fields: vec![ - FieldDefinition { - name: "type_field".to_string(), - typ: FieldType::String, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "from".to_string(), - typ: FieldType::String, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "to".to_string(), - typ: FieldType::String, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "value".to_string(), - typ: FieldType::UInt, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "gas".to_string(), - typ: FieldType::UInt, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "gas_used".to_string(), - typ: FieldType::UInt, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "input".to_string(), - typ: FieldType::Text, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "output".to_string(), - typ: FieldType::Text, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }, - ], - primary_index: vec![], - } -} - -pub fn map_trace_to_ops(trace: &Trace) -> Vec { - let mut ops = vec![]; - let op = Operation::Insert { - new: Record { - values: vec![ - Field::String(trace.type_field.clone()), - Field::String(format!("{:?}", trace.from)), - Field::String(format!("{:?}", trace.to)), - Field::UInt(trace.value.unwrap_or(U256::zero()).low_u64()), - Field::UInt(trace.gas.low_u64()), - Field::UInt(trace.gas_used.low_u64()), - Field::Text(format!("{:?}", trace.input)), - Field::Text(format!("{:?}", trace.output)), - ], - lifetime: None, - }, - }; - ops.push(op); - if let Some(calls) = &trace.calls { - for call in calls { - ops.append(&mut map_trace_to_ops(call)); - } - } - ops -} diff --git a/dozer-ingestion/ethereum/src/trace/mod.rs b/dozer-ingestion/ethereum/src/trace/mod.rs deleted file mode 100644 index 5b9d78cd25..0000000000 --- a/dozer-ingestion/ethereum/src/trace/mod.rs +++ /dev/null @@ -1,5 +0,0 @@ -mod connector; -pub mod helper; -pub use connector::EthTraceConnector; -#[cfg(test)] -mod tests; diff --git a/dozer-ingestion/ethereum/src/trace/tests.rs b/dozer-ingestion/ethereum/src/trace/tests.rs deleted file mode 100644 index a4da1e1ea1..0000000000 --- a/dozer-ingestion/ethereum/src/trace/tests.rs +++ /dev/null @@ -1,90 +0,0 @@ -use std::{env, time::Duration}; - -use dozer_ingestion_connector::{ - dozer_types::{ - log::info, - models::ingestion_types::{EthTraceConfig, IngestionMessage}, - types::{Field, Operation}, - }, - test_util::{create_test_runtime, spawn_connector}, - tokio, Connector, -}; - -use crate::{helper, trace::helper::get_block_traces, EthTraceConnector}; - -use super::connector::BatchIterator; - -#[test] -fn test_iterator() { - let mut iter = BatchIterator::new(1, Some(2), 1); - assert_eq!(iter.next(), Some((1, 2))); - assert_eq!(iter.next(), Some((2, 3))); - assert_eq!(iter.next(), None); - - let mut iter = BatchIterator::new(1, Some(1), 3); - assert_eq!(iter.next(), Some((1, 2))); - assert_eq!(iter.next(), None); -} - -#[tokio::test] -#[ignore] -async fn test_get_block_traces() { - let url = env::var("ETH_HTTPS_URL").unwrap(); - let client = helper::get_batch_http_client(&url).await.unwrap(); - let traces = get_block_traces(client, (1000000, 1000005)).await.unwrap(); - assert!(!traces.is_empty(), "Failed to get traces found"); -} - -#[test] -#[ignore] -fn test_trace_iterator() { - let runtime = create_test_runtime(); - let https_url = env::var("ETH_HTTPS_URL").unwrap(); - - dozer_tracing::init_telemetry(None, &Default::default()); - let orig_hook = std::panic::take_hook(); - std::panic::set_hook(Box::new(move |panic_info| { - // invoke the default handler and exit the process - orig_hook(panic_info); - })); - - info!("Initializing with WSS: {}", https_url); - - let mut connector = EthTraceConnector::new( - EthTraceConfig { - https_url, - from_block: 1000000, - to_block: Some(1000001), - batch_size: Some(100), - }, - "test".to_string(), - ); - - let (tables, schemas) = runtime.block_on(connector.list_all_schemas()).unwrap(); - for s in schemas { - info!("\n{}", s.schema.print()); - } - let (mut iterator, _) = spawn_connector(runtime.clone(), connector, tables); - - runtime.block_on(async move { - if let Some(IngestionMessage::OperationEvent { op, .. }) = - iterator.next_timeout(Duration::from_millis(1000)).await - { - assert!(matches!(op, Operation::Insert { .. })); - if let Operation::Insert { new } = op { - assert!(matches!(new.values[0], Field::String(_))); - assert!(matches!(new.values[1], Field::String(_))); - assert!(matches!(new.values[2], Field::String(_))); - assert!(matches!(new.values[3], Field::UInt(_))); - assert!(matches!(new.values[4], Field::UInt(_))); - assert!(matches!(new.values[5], Field::UInt(_))); - assert!(matches!(new.values[6], Field::Text(_))); - assert!(matches!(new.values[7], Field::Text(_))); - } else { - panic!("Expected insert"); - } - } else { - panic!("No message received"); - } - }); -} diff --git a/dozer-ingestion/grpc/Cargo.toml b/dozer-ingestion/grpc/Cargo.toml deleted file mode 100644 index 971afa5ba0..0000000000 --- a/dozer-ingestion/grpc/Cargo.toml +++ /dev/null @@ -1,13 +0,0 @@ -[package] -name = "dozer-ingestion-grpc" -version = "0.4.0" -edition = "2021" -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-ingestion-connector = { path = "../connector" } -tower-http = { version = "0.4", features = ["full"] } -tonic-web = "0.11.0" -tonic-reflection = "0.11.0" diff --git a/dozer-ingestion/grpc/src/adapter/arrow.rs b/dozer-ingestion/grpc/src/adapter/arrow.rs deleted file mode 100644 index 3e8b258eb6..0000000000 --- a/dozer-ingestion/grpc/src/adapter/arrow.rs +++ /dev/null @@ -1,152 +0,0 @@ -use std::collections::HashMap; - -use dozer_ingestion_connector::{ - async_trait, - dozer_types::{ - arrow::datatypes::Schema as ArrowSchema, - arrow::{self, ipc::reader::StreamReader}, - arrow_types::{self, from_arrow::map_record_batch_to_dozer_records}, - bytes::{Buf, Bytes}, - grpc_types::ingest::IngestArrowRequest, - models::ingestion_types::{IngestionMessage, TransactionInfo}, - serde::{Deserialize, Serialize}, - serde_json, - types::{Operation, Record, Schema}, - }, - CdcType, Ingestor, SourceSchema, -}; - -use crate::Error; - -use super::{GrpcIngestMessage, IngestAdapter}; - -#[derive(Clone, Serialize, Deserialize, Debug)] -#[serde(crate = "dozer_ingestion_connector::dozer_types::serde")] -pub struct GrpcArrowSchema { - pub name: String, - pub schema: arrow::datatypes::Schema, - #[serde(default)] - pub cdc_type: CdcType, -} - -// Input is a JSON string or a path to a JSON file -// Takes name, arrow schema, and optionally replication type - -#[derive(Debug)] -pub struct ArrowAdapter { - schema_map: HashMap, - _arrow_schemas: HashMap, -} - -impl ArrowAdapter { - #[allow(clippy::type_complexity)] - fn parse_schemas( - schemas_str: &str, - ) -> Result<(Vec<(String, SourceSchema)>, HashMap), Error> { - let grpc_schemas: Vec = serde_json::from_str(schemas_str)?; - let mut schemas = vec![]; - - let mut arrow_schemas = HashMap::new(); - - for (id, grpc_schema) in grpc_schemas.into_iter().enumerate() { - let schema = arrow_types::from_arrow::map_schema_to_dozer(&grpc_schema.schema)?; - - arrow_schemas.insert(id as u32, grpc_schema.schema); - - schemas.push(( - grpc_schema.name, - SourceSchema::new(schema, grpc_schema.cdc_type), - )); - } - Ok((schemas, arrow_schemas)) - } -} - -#[async_trait] -impl IngestAdapter for ArrowAdapter { - fn new(schemas_str: String) -> Result { - let (schemas, arrow_schemas) = Self::parse_schemas(&schemas_str)?; - let schema_map = schemas.into_iter().collect(); - Ok(Self { - schema_map, - _arrow_schemas: arrow_schemas, - }) - } - - fn get_schemas(&self) -> Vec<(String, SourceSchema)> { - self.schema_map - .iter() - .map(|(key, value)| (key.clone(), value.clone())) - .collect() - } - - async fn handle_message( - &self, - table_index: usize, - msg: GrpcIngestMessage, - ingestor: &'static Ingestor, - ) -> Result<(), Error> { - match msg { - GrpcIngestMessage::Default(_) => Err(Error::CannotHandleDefaultMessage), - GrpcIngestMessage::Arrow(msg) => { - handle_message(table_index, msg, &self.schema_map, ingestor).await - } - } - } -} - -pub async fn handle_message( - table_index: usize, - req: IngestArrowRequest, - schema_map: &HashMap, - ingestor: &'static Ingestor, -) -> Result<(), Error> { - let schema = &schema_map - .get(&req.schema_name) - .ok_or_else(|| Error::SchemaNotFound(req.schema_name.clone()))? - .schema; - - let records = map_record_batch(req, schema)?; - - for r in records { - let op = Operation::Insert { new: r }; - - if ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index, - op, - id: None, - }) - .await - .is_err() - { - // If receiving end is closed, then we can just ignore the message - return Ok(()); - } - } - if ingestor - .handle_message(IngestionMessage::TransactionInfo(TransactionInfo::Commit { - id: None, - source_time: None, - })) - .await - .is_err() - { - return Ok(()); - } - - Ok(()) -} - -fn map_record_batch(req: IngestArrowRequest, schema: &Schema) -> Result, Error> { - let mut buf = Bytes::from(req.records).reader(); - // read stream back - let mut reader = StreamReader::try_new(&mut buf, None)?; - let mut records = Vec::new(); - while let Some(Ok(batch)) = reader.next() { - let b_recs = map_record_batch_to_dozer_records(batch, schema)?; - records.extend(b_recs); - } - - Ok(records) -} diff --git a/dozer-ingestion/grpc/src/adapter/default.rs b/dozer-ingestion/grpc/src/adapter/default.rs deleted file mode 100644 index 09de2ec905..0000000000 --- a/dozer-ingestion/grpc/src/adapter/default.rs +++ /dev/null @@ -1,200 +0,0 @@ -use dozer_ingestion_connector::{ - async_trait, - dozer_types::{ - self, chrono, - grpc_types::{self, ingest::IngestRequest}, - json_types::prost_to_json_value, - models::ingestion_types::{IngestionMessage, TransactionInfo}, - ordered_float::OrderedFloat, - rust_decimal::Decimal, - serde_json, - types::{Field, Operation, Record, Schema}, - }, - Ingestor, SourceSchema, -}; - -use crate::Error; - -use super::{GrpcIngestMessage, IngestAdapter}; - -use std::collections::HashMap; - -#[derive(Debug)] -pub struct DefaultAdapter { - schema_map: HashMap, -} - -impl DefaultAdapter { - fn parse_schemas(schemas_str: &str) -> Result, Error> { - let schemas: HashMap = serde_json::from_str(schemas_str)?; - - Ok(schemas) - } -} - -#[async_trait] -impl IngestAdapter for DefaultAdapter { - fn new(schemas_str: String) -> Result { - let schema_map = Self::parse_schemas(&schemas_str)?; - Ok(Self { schema_map }) - } - fn get_schemas(&self) -> Vec<(String, SourceSchema)> { - self.schema_map - .iter() - .map(|(key, value)| (key.clone(), value.clone())) - .collect() - } - - async fn handle_message( - &self, - table_index: usize, - msg: GrpcIngestMessage, - ingestor: &'static Ingestor, - ) -> Result<(), Error> { - match msg { - GrpcIngestMessage::Default(msg) => { - handle_message(table_index, msg, &self.schema_map, ingestor).await - } - GrpcIngestMessage::Arrow(_) => Err(Error::CannotHandleArrowMessage), - } - } -} - -pub async fn handle_message( - table_index: usize, - req: IngestRequest, - schema_map: &HashMap, - ingestor: &'static Ingestor, -) -> Result<(), Error> { - let schema = &schema_map - .get(&req.schema_name) - .ok_or_else(|| Error::SchemaNotFound(req.schema_name.clone()))? - .schema; - - let op = match req.typ() { - grpc_types::types::OperationType::Insert => Operation::Insert { - new: map_record(req.new, schema)?, - }, - grpc_types::types::OperationType::Delete => Operation::Delete { - old: map_record(req.old, schema)?, - }, - grpc_types::types::OperationType::Update => Operation::Update { - old: map_record(req.old, schema)?, - new: map_record(req.new, schema)?, - }, - }; - // If receiving end is closed, then we can just ignore the message - let _ = ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index, - op, - id: None, - }) - .await; - let _ = ingestor - .handle_message(IngestionMessage::TransactionInfo(TransactionInfo::Commit { - id: None, - source_time: None, - })) - .await; - Ok(()) -} - -fn map_record(rec: Vec, schema: &Schema) -> Result { - let mut values: Vec = vec![]; - let values_count = rec.len(); - let schema_fields_count = schema.fields.len(); - if values_count != schema_fields_count { - return Err(Error::NumFieldsMismatch { - values_count, - schema_fields_count, - }); - } - - for (idx, v) in rec.into_iter().enumerate() { - let typ = schema.fields[idx].typ; - - let val = v.value.map(|value| match (value, typ) { - ( - grpc_types::types::value::Value::UintValue(a), - dozer_types::types::FieldType::UInt, - ) => Ok(dozer_types::types::Field::UInt(a)), - - (grpc_types::types::value::Value::IntValue(a), dozer_types::types::FieldType::Int) => { - Ok(dozer_types::types::Field::Int(a)) - } - - ( - grpc_types::types::value::Value::FloatValue(a), - dozer_types::types::FieldType::Float, - ) => Ok(dozer_types::types::Field::Float(OrderedFloat(a))), - - ( - grpc_types::types::value::Value::BoolValue(a), - dozer_types::types::FieldType::Boolean, - ) => Ok(dozer_types::types::Field::Boolean(a)), - - ( - grpc_types::types::value::Value::StringValue(a), - dozer_types::types::FieldType::String, - ) => Ok(dozer_types::types::Field::String(a)), - - ( - grpc_types::types::value::Value::BytesValue(a), - dozer_types::types::FieldType::Binary, - ) => Ok(dozer_types::types::Field::Binary(a)), - ( - grpc_types::types::value::Value::StringValue(a), - dozer_types::types::FieldType::Text, - ) => Ok(dozer_types::types::Field::Text(a)), - ( - grpc_types::types::value::Value::JsonValue(a), - dozer_types::types::FieldType::Json, - ) => Ok(dozer_types::types::Field::Json(prost_to_json_value(a))), - ( - grpc_types::types::value::Value::TimestampValue(a), - dozer_types::types::FieldType::Timestamp, - ) => Ok( - chrono::NaiveDateTime::from_timestamp_opt(a.seconds, a.nanos as u32) - .map(|t| { - dozer_types::types::Field::Timestamp( - chrono::DateTime::::from_naive_utc_and_offset( - t, - chrono::Utc, - ) - .into(), - ) - }) - .unwrap_or(dozer_types::types::Field::Null), - ), - ( - grpc_types::types::value::Value::DecimalValue(d), - dozer_types::types::FieldType::Decimal, - ) => Ok(dozer_types::types::Field::Decimal(Decimal::from_parts( - d.lo, d.mid, d.hi, d.negative, d.scale, - ))), - ( - grpc_types::types::value::Value::DateValue(_), - dozer_types::types::FieldType::UInt, - ) - | ( - grpc_types::types::value::Value::DateValue(_), - dozer_types::types::FieldType::Date, - ) - | ( - grpc_types::types::value::Value::PointValue(_), - dozer_types::types::FieldType::Point, - ) => Ok(dozer_types::types::Field::Null), - (a, b) => Err(Error::FieldTypeMismatch { - index: idx, - value: a, - field_type: b, - }), - }); - values.push(val.unwrap_or(Ok(dozer_types::types::Field::Null))?); - } - Ok(Record { - values, - lifetime: None, - }) -} diff --git a/dozer-ingestion/grpc/src/adapter/mod.rs b/dozer-ingestion/grpc/src/adapter/mod.rs deleted file mode 100644 index 07f4262e69..0000000000 --- a/dozer-ingestion/grpc/src/adapter/mod.rs +++ /dev/null @@ -1,70 +0,0 @@ -use std::fmt::Debug; - -mod default; - -mod arrow; - -pub use arrow::ArrowAdapter; -pub use default::DefaultAdapter; -use dozer_ingestion_connector::{ - async_trait, - dozer_types::grpc_types::ingest::{IngestArrowRequest, IngestRequest}, - Ingestor, SourceSchema, -}; - -use crate::Error; - -#[async_trait] -pub trait IngestAdapter: Debug -where - Self: Send + Sync + 'static + Sized, -{ - fn new(schemas_str: String) -> Result; - fn get_schemas(&self) -> Vec<(String, SourceSchema)>; - async fn handle_message( - &self, - table_index: usize, - msg: GrpcIngestMessage, - ingestor: &'static Ingestor, - ) -> Result<(), Error>; -} - -pub enum GrpcIngestMessage { - Default(IngestRequest), - Arrow(IngestArrowRequest), -} -pub struct GrpcIngestor -where - A: IngestAdapter, -{ - adapter: A, -} -impl GrpcIngestor -where - T: IngestAdapter, -{ - pub fn new(schemas_str: String) -> Result { - let adapter = T::new(schemas_str)?; - Ok(Self { adapter }) - } -} - -impl GrpcIngestor -where - A: IngestAdapter, -{ - pub fn get_schemas(&self) -> Result, Error> { - Ok(self.adapter.get_schemas()) - } - - pub async fn handle_message( - &self, - table_index: usize, - msg: GrpcIngestMessage, - ingestor: &'static Ingestor, - ) -> Result<(), Error> { - self.adapter - .handle_message(table_index, msg, ingestor) - .await - } -} diff --git a/dozer-ingestion/grpc/src/connector.rs b/dozer-ingestion/grpc/src/connector.rs deleted file mode 100644 index 33ecb3e9dc..0000000000 --- a/dozer-ingestion/grpc/src/connector.rs +++ /dev/null @@ -1,208 +0,0 @@ -use std::fmt::Debug; - -use crate::Error; - -use super::adapter::{GrpcIngestor, IngestAdapter}; -use super::ingest::IngestorServiceImpl; -use dozer_ingestion_connector::dozer_types::node::OpIdentifier; -use dozer_ingestion_connector::schema_parser::SchemaParser; -use dozer_ingestion_connector::utils::TableNotFound; -use dozer_ingestion_connector::{ - async_trait, dozer_types, - dozer_types::{ - errors::internal::BoxedError, - grpc_types::ingest::ingest_service_server::IngestServiceServer, - log::{info, warn}, - models::ingestion_types::{default_ingest_host, default_ingest_port, GrpcConfig}, - tonic::transport::Server, - tracing::Level, - }, - Connector, Ingestor, SourceSchema, SourceSchemaResult, TableIdentifier, TableInfo, -}; -use tower_http::trace::{self, TraceLayer}; - -#[derive(Debug)] -pub struct GrpcConnector -where - T: IngestAdapter, -{ - pub name: String, - pub config: GrpcConfig, - _phantom: std::marker::PhantomData, -} - -impl GrpcConnector -where - T: IngestAdapter, -{ - pub fn new(name: String, config: GrpcConfig) -> Self { - Self { - name, - config, - _phantom: std::marker::PhantomData, - } - } - - pub async fn serve(&self, ingestor: &Ingestor, tables: Vec) -> Result<(), Error> { - let host = self.config.host.clone().unwrap_or_else(default_ingest_host); - let port = self.config.port.unwrap_or_else(default_ingest_port); - - let addr = format!("{host}:{port}").parse()?; - - let schemas_str = SchemaParser::parse_config(&self.config.schemas)?; - let adapter = GrpcIngestor::::new(schemas_str)?; - - // Ingestor will live as long as the server - // Refactor to use Arc - let ingestor = unsafe { std::mem::transmute::<&'_ Ingestor, &'static Ingestor>(ingestor) }; - - let ingest_service = IngestorServiceImpl::new(adapter, ingestor, tables); - let ingest_service = tonic_web::enable(IngestServiceServer::new(ingest_service)); - - let reflection_service = tonic_reflection::server::Builder::configure() - .register_encoded_file_descriptor_set( - dozer_types::grpc_types::ingest::FILE_DESCRIPTOR_SET, - ) - .build() - .unwrap(); - info!("Starting Dozer GRPC Ingestor on http://{}:{} ", host, port,); - Server::builder() - .layer( - TraceLayer::new_for_http() - .make_span_with(trace::DefaultMakeSpan::new().level(Level::INFO)) - .on_response(trace::DefaultOnResponse::new().level(Level::INFO)), - ) - .accept_http1(true) - .add_service(ingest_service) - .add_service(reflection_service) - .serve(addr) - .await - .map_err(Into::into) - } -} - -impl GrpcConnector { - fn get_all_schemas(&self) -> Result, Error> { - let schemas_str = SchemaParser::parse_config(&self.config.schemas)?; - let adapter = GrpcIngestor::::new(schemas_str)?; - adapter.get_schemas() - } -} - -#[async_trait] -impl Connector for GrpcConnector -where - T: IngestAdapter, -{ - fn types_mapping() -> Vec<(String, Option)> - where - Self: Sized, - { - todo!() - } - - async fn validate_connection(&mut self) -> Result<(), BoxedError> { - self.get_all_schemas()?; - Ok(()) - } - - async fn list_tables(&mut self) -> Result, BoxedError> { - Ok(self - .get_all_schemas()? - .into_iter() - .map(|(name, _)| TableIdentifier::from_table_name(name)) - .collect()) - } - - async fn validate_tables(&mut self, tables: &[TableIdentifier]) -> Result<(), BoxedError> { - let schemas = self.get_all_schemas()?; - for table in tables { - if !schemas - .iter() - .any(|(name, _)| name == &table.name && table.schema.is_none()) - { - return Err(TableNotFound { - schema: table.schema.clone(), - name: table.name.clone(), - } - .into()); - } - } - Ok(()) - } - - async fn list_columns( - &mut self, - tables: Vec, - ) -> Result, BoxedError> { - let schemas = self.get_all_schemas()?; - let mut result = vec![]; - for table in tables { - if let Some((_, schema)) = schemas - .iter() - .find(|(name, _)| name == &table.name && table.schema.is_none()) - { - let column_names = schema - .schema - .fields - .iter() - .map(|field| field.name.clone()) - .collect(); - result.push(TableInfo { - schema: table.schema, - name: table.name, - column_names, - }) - } else { - return Err(TableNotFound { - schema: table.schema.clone(), - name: table.name.clone(), - } - .into()); - } - } - Ok(result) - } - - async fn get_schemas( - &mut self, - table_infos: &[TableInfo], - ) -> Result, BoxedError> { - let schemas_str = SchemaParser::parse_config(&self.config.schemas)?; - let adapter = GrpcIngestor::::new(schemas_str)?; - - let schemas = adapter.get_schemas()?; - - let mut result = vec![]; - for table in table_infos { - if let Some((_, schema)) = schemas - .iter() - .find(|(name, _)| name == &table.name && table.schema.is_none()) - { - warn!("TODO: filter columns"); - result.push(Ok(schema.clone())); - } else { - result.push(Err(TableNotFound { - schema: table.schema.clone(), - name: table.name.clone(), - } - .into())); - } - } - - Ok(result) - } - - async fn serialize_state(&self) -> Result, BoxedError> { - Ok(vec![]) - } - - async fn start( - &mut self, - ingestor: &Ingestor, - tables: Vec, - _last_checkpoint: Option, - ) -> Result<(), BoxedError> { - self.serve(ingestor, tables).await.map_err(Into::into) - } -} diff --git a/dozer-ingestion/grpc/src/ingest.rs b/dozer-ingestion/grpc/src/ingest.rs deleted file mode 100644 index 21beed123d..0000000000 --- a/dozer-ingestion/grpc/src/ingest.rs +++ /dev/null @@ -1,173 +0,0 @@ -use std::sync::Arc; - -use dozer_ingestion_connector::{ - dozer_types::{ - grpc_types::ingest::{ - ingest_service_server::IngestService, IngestArrowRequest, IngestRequest, IngestResponse, - }, - log::error, - tonic::{self, Streaming}, - }, - futures::StreamExt, - tokio, Ingestor, TableInfo, -}; - -use super::adapter::{GrpcIngestMessage, GrpcIngestor, IngestAdapter}; - -pub struct IngestorServiceImpl -where - T: IngestAdapter, -{ - adapter: Arc>, - ingestor: &'static Ingestor, - tables: Vec, -} -impl IngestorServiceImpl -where - T: IngestAdapter, -{ - pub fn new( - adapter: GrpcIngestor, - ingestor: &'static Ingestor, - tables: Vec, - ) -> Self { - Self { - adapter: Arc::new(adapter), - ingestor, - tables, - } - } -} -#[tonic::async_trait] -impl IngestService for IngestorServiceImpl -where - T: IngestAdapter, -{ - async fn ingest( - &self, - request: tonic::Request, - ) -> Result, tonic::Status> { - let req = request.into_inner(); - let table_index = self - .tables - .iter() - .position(|table| table.name == req.schema_name) - .ok_or(tonic::Status::not_found(format!( - "schema name not found: {}", - req.schema_name - )))?; - - let seq_no = req.seq_no; - self.adapter - .handle_message(table_index, GrpcIngestMessage::Default(req), self.ingestor) - .await - .map_err(|e| tonic::Status::internal(format!("ingestion stream error: {e}")))?; - - Ok(tonic::Response::new(IngestResponse { seq_no })) - } - - async fn ingest_stream( - &self, - req: tonic::Request>, - ) -> Result, tonic::Status> { - let mut in_stream = req.into_inner(); - - let adapter = self.adapter.clone(); - let ingestor = self.ingestor; - let table_names = self.tables.clone(); - let seq_no = tokio::spawn(async move { - let mut seq_no = 0; - while let Some(result) = in_stream.next().await { - if let Ok(req) = result { - let Some(table_index) = table_names - .iter() - .position(|table| table.name == req.schema_name) - else { - error!("schema name not found: {}", req.schema_name); - break; - }; - - seq_no = req.seq_no; - let res = adapter - .handle_message(table_index, GrpcIngestMessage::Default(req), ingestor) - .await; - if let Err(e) = res { - error!("ingestion stream insertion errored: {:#?}", e); - break; - } - } else { - error!("ingestion stream errored: {:#?}", result); - break; - } - } - seq_no - }) - .await - .map_err(|e| tonic::Status::internal(format!("ingestion stream error: {e}")))?; - Ok(tonic::Response::new(IngestResponse { seq_no })) - } - - async fn ingest_arrow( - &self, - request: tonic::Request, - ) -> Result, tonic::Status> { - let req = request.into_inner(); - let table_index = self - .tables - .iter() - .position(|table| table.name == req.schema_name) - .ok_or(tonic::Status::not_found(format!( - "schema name not found: {}", - req.schema_name - )))?; - - let seq_no = req.seq_no; - self.adapter - .handle_message(table_index, GrpcIngestMessage::Arrow(req), self.ingestor) - .await - .map_err(|e| tonic::Status::internal(format!("ingestion stream error: {e}")))?; - - Ok(tonic::Response::new(IngestResponse { seq_no })) - } - - async fn ingest_arrow_stream( - &self, - req: tonic::Request>, - ) -> Result, tonic::Status> { - let mut in_stream = req.into_inner(); - - let adapter = self.adapter.clone(); - let ingestor = self.ingestor; - let table_names = self.tables.clone(); - let seq_no = tokio::spawn(async move { - let mut seq_no = 0; - while let Some(result) = in_stream.next().await { - if let Ok(req) = result { - let Some(table_index) = table_names - .iter() - .position(|table| table.name == req.schema_name) - else { - error!("schema name not found: {}", req.schema_name); - break; - }; - - seq_no = req.seq_no; - let res = adapter - .handle_message(table_index, GrpcIngestMessage::Arrow(req), ingestor) - .await; - if let Err(e) = res { - error!("ingestion stream insertion errored: {:#?}", e); - break; - } - } else { - error!("ingestion stream errored: {:#?}", result); - break; - } - } - seq_no - }) - .await - .map_err(|e| tonic::Status::internal(format!("ingestion stream error: {e}")))?; - Ok(tonic::Response::new(IngestResponse { seq_no })) - } -} diff --git a/dozer-ingestion/grpc/src/lib.rs b/dozer-ingestion/grpc/src/lib.rs deleted file mode 100644 index 79908d9f9e..0000000000 --- a/dozer-ingestion/grpc/src/lib.rs +++ /dev/null @@ -1,52 +0,0 @@ -pub mod connector; -mod ingest; - -mod adapter; -use std::net::AddrParseError; - -pub use adapter::{ArrowAdapter, DefaultAdapter, GrpcIngestMessage, GrpcIngestor, IngestAdapter}; -use dozer_ingestion_connector::dozer_types::{ - arrow::error::ArrowError, - arrow_types::errors::FromArrowError, - grpc_types, serde_json, - thiserror::{self, Error}, - tonic::transport, - types::FieldType, -}; -use dozer_ingestion_connector::schema_parser::SchemaParserError; - -#[cfg(test)] -mod tests; - -#[derive(Debug, Error)] -pub enum Error { - #[error("Schema parser error: {0}")] - CannotReadFile(#[from] SchemaParserError), - #[error("serde json error: {0}")] - SerdeJson(#[from] serde_json::Error), - #[error("from arrow error: {0}")] - FromArrow(#[from] FromArrowError), - #[error("arrow error: {0}")] - Arrow(#[from] ArrowError), - #[error("cannot parse address: {0}")] - AddrParse(#[from] AddrParseError), - #[error("tonic transport error: {0}")] - TonicTransport(#[from] transport::Error), - #[error("default adapter cannot handle arrow ingest message")] - CannotHandleArrowMessage, - #[error("arrow adapter cannot handle default ingest message")] - CannotHandleDefaultMessage, - #[error("schema not found: {0}")] - SchemaNotFound(String), - #[error("record is not properly formed. Length of values {values_count} does not match schema: {schema_fields_count}")] - NumFieldsMismatch { - values_count: usize, - schema_fields_count: usize, - }, - #[error("data is not valid at index: {index}, Type: {value:?}, Expected Type: {field_type}")] - FieldTypeMismatch { - index: usize, - value: grpc_types::types::value::Value, - field_type: FieldType, - }, -} diff --git a/dozer-ingestion/grpc/src/tests.rs b/dozer-ingestion/grpc/src/tests.rs deleted file mode 100644 index 99ce956664..0000000000 --- a/dozer-ingestion/grpc/src/tests.rs +++ /dev/null @@ -1,259 +0,0 @@ -use std::collections::HashMap; -use std::{sync::Arc, thread}; - -use dozer_ingestion_connector::dozer_types::{ - arrow::array::{Int32Array, StringArray}, - arrow::{datatypes as arrow_types, record_batch::RecordBatch}, - arrow_types::from_arrow::serialize_record_batch, - arrow_types::to_arrow::DOZER_SCHEMA_KEY, - grpc_types::{ - ingest::{ingest_service_client::IngestServiceClient, IngestArrowRequest, IngestRequest}, - types, - }, - json_types::json as dozer_json, - models::ingestion_types::IngestionMessage, - models::ingestion_types::{ConfigSchemas, GrpcConfig}, - serde_json, - serde_json::json, - serde_json::Value, - tonic::transport::Channel, - types::Operation, - types::{FieldDefinition, FieldType, Schema as DozerSchema, SourceDefinition}, -}; -use dozer_ingestion_connector::test_util::{create_test_runtime, spawn_connector_all_tables}; -use dozer_ingestion_connector::tokio::runtime::Runtime; -use dozer_ingestion_connector::{dozer_types, IngestionIterator}; - -use crate::{ArrowAdapter, DefaultAdapter}; - -use super::connector::GrpcConnector; -use super::IngestAdapter; - -fn ingest_grpc( - runtime: Arc, - schemas: Value, - adapter: String, - port: u32, -) -> (IngestServiceClient, IngestionIterator) { - let grpc_connector = GrpcConnector::::new( - "grpc".to_string(), - GrpcConfig { - schemas: ConfigSchemas::Inline(schemas.to_string()), - adapter: Some(adapter), - port: Some(port), - host: None, - }, - ); - - let (iterator, _) = spawn_connector_all_tables(runtime.clone(), grpc_connector); - - let retries = 10; - let url = format!("http://0.0.0.0:{port}"); - let mut res = runtime.block_on(IngestServiceClient::connect(url.clone())); - for r in 0..retries { - if res.is_ok() { - break; - } - if r == retries - 1 { - panic!("failed to connect after {r} times"); - } - thread::sleep(std::time::Duration::from_millis(300)); - res = runtime.block_on(IngestServiceClient::connect(url.clone())); - } - - (res.unwrap(), iterator) -} - -#[test] -fn ingest_grpc_default() { - let runtime = create_test_runtime(); - let schemas = json!({ - "users": { - "schema": { - "fields": [ - { - "name": "id", - "typ": "Int", - "nullable": false - }, - { - "name": "name", - "typ": "String", - "nullable": true - } - ] - } - } - }); - - let (mut ingest_client, mut iterator) = - ingest_grpc::(runtime.clone(), schemas, "default".to_string(), 45678); - - // Ingest a record - runtime - .block_on(ingest_client.ingest(IngestRequest { - schema_name: "users".to_string(), - new: vec![ - types::Value { - value: Some(types::value::Value::IntValue(1675)), - }, - types::Value { - value: Some(types::value::Value::StringValue("dario".to_string())), - }, - ], - seq_no: 1, - ..Default::default() - })) - .unwrap(); - - let msg = iterator.next().unwrap(); - - if let IngestionMessage::OperationEvent { op, .. } = msg { - if let Operation::Insert { new: record } = op { - assert_eq!(record.values[0].as_int(), Some(1675)); - assert_eq!(record.values[1].as_string(), Some("dario")); - } else { - panic!("wrong operation kind"); - } - } else { - panic!("wrong message kind"); - } -} - -#[test] -#[ignore] -fn test_serialize_arrow_schema() { - let schema = arrow_types::Schema::new(vec![ - arrow_types::Field::new("id", arrow_types::DataType::Int32, false), - arrow_types::Field::new( - "time", - arrow_types::DataType::Timestamp( - arrow_types::TimeUnit::Millisecond, - Some("SGT".into()), - ), - false, - ), - ]); - - let str = dozer_types::serde_json::to_string(&schema).unwrap(); - println!("{str}"); -} - -#[test] -fn ingest_grpc_arrow() { - let runtime = create_test_runtime(); - let schema_str = serde_json::to_string( - &DozerSchema::default() - .field( - FieldDefinition::new( - "id".to_string(), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - true, - ) - .field( - FieldDefinition::new( - "name".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "json".to_string(), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - ) - .expect("Schema can always be serialized as JSON"); - - let schemas = json!([{ - "name": "users", - "schema": { - "fields": [ - { - "name": "id", - "data_type": "Int32", - "nullable": false, - "dict_id": 0, - "dict_is_ordered": false, - "metadata": {} - }, - { - "name": "name", - "data_type": "Utf8", - "nullable": true, - "dict_id": 0, - "dict_is_ordered": false, - "metadata": {} - }, - { - "name": "json", - "data_type": "Utf8", - "nullable": true, - "dict_id": 0, - "dict_is_ordered": false, - "metadata": {} - } - ], - "metadata": { DOZER_SCHEMA_KEY.to_string() : schema_str } - } - }]); - - let (mut ingest_client, mut iterator) = - ingest_grpc::(runtime.clone(), schemas, "arrow".to_string(), 45679); - - // Ingest a record - let schema = arrow_types::Schema::new_with_metadata( - vec![ - arrow_types::Field::new("id", arrow_types::DataType::Int32, false), - arrow_types::Field::new("name", arrow_types::DataType::Utf8, false), - arrow_types::Field::new("json", arrow_types::DataType::Utf8, false), - ], - HashMap::from([(DOZER_SCHEMA_KEY.to_string(), schema_str)]), - ); - - let a = Int32Array::from_iter([1675, 1676, 1677]); - let b = StringArray::from_iter_values(vec!["dario", "mario", "vario"]); - let c = StringArray::from_iter_values(vec!["[1, 2, 3]", "{\"a\": \"b\"}", "\"s\""]); - - let record_batch = RecordBatch::try_new( - Arc::new(schema), - vec![Arc::new(a), Arc::new(b), Arc::new(c)], - ) - .unwrap(); - - runtime - .block_on(ingest_client.ingest_arrow(IngestArrowRequest { - schema_name: "users".to_string(), - records: serialize_record_batch(&record_batch), - seq_no: 1, - ..Default::default() - })) - .unwrap(); - - let msg = iterator.next().unwrap(); - - if let IngestionMessage::OperationEvent { op, .. } = msg { - if let Operation::Insert { new: record } = op { - assert_eq!(record.values[0].as_int(), Some(1675)); - assert_eq!(record.values[1].as_string(), Some("dario")); - assert_eq!( - record.values[2].as_json(), - Some(&dozer_json!([1_f64, 2_f64, 3_f64])) - ); - } else { - panic!("wrong operation kind"); - } - } else { - panic!("wrong message kind"); - } -} diff --git a/dozer-ingestion/javascript/Cargo.toml b/dozer-ingestion/javascript/Cargo.toml deleted file mode 100644 index c29f92c856..0000000000 --- a/dozer-ingestion/javascript/Cargo.toml +++ /dev/null @@ -1,15 +0,0 @@ -[package] -name = "dozer-ingestion-javascript" -version = "0.4.0" -edition = "2021" -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-ingestion-connector = { path = "../connector" } -dozer-deno = { path = "../../dozer-deno" } -deno_core = { workspace = true } - -[dev-dependencies] -camino = "1.1.6" diff --git a/dozer-ingestion/javascript/src/js_extension/ingest.js b/dozer-ingestion/javascript/src/js_extension/ingest.js deleted file mode 100644 index ee9899b8ae..0000000000 --- a/dozer-ingestion/javascript/src/js_extension/ingest.js +++ /dev/null @@ -1,18 +0,0 @@ -(async () => { - const url = 'https://api.github.com/repos/getdozer/dozer/commits'; - const response = await fetch(url); - - const commits = await response.json(); - - const snapshot_msg = { typ: "SnapshottingDone", old_val: null, new_val: null }; - await Deno[Deno.internal].core.ops.ingest(snapshot_msg); - - for (const commit of commits) { - const msg = { - typ: "Insert", - old_val: null, - new_val: { commit: commit.sha }, - }; - await Deno[Deno.internal].core.ops.ingest(msg); - } -})(); diff --git a/dozer-ingestion/javascript/src/js_extension/mod.rs b/dozer-ingestion/javascript/src/js_extension/mod.rs deleted file mode 100644 index 57c52fd959..0000000000 --- a/dozer-ingestion/javascript/src/js_extension/mod.rs +++ /dev/null @@ -1,159 +0,0 @@ -use std::{future::Future, sync::Arc}; - -use deno_core::*; - -use dozer_deno::JsWorker; -use dozer_ingestion_connector::{ - dozer_types::{ - errors::{internal::BoxedError, types::DeserializationError}, - json_types::serde_json_to_json_value, - models::ingestion_types::{IngestionMessage, TransactionInfo}, - serde::{Deserialize, Serialize}, - serde_json, - thiserror::{self, Error}, - types::{Field, Operation, Record}, - }, - tokio::{runtime::Runtime, task::LocalSet}, - Ingestor, -}; - -#[derive(Deserialize, Serialize, Debug)] -#[serde(crate = "dozer_ingestion_connector::dozer_types::serde")] -pub enum MsgType { - SnapshottingStarted, - SnapshottingDone, - Insert, - Delete, - Update, -} - -#[derive(Deserialize, Serialize, Debug)] -#[serde(crate = "dozer_ingestion_connector::dozer_types::serde")] -pub struct JsMessage { - typ: MsgType, - old_val: serde_json::Value, - new_val: serde_json::Value, -} - -#[op2(async)] -fn ingest( - #[state] ingestor: &Ingestor, - #[serde] val: JsMessage, -) -> impl Future> { - send(ingestor.clone(), val) -} - -extension!( - dozer_extension, - ops = [ingest], - options = { ingestor: Ingestor }, - state = |state, options| { - state.put(options.ingestor); - }, -); - -pub struct JsExtension { - runtime: Arc, - ingestor: Ingestor, - module_specifier: ModuleSpecifier, -} - -#[derive(Debug, Error)] -pub enum JsExtensionError { - #[error("Failed to canonicalize path {0}: {1}")] - CanonicalizePath(String, #[source] std::io::Error), -} - -impl JsExtension { - pub fn new( - runtime: Arc, - ingestor: Ingestor, - js_path: String, - ) -> Result { - let path = std::fs::canonicalize(js_path.clone()) - .map_err(|e| JsExtensionError::CanonicalizePath(js_path, e))?; - let module_specifier = - ModuleSpecifier::from_file_path(path).expect("we just canonicalized it"); - Ok(Self { - runtime, - ingestor, - module_specifier, - }) - } - - pub async fn run(self) -> Result<(), BoxedError> { - let runtime = self.runtime.clone(); - runtime - .spawn_blocking(move || { - let local_set = LocalSet::new(); - local_set.block_on(&self.runtime, async move { - let mut worker = JsWorker::new(vec![dozer_extension::init_ops(self.ingestor)])?; - worker.execute_main_module(&self.module_specifier).await?; - worker.js_runtime.run_event_loop(Default::default()).await - }) - }) - .await - .unwrap() // Propagate panics. - .map_err(Into::into) - } -} - -async fn send(ingestor: Ingestor, val: JsMessage) -> Result<(), anyhow::Error> { - let msg = match val.typ { - MsgType::SnapshottingStarted => { - IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingStarted) - } - MsgType::SnapshottingDone => { - IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingDone { id: None }) - } - MsgType::Insert | MsgType::Delete | MsgType::Update => { - let op = map_operation(val)?; - IngestionMessage::OperationEvent { - table_index: 0, - op, - id: None, - } - } - }; - - // Ignore if the receiver is closed. - let _ = ingestor.handle_message(msg).await; - let _ = ingestor - .handle_message(IngestionMessage::TransactionInfo(TransactionInfo::Commit { - id: None, - source_time: None, - })) - .await; - Ok(()) -} - -fn map_operation(msg: JsMessage) -> Result { - Ok(match msg.typ { - MsgType::Insert => Operation::Insert { - new: Record { - values: vec![Field::Json(serde_json_to_json_value(msg.new_val)?)], - lifetime: None, - }, - }, - MsgType::Delete => Operation::Delete { - old: Record { - values: vec![Field::Json(serde_json_to_json_value(msg.old_val)?)], - lifetime: None, - }, - }, - MsgType::Update => Operation::Update { - old: Record { - values: vec![Field::Json(serde_json_to_json_value(msg.old_val)?)], - lifetime: None, - }, - new: Record { - values: vec![Field::Json(serde_json_to_json_value(msg.new_val)?)], - lifetime: None, - }, - }, - _ => unreachable!(), - }) -} - -#[cfg(test)] -mod tests; diff --git a/dozer-ingestion/javascript/src/js_extension/tests.rs b/dozer-ingestion/javascript/src/js_extension/tests.rs deleted file mode 100644 index b4f0a5d9bc..0000000000 --- a/dozer-ingestion/javascript/src/js_extension/tests.rs +++ /dev/null @@ -1,31 +0,0 @@ -use std::time::Duration; - -use camino::Utf8Path; -use dozer_ingestion_connector::{test_util::create_test_runtime, IngestionConfig, Ingestor}; - -use super::JsExtension; - -#[test] -#[ignore = "this test fails if it runs together with tests in `dozer-deno`. Not sure why."] -fn test_deno() { - let js_path = Utf8Path::new(env!("CARGO_MANIFEST_DIR")).join("./src/js_extension/ingest.js"); - - let (ingestor, mut iterator) = Ingestor::initialize_channel(IngestionConfig::default()); - - let runtime = create_test_runtime(); - let ext = JsExtension::new(runtime.clone(), ingestor, js_path.into()).unwrap(); - - runtime.spawn(ext.run()); - - runtime.block_on(async move { - let mut count = 0; - loop { - let msg = iterator.next_timeout(Duration::from_secs(5)).await.unwrap(); - count += 1; - println!("i: {:?}, msg: {:?}", count, msg); - if count > 3 { - break; - } - } - }); -} diff --git a/dozer-ingestion/javascript/src/lib.rs b/dozer-ingestion/javascript/src/lib.rs deleted file mode 100644 index ba9fef603a..0000000000 --- a/dozer-ingestion/javascript/src/lib.rs +++ /dev/null @@ -1,105 +0,0 @@ -use std::sync::Arc; - -use dozer_ingestion_connector::{ - async_trait, - dozer_types::{ - errors::internal::BoxedError, - models::ingestion_types::{default_bootstrap_path, JavaScriptConfig}, - node::OpIdentifier, - types::{FieldDefinition, FieldType, Schema, SourceDefinition}, - }, - tokio::runtime::Runtime, - CdcType, Connector, Ingestor, SourceSchema, SourceSchemaResult, TableIdentifier, TableInfo, -}; -use js_extension::JsExtension; - -#[derive(Debug)] -pub struct JavaScriptConnector { - runtime: Arc, - config: JavaScriptConfig, -} - -#[async_trait] -impl Connector for JavaScriptConnector { - // We will return one field, named "value", of type Json to - // give maximum flexibility to the user. - fn types_mapping() -> Vec<(String, Option)> - where - Self: Sized, - { - vec![(String::from("value"), Some(FieldType::Json))] - } - - async fn validate_connection(&mut self) -> Result<(), BoxedError> { - Ok(()) - } - - async fn list_tables(&mut self) -> Result, BoxedError> { - Ok(vec![TableIdentifier { - schema: None, - name: "json_records".to_string(), - }]) - } - - async fn validate_tables(&mut self, _tables: &[TableIdentifier]) -> Result<(), BoxedError> { - Ok(()) - } - - async fn list_columns( - &mut self, - _tables: Vec, - ) -> Result, BoxedError> { - Ok(vec![TableInfo { - schema: None, - name: "json_records".to_string(), - column_names: vec!["value".to_string()], - }]) - } - - async fn get_schemas( - &mut self, - _table_infos: &[TableInfo], - ) -> Result, BoxedError> { - Ok(vec![Ok(SourceSchema { - schema: Schema { - fields: vec![FieldDefinition { - name: "value".to_string(), - typ: FieldType::Json, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }], - primary_index: vec![], - }, - cdc_type: CdcType::Nothing, - })]) - } - - async fn serialize_state(&self) -> Result, BoxedError> { - Ok(vec![]) - } - - async fn start( - &mut self, - ingestor: &Ingestor, - _tables: Vec, - _last_checkpoint: Option, - ) -> Result<(), BoxedError> { - let js_path = self - .config - .bootstrap_path - .clone() - .unwrap_or_else(default_bootstrap_path); - let ingestor = ingestor.clone(); - let ext = JsExtension::new(self.runtime.clone(), ingestor, js_path)?; - ext.run().await - } -} - -impl JavaScriptConnector { - pub fn new(runtime: Arc, config: JavaScriptConfig) -> Self { - Self { runtime, config } - } -} - -mod js_extension; diff --git a/dozer-ingestion/kafka/Cargo.toml b/dozer-ingestion/kafka/Cargo.toml deleted file mode 100644 index ed4dc987e6..0000000000 --- a/dozer-ingestion/kafka/Cargo.toml +++ /dev/null @@ -1,13 +0,0 @@ -[package] -name = "dozer-ingestion-kafka" -version = "0.4.0" -edition = "2021" -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-ingestion-connector = { path = "../connector" } -rdkafka = "0.36.0" -schema_registry_converter = { version = "4.0.0", features = ["avro"] } -base64 = "0.21.0" diff --git a/dozer-ingestion/kafka/src/connector.rs b/dozer-ingestion/kafka/src/connector.rs deleted file mode 100644 index e48d672d58..0000000000 --- a/dozer-ingestion/kafka/src/connector.rs +++ /dev/null @@ -1,176 +0,0 @@ -use dozer_ingestion_connector::async_trait; -use dozer_ingestion_connector::dozer_types::errors::internal::BoxedError; -use dozer_ingestion_connector::dozer_types::models::ingestion_types::KafkaConfig; -use dozer_ingestion_connector::dozer_types::node::OpIdentifier; -use dozer_ingestion_connector::dozer_types::types::FieldType; -use dozer_ingestion_connector::Connector; -use dozer_ingestion_connector::Ingestor; -use dozer_ingestion_connector::SourceSchema; -use dozer_ingestion_connector::SourceSchemaResult; -use dozer_ingestion_connector::TableIdentifier; -use dozer_ingestion_connector::TableInfo; -use rdkafka::consumer::BaseConsumer; -use rdkafka::consumer::Consumer; -use rdkafka::util::Timeout; -use rdkafka::ClientConfig; - -use crate::no_schema_registry_basic::NoSchemaRegistryBasic; -use crate::schema_registry_basic::SchemaRegistryBasic; -use crate::stream_consumer::StreamConsumer; -use crate::stream_consumer_basic::StreamConsumerBasic; -use crate::KafkaError; - -#[derive(Debug)] -pub struct KafkaConnector { - config: KafkaConfig, -} - -impl KafkaConnector { - pub fn new(config: KafkaConfig) -> Self { - Self { config } - } - - async fn get_schemas_impl( - &self, - table_names: Option<&[String]>, - ) -> Result, KafkaError> { - if let Some(schema_registry_url) = &self.config.schema_registry_url { - SchemaRegistryBasic::get_schema(table_names, schema_registry_url.clone()).await - } else { - NoSchemaRegistryBasic::get_schema(table_names) - } - } -} - -#[async_trait] -impl Connector for KafkaConnector { - fn types_mapping() -> Vec<(String, Option)> - where - Self: Sized, - { - todo!() - } - - async fn validate_connection(&mut self) -> Result<(), BoxedError> { - Ok(()) - } - - async fn list_tables(&mut self) -> Result, BoxedError> { - let consumer = ClientConfig::new() - .set("bootstrap.servers", &self.config.broker.clone()) - .set("api.version.request", "true") - .create::()?; - - let metadata = - consumer.fetch_metadata(None, Timeout::After(std::time::Duration::new(60, 0)))?; - let topics = metadata.topics(); - - let mut tables = vec![]; - for topic in topics { - tables.push(TableIdentifier { - schema: None, - name: topic.name().to_string(), - }); - } - - Ok(tables) - } - - async fn validate_tables(&mut self, tables: &[TableIdentifier]) -> Result<(), BoxedError> { - let table_names = tables - .iter() - .map(|table| table.name.clone()) - .collect::>(); - self.get_schemas_impl(Some(&table_names)).await?; - Ok(()) - } - - async fn list_columns( - &mut self, - tables: Vec, - ) -> Result, BoxedError> { - let table_names = tables - .iter() - .map(|table| table.name.clone()) - .collect::>(); - let schemas = self.get_schemas_impl(Some(&table_names)).await?; - let mut result = vec![]; - for (table, schema) in tables.into_iter().zip(schemas) { - let column_names = schema - .schema - .fields - .into_iter() - .map(|field| field.name) - .collect(); - result.push(TableInfo { - schema: table.schema, - name: table.name, - column_names, - }); - } - Ok(result) - } - - async fn get_schemas( - &mut self, - table_infos: &[TableInfo], - ) -> Result, BoxedError> { - let table_names = table_infos - .iter() - .map(|table| table.name.clone()) - .collect::>(); - Ok(self - .get_schemas_impl(Some(&table_names)) - .await? - .into_iter() - .map(Ok) - .collect()) - } - - async fn serialize_state(&self) -> Result, BoxedError> { - Ok(vec![]) - } - - async fn start( - &mut self, - ingestor: &Ingestor, - tables: Vec, - last_checkpoint: Option, - ) -> Result<(), BoxedError> { - let broker = self.config.broker.to_owned(); - run( - broker, - tables, - last_checkpoint, - ingestor, - &self.config.schema_registry_url, - ) - .await - .map_err(Into::into) - } -} - -async fn run( - broker: String, - tables: Vec, - last_checkpoint: Option, - ingestor: &Ingestor, - schema_registry_url: &Option, -) -> Result<(), KafkaError> { - let mut client_config = ClientConfig::new(); - client_config - .set("bootstrap.servers", broker) - .set("group.id", "dozer") - .set("enable.auto.commit", "true"); - - let consumer = StreamConsumerBasic::default(); - consumer - .run( - client_config, - ingestor, - tables, - last_checkpoint, - schema_registry_url, - ) - .await -} diff --git a/dozer-ingestion/kafka/src/debezium/mapper.rs b/dozer-ingestion/kafka/src/debezium/mapper.rs deleted file mode 100644 index 8f496f18a6..0000000000 --- a/dozer-ingestion/kafka/src/debezium/mapper.rs +++ /dev/null @@ -1,452 +0,0 @@ -use base64::{engine, Engine}; -use dozer_ingestion_connector::dozer_types::{ - serde_json::Value, - types::{Field, Schema}, -}; -use std::collections::HashMap; - -use crate::KafkaSchemaError; - -use super::stream_consumer::DebeziumSchemaStruct; - -// fn convert_decimal(value: &str, scale: u32) -> Result { -// let decoded_value = engine::general_purpose::STANDARD -// .decode(value) -// .map_err(BinaryDecodeError) -// .unwrap(); -// -// let mut multiplier: u64 = 1; -// let mut result: u64 = 0; -// decoded_value.iter().rev().for_each(|w| { -// let number = *w as u64; -// result += number * multiplier; -// multiplier *= 256; -// }); -// -// Ok(Field::from( -// Decimal::try_new(result as i64, scale).map_err(DecimalConvertError)?, -// )) -// } - -fn convert_value(value: Value, schema: &DebeziumSchemaStruct) -> Result { - // match schema.name.clone() { - match schema.r#type.clone() { - Value::String(typ) => match typ.as_str() { - "int" | "int8" | "int16" | "int32" | "int64" => value - .as_i64() - .map_or(Ok(Field::Null), |v| Ok(Field::from(v))), - "string" => value - .as_str() - .map_or(Ok(Field::Null), |s| Ok(Field::from(s.to_string()))), - "bytes" => value.as_str().map_or(Ok(Field::Null), |s| { - Ok(Field::Binary( - engine::general_purpose::STANDARD - .decode(s) - .map_err(KafkaSchemaError::BinaryDecodeError)?, - )) - }), - "float" | "float32" | "float64" | "double" => value - .as_f64() - .map_or(Ok(Field::Null), |s| Ok(Field::from(s))), - "boolean" => value - .as_bool() - .map_or(Ok(Field::Null), |s| Ok(Field::from(s))), - _ => Err(KafkaSchemaError::TypeNotSupported(typ)), - }, - _ => Err(KafkaSchemaError::TypeNotSupported( - "Unexpected value type".to_string(), - )), - } - // Some(name) => { - // match name.as_str() { - // "io.debezium.time.MicroTimestamp" => value.as_i64().map_or(Ok(Field::Null), |v| { - // let sec = v / 1000000; - // let nsecs = (v % 1000000) * 1000; - // let date = NaiveDateTime::from_timestamp_opt(sec, nsecs as u32) - // .map_or_else(|| Err(InvalidTimestampError), Ok)?; - // Ok(Field::from(date)) - // }), - // "io.debezium.time.Timestamp" | "org.apache.kafka.connect.data.Timestamp" => { - // value.as_i64().map_or(Ok(Field::Null), |v| { - // let sec = v / 1000; - // let nsecs = (v % 1000) * 1000000; - // let date = NaiveDateTime::from_timestamp_opt(sec, nsecs as u32) - // .map_or_else(|| Err(InvalidTimestampError), Ok)?; - // Ok(Field::from(date)) - // }) - // } - // "org.apache.kafka.connect.data.Decimal" => { - // let parameters = schema.parameters.as_ref().ok_or(ScaleNotFound)?; - // let scale: u32 = parameters - // .scale - // .as_ref() - // .ok_or(ScaleNotFound)? - // .parse() - // .map_err(|_| ScaleIsInvalid)?; - // value - // .as_str() - // .map_or(Ok(Field::Null), |value| convert_decimal(value, scale)) - // } - // "io.debezium.data.VariableScaleDecimal" => { - // value.as_object().map_or(Ok(Field::Null), |map| { - // let scale: u32 = map - // .get("scale") - // .ok_or(ScaleNotFound)? - // .as_u64() - // .ok_or(ScaleIsInvalid)? as u32; - // map.get("value").map_or(Ok(Field::Null), |dec_val| { - // dec_val - // .as_str() - // .map_or(Ok(Field::Null), |v| convert_decimal(v, scale)) - // }) - // }) - // } - // "io.debezium.time.Date" | "org.apache.kafka.connect.data.Date" => { - // value.as_i64().map_or(Ok(Field::Null), |v| { - // Ok(Field::from( - // NaiveDate::from_num_days_from_ce_opt(v as i32) - // .map_or_else(|| Err(InvalidDateError), Ok)?, - // )) - // }) - // } - // "io.debezium.time.MicroTime" => Ok(Field::Null), - // "io.debezium.data.Json" => value.as_str().map_or(Ok(Field::Null), |s| { - // Ok(Field::Json( - // JsonValue::from_str(s).map_err(|e| InvalidJsonError(e.to_string()))?, - // )) - // }), - // // | "io.debezium.time.MicroTime" | "org.apache.kafka.connect.data.Time" => Ok(FieldType::Timestamp), - // _ => Err(TypeNotSupported(name)), - // } - // } - // } -} - -pub fn convert_value_to_schema( - value: Value, - schema: &Schema, - fields_map: &HashMap, -) -> Result, KafkaSchemaError> { - schema - .fields - .iter() - .map(|f| match value.get(f.name.clone()).cloned() { - None => Ok(Field::Null), - Some(field_value) => { - let schema_struct = fields_map - .get(&f.name) - .ok_or_else(|| KafkaSchemaError::FieldNotFound(f.name.clone()))?; - convert_value(field_value, schema_struct) - } - }) - .collect() -} - -#[cfg(test)] -mod tests { - use dozer_ingestion_connector::dozer_types::{ - chrono::NaiveDateTime, - serde_json::Map, - types::{FieldDefinition, FieldType, SourceDefinition}, - }; - - use super::*; - - #[macro_export] - macro_rules! test_conversion_debezium { - ($a:expr,$b:expr,$c:expr,$d:expr,$e:expr) => { - let value = convert_value( - Value::from($a), - &&DebeziumSchemaStruct { - r#type: Value::String($b.to_string()), - fields: None, - optional: Some(false), - name: $c, - field: None, - version: None, - parameters: $e, - }, - ); - assert_eq!(value.unwrap(), $d); - }; - } - - macro_rules! test_conversion_debezium_error { - ($a:expr,$b:expr,$c:expr,$d:expr,$e:expr) => { - let actual_error = convert_value( - Value::from($a), - &&DebeziumSchemaStruct { - r#type: Value::String($b.to_string()), - fields: None, - optional: Some(false), - name: $c, - field: None, - version: None, - parameters: $e, - }, - ) - .unwrap_err(); - assert_eq!(actual_error, $d); - }; - } - #[test] - fn it_converts_value_to_field() { - test_conversion_debezium!(159, "int8", None, Field::from(159), None); - test_conversion_debezium!( - "ABC165", - "string", - None, - Field::from("ABC165".to_string()), - None - ); - // MTIzNA== -> 1234 - test_conversion_debezium!( - "MTIzNA==", - "bytes", - None, - Field::Binary(vec![49, 50, 51, 52]), - None - ); - test_conversion_debezium!(16.8, "float32", None, Field::from(16.8), None); - test_conversion_debezium!(false, "boolean", None, Field::from(false), None); - let _current_date = - NaiveDateTime::parse_from_str("2022-11-28 16:55:43", "%Y-%m-%d %H:%M:%S").unwrap(); - // test_conversion_debezium!( - // 1669654543000000_i64, - // "-", - // Some("io.debezium.time.MicroTimestamp".to_string()), - // Field::from(current_date), - // None - // ); - // test_conversion_debezium!( - // 1669654543000_i64, - // "-", - // Some("io.debezium.time.Timestamp".to_string()), - // Field::from(current_date), - // None - // ); - - // 4 x 256 + 210 = 1234 - // test_conversion_debezium!( - // engine::general_purpose::STANDARD.encode(vec![4, 210]), - // "-", - // Some("org.apache.kafka.connect.data.Decimal".to_string()), - // Field::from(rust_decimal::Decimal::new(1234, 2)), - // Some(DebeziumSchemaParameters { - // scale: Some(2.to_string()), - // precision: None - // }) - // ); - - let mut v: Map = Map::new(); - v.insert( - "value".to_string(), - Value::from(engine::general_purpose::STANDARD.encode(vec![4, 211])), - ); - v.insert("scale".to_string(), Value::from(2_u64)); - // test_conversion_debezium!( - // v, - // "-", - // Some("io.debezium.data.VariableScaleDecimal".to_string()), - // Field::from(rust_decimal::Decimal::new(1235, 2)), - // None - // ); - // - // let current_date = NaiveDate::from_ymd_opt(2022, 11, 28).unwrap(); - // test_conversion_debezium!( - // 738487, - // "-", - // Some("io.debezium.time.Date".to_string()), - // Field::from(current_date), - // None - // ); - // test_conversion_debezium!( - // "{\"abc\":123}", - // "-", - // Some("io.debezium.data.Json".to_string()), - // Field::Json(JsonValue::Object(BTreeMap::from([( - // String::from("abc"), - // JsonValue::Number(OrderedFloat(123_f64)) - // )]))), - // None - // ); - } - - #[test] - fn it_converts_value_to_field_error() { - test_conversion_debezium_error!( - "Unknown type value", - "Unknown type", - None, - KafkaSchemaError::TypeNotSupported("Unknown type".to_string()), - None - ); - // test_conversion_debezium_error!( - // 1234, - // "-", - // Some("org.apache.kafka.connect.data.Decimal".to_string()), - // crate::errors::DebeziumSchemaError::ScaleNotFound, - // None - // ); - // test_conversion_debezium_error!( - // 1234, - // "-", - // Some("org.apache.kafka.connect.data.Decimal".to_string()), - // crate::errors::DebeziumSchemaError::ScaleIsInvalid, - // Some(DebeziumSchemaParameters { - // scale: Some("ABCD".to_string()), - // precision: None - // }) - // ); - } - - #[test] - fn it_converts_value_to_schema() { - let mut v: Map = Map::new(); - v.insert("id".to_string(), Value::from(1)); - v.insert("name".to_string(), Value::from("Product")); - v.insert("description".to_string(), Value::from("Description")); - v.insert("weight".to_string(), Value::from(12.34)); - - let value = Value::from(v); - - let schema = Schema { - fields: vec![ - FieldDefinition { - name: "id".to_string(), - typ: FieldType::Int, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "name".to_string(), - typ: FieldType::String, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "description".to_string(), - typ: FieldType::String, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "weight".to_string(), - typ: FieldType::Float, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - ], - primary_index: vec![], - }; - - let mut fields_map: HashMap = HashMap::new(); - let id_struct = DebeziumSchemaStruct { - r#type: Value::String("int64".to_string()), - fields: None, - optional: Some(false), - name: None, - field: None, - version: None, - parameters: None, - }; - fields_map.insert("id".to_string(), id_struct); - let name_struct = DebeziumSchemaStruct { - r#type: Value::String("string".to_string()), - fields: None, - optional: Some(false), - name: None, - field: None, - version: None, - parameters: None, - }; - fields_map.insert("name".to_string(), name_struct); - let description_struct = DebeziumSchemaStruct { - r#type: Value::String("string".to_string()), - fields: None, - optional: Some(false), - name: None, - field: None, - version: None, - parameters: None, - }; - fields_map.insert("description".to_string(), description_struct); - let weight_struct = DebeziumSchemaStruct { - r#type: Value::String("float64".to_string()), - fields: None, - optional: Some(false), - name: None, - field: None, - version: None, - parameters: None, - }; - fields_map.insert("weight".to_string(), weight_struct); - - let fields = convert_value_to_schema(value, &schema, &fields_map).unwrap(); - assert_eq!(*fields.first().unwrap(), Field::from(1)); - assert_eq!(*fields.get(1).unwrap(), Field::from("Product".to_string())); - assert_eq!( - *fields.get(2).unwrap(), - Field::from("Description".to_string()) - ); - assert_eq!(*fields.get(3).unwrap(), Field::from(12.34)); - } - - #[test] - fn it_converts_null_value_to_schema() { - let mut v: Map = Map::new(); - v.insert("id".to_string(), Value::from(1)); - - let value = Value::from(v); - - let schema = Schema { - fields: vec![ - FieldDefinition { - name: "id".to_string(), - typ: FieldType::Int, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "name".to_string(), - typ: FieldType::String, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }, - ], - primary_index: vec![], - }; - - let mut fields_map: HashMap = HashMap::new(); - let id_struct = DebeziumSchemaStruct { - r#type: Value::String("int64".to_string()), - fields: None, - optional: Some(false), - name: None, - field: None, - version: None, - parameters: None, - }; - fields_map.insert("id".to_string(), id_struct); - let name_struct = DebeziumSchemaStruct { - r#type: Value::String("string".to_string()), - fields: None, - optional: Some(false), - name: None, - field: None, - version: None, - parameters: None, - }; - fields_map.insert("name".to_string(), name_struct); - - let fields = convert_value_to_schema(value, &schema, &fields_map).unwrap(); - assert_eq!(*fields.first().unwrap(), Field::from(1)); - assert_eq!(*fields.get(1).unwrap(), Field::Null); - } -} diff --git a/dozer-ingestion/kafka/src/debezium/mod.rs b/dozer-ingestion/kafka/src/debezium/mod.rs deleted file mode 100644 index 35a9de8f05..0000000000 --- a/dozer-ingestion/kafka/src/debezium/mod.rs +++ /dev/null @@ -1,5 +0,0 @@ -pub mod mapper; -pub mod no_schema_registry; -pub mod schema; -pub mod schema_registry; -pub mod stream_consumer; diff --git a/dozer-ingestion/kafka/src/debezium/no_schema_registry.rs b/dozer-ingestion/kafka/src/debezium/no_schema_registry.rs deleted file mode 100644 index dc5e2b528b..0000000000 --- a/dozer-ingestion/kafka/src/debezium/no_schema_registry.rs +++ /dev/null @@ -1,63 +0,0 @@ -use dozer_ingestion_connector::dozer_types::serde_json; -use dozer_ingestion_connector::{CdcType, SourceSchema}; -use rdkafka::config::RDKafkaLogLevel; -use rdkafka::consumer::stream_consumer::StreamConsumer as RdkafkaStreamConsumer; -use rdkafka::consumer::{Consumer, DefaultConsumerContext}; -use rdkafka::{ClientConfig, Message}; - -use crate::{KafkaError, KafkaStreamError}; - -use super::schema::map_schema; -use super::stream_consumer::DebeziumMessage; - -pub struct NoSchemaRegistry {} - -impl NoSchemaRegistry { - pub async fn get_schema( - table_names: Option<&[String]>, - broker: String, - ) -> Result, KafkaError> { - let mut schemas = vec![]; - match table_names { - None => {} - Some(tables) => { - for table in tables { - let context = DefaultConsumerContext; - - let con: RdkafkaStreamConsumer = ClientConfig::new() - .set("bootstrap.servers", broker.clone()) - .set("enable.partition.eof", "false") - .set("session.timeout.ms", "6000") - .set("enable.auto.commit", "true") - .set_log_level(RDKafkaLogLevel::Debug) - .create_with_context(context)?; - - con.subscribe(&[table])?; - - let m = con.recv().await.map_err(|e| { - KafkaError::KafkaStreamError(KafkaStreamError::PollingError(e)) - })?; - - if let (Some(message), Some(key)) = (m.payload(), m.key()) { - let value_struct: DebeziumMessage = serde_json::from_str( - std::str::from_utf8(message).map_err(KafkaError::BytesConvertError)?, - ) - .map_err(KafkaError::JsonDecodeError)?; - let key_struct: DebeziumMessage = serde_json::from_str( - std::str::from_utf8(key).map_err(KafkaError::BytesConvertError)?, - ) - .map_err(KafkaError::JsonDecodeError)?; - - let (mapped_schema, _fields_map) = - map_schema(&value_struct.schema, &key_struct.schema) - .map_err(KafkaError::KafkaSchemaError)?; - - schemas.push(SourceSchema::new(mapped_schema, CdcType::FullChanges)); - } - } - } - } - - Ok(schemas) - } -} diff --git a/dozer-ingestion/kafka/src/debezium/schema.rs b/dozer-ingestion/kafka/src/debezium/schema.rs deleted file mode 100644 index 76ee9bb5a0..0000000000 --- a/dozer-ingestion/kafka/src/debezium/schema.rs +++ /dev/null @@ -1,310 +0,0 @@ -use std::collections::HashMap; - -use dozer_ingestion_connector::dozer_types::{ - serde_json::Value, - types::{FieldDefinition, FieldType, Schema, SourceDefinition}, -}; - -use crate::KafkaSchemaError; - -use super::stream_consumer::DebeziumSchemaStruct; - -// Reference: https://debezium.io/documentation/reference/0.9/connectors/postgresql.html -pub fn map_type(schema: &DebeziumSchemaStruct) -> Result { - match schema.name.clone() { - None => match schema.r#type.clone() { - Value::String(typ) => match typ.as_str() { - "int" | "int8" | "int16" | "int32" | "int64" => Ok(FieldType::Int), - "string" => Ok(FieldType::String), - "bytes" => Ok(FieldType::Binary), - "float" | "float32" | "float64" | "double" => Ok(FieldType::Float), - "boolean" => Ok(FieldType::Boolean), - _ => Err(KafkaSchemaError::TypeNotSupported(typ)), - }, - _ => Err(KafkaSchemaError::TypeNotSupported( - "Unexpected value type".to_string(), - )), - }, - Some(name) => match name.as_str() { - "io.debezium.time.MicroTime" - | "io.debezium.time.Timestamp" - | "io.debezium.time.MicroTimestamp" - | "org.apache.kafka.connect.data.Time" - | "org.apache.kafka.connect.data.Timestamp" => Ok(FieldType::Timestamp), - "io.debezium.time.Date" | "org.apache.kafka.connect.data.Date" => Ok(FieldType::Date), - "org.apache.kafka.connect.data.Decimal" | "io.debezium.data.VariableScaleDecimal" => { - Ok(FieldType::Decimal) - } - "io.debezium.data.Json" => Ok(FieldType::Json), - _ => Err(KafkaSchemaError::TypeNotSupported(name)), - }, - } -} - -pub fn map_schema( - schema: &DebeziumSchemaStruct, - key_schema: &DebeziumSchemaStruct, -) -> Result<(Schema, HashMap), KafkaSchemaError> { - let pk_fields = match &key_schema.fields { - None => vec![], - Some(fields) => fields.iter().map(|f| f.field.clone().unwrap()).collect(), - }; - - match &schema.fields { - None => Err(KafkaSchemaError::SchemaDefinitionNotFound), - Some(fields) => { - let new_schema_struct = fields.iter().find(|f| { - if let Some(val) = f.field.clone() { - val == *"after" - } else { - false - } - }); - - if let Some(schema) = new_schema_struct { - let mut pk_keys_indexes = vec![]; - let mut fields_schema_map: HashMap = HashMap::new(); - - let defined_fields: Result, _> = match &schema.fields { - None => Ok(vec![]), - Some(fields) => fields - .iter() - .enumerate() - .map(|(idx, f)| { - let typ = map_type(f)?; - let name = f.field.clone().unwrap(); - if pk_fields.contains(&name) { - pk_keys_indexes.push(idx); - } - fields_schema_map.insert(name.clone(), f.clone()); - Ok(FieldDefinition { - name, - typ, - nullable: f.optional.map_or(false, |o| o), - source: SourceDefinition::Dynamic, - description: None, - }) - }) - .collect(), - }; - - Ok(( - Schema { - fields: defined_fields?, - primary_index: pk_keys_indexes, - }, - fields_schema_map, - )) - } else { - Err(KafkaSchemaError::SchemaDefinitionNotFound) - } - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn test_it_fails_when_schema_empty() { - let schema = DebeziumSchemaStruct { - r#type: Value::String("empty".to_string()), - fields: None, - optional: Some(false), - name: None, - field: None, - version: None, - parameters: None, - }; - - let key_schema = DebeziumSchemaStruct { - r#type: Value::String("before".to_string()), - fields: None, - optional: Some(false), - name: None, - field: None, - version: None, - parameters: None, - }; - - let actual_error = map_schema(&schema, &key_schema).unwrap_err(); - assert_eq!(actual_error, KafkaSchemaError::SchemaDefinitionNotFound); - } - - #[test] - fn test_it_converts_schema() { - let schema = DebeziumSchemaStruct { - r#type: Value::String("empty".to_string()), - fields: Some(vec![DebeziumSchemaStruct { - r#type: Value::String("after".to_string()), - fields: Some(vec![ - DebeziumSchemaStruct { - r#type: Value::String("int32".to_string()), - fields: None, - optional: Some(false), - name: None, - field: Some("id".to_string()), - version: None, - parameters: None, - }, - DebeziumSchemaStruct { - r#type: Value::String("string".to_string()), - fields: None, - optional: Some(true), - name: None, - field: Some("name".to_string()), - version: None, - parameters: None, - }, - ]), - optional: Some(false), - name: None, - field: Some("after".to_string()), - version: None, - parameters: None, - }]), - optional: Some(false), - name: None, - field: Some("struct".to_string()), - version: None, - parameters: None, - }; - - let key_schema = DebeziumSchemaStruct { - r#type: Value::String("-".to_string()), - fields: Some(vec![DebeziumSchemaStruct { - r#type: Value::String("int32".to_string()), - fields: None, - optional: Some(false), - name: None, - field: Some("id".to_string()), - version: None, - parameters: None, - }]), - optional: Some(false), - name: None, - field: None, - version: None, - parameters: None, - }; - - let (schema, _) = map_schema(&schema, &key_schema).unwrap(); - let expected_schema = Schema { - fields: vec![ - FieldDefinition { - name: "id".to_string(), - typ: FieldType::Int, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "name".to_string(), - typ: FieldType::String, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }, - ], - primary_index: vec![0], - }; - assert_eq!(schema, expected_schema); - } - - #[test] - fn test_it_converts_empty_schema() { - let schema = DebeziumSchemaStruct { - r#type: Value::String("empty".to_string()), - fields: Some(vec![DebeziumSchemaStruct { - r#type: Value::String("after".to_string()), - fields: None, - optional: Some(false), - name: None, - field: Some("after".to_string()), - version: None, - parameters: None, - }]), - optional: Some(false), - name: None, - field: Some("struct".to_string()), - version: None, - parameters: None, - }; - - let key_schema = DebeziumSchemaStruct { - r#type: Value::String("-".to_string()), - fields: Some(vec![]), - optional: Some(false), - name: None, - field: None, - version: None, - parameters: None, - }; - - let (schema, _) = map_schema(&schema, &key_schema).unwrap(); - let expected_schema = Schema { - fields: vec![], - primary_index: vec![], - }; - assert_eq!(schema, expected_schema); - } - - macro_rules! test_map_type { - ($a:expr,$b:expr,$c:expr) => { - let schema = DebeziumSchemaStruct { - r#type: Value::String($a.to_string()), - fields: None, - optional: Some(false), - name: $b, - field: None, - version: None, - parameters: None, - }; - - let typ = map_type(&schema); - assert_eq!(typ, $c); - }; - } - - #[test] - fn test_map_type() { - test_map_type!("int8", None, Ok(FieldType::Int)); - test_map_type!("string", None, Ok(FieldType::String)); - test_map_type!("bytes", None, Ok(FieldType::Binary)); - test_map_type!("float32", None, Ok(FieldType::Float)); - test_map_type!("boolean", None, Ok(FieldType::Boolean)); - test_map_type!( - "not found", - None, - Err(KafkaSchemaError::TypeNotSupported("not found".to_string())) - ); - test_map_type!( - "int8", - Some("io.debezium.time.MicroTime".to_string()), - Ok(FieldType::Timestamp) - ); - test_map_type!( - "int8", - Some("io.debezium.time.Date".to_string()), - Ok(FieldType::Date) - ); - test_map_type!( - "int8", - Some("org.apache.kafka.connect.data.Decimal".to_string()), - Ok(FieldType::Decimal) - ); - test_map_type!( - "string", - Some("io.debezium.data.Json".to_string()), - Ok(FieldType::Json) - ); - test_map_type!( - "string", - Some("not existing".to_string()), - Err(KafkaSchemaError::TypeNotSupported( - "not existing".to_string() - )) - ); - } -} diff --git a/dozer-ingestion/kafka/src/debezium/schema_registry.rs b/dozer-ingestion/kafka/src/debezium/schema_registry.rs deleted file mode 100644 index d88644ca46..0000000000 --- a/dozer-ingestion/kafka/src/debezium/schema_registry.rs +++ /dev/null @@ -1,151 +0,0 @@ -use std::collections::HashMap; - -use dozer_ingestion_connector::dozer_types::log::error; -use dozer_ingestion_connector::dozer_types::serde_json::{self, Value}; -use dozer_ingestion_connector::dozer_types::types::{ - FieldDefinition, FieldType, Schema, SourceDefinition, -}; -use dozer_ingestion_connector::{tokio, CdcType, SourceSchema}; -use schema_registry_converter::async_impl::schema_registry::SrSettings; -use schema_registry_converter::schema_registry_common::SubjectNameStrategy; - -use crate::{KafkaError, KafkaSchemaError}; - -use super::schema::map_type; -use super::stream_consumer::DebeziumSchemaStruct; - -pub struct SchemaRegistry {} - -impl SchemaRegistry { - pub fn map_typ(schema: &DebeziumSchemaStruct) -> Result<(FieldType, bool), KafkaSchemaError> { - let nullable = schema.optional.map_or(false, |o| !o); - match schema.r#type.clone() { - Value::String(_) => map_type(&DebeziumSchemaStruct { - r#type: schema.r#type.clone(), - fields: None, - optional: None, - name: None, - field: None, - version: None, - parameters: None, - }) - .map(|s| (s, nullable)), - Value::Array(types) => { - let nullable = types.contains(&Value::from("null")); - for typ in types { - if typ.as_str().unwrap() != "null" { - return Self::map_typ(&DebeziumSchemaStruct { - r#type: typ, - fields: None, - optional: Some(nullable), - name: None, - field: None, - version: None, - parameters: None, - }); - } - } - - Err(KafkaSchemaError::TypeNotSupported("Array".to_string())) - } - Value::Object(obj) => SchemaRegistry::map_typ(&DebeziumSchemaStruct { - r#type: obj.get("type").unwrap().clone(), - fields: None, - optional: None, - name: None, - field: None, - version: None, - parameters: None, - }), - _ => Err(KafkaSchemaError::TypeNotSupported( - "Unexpected value".to_string(), - )), - } - } - - pub async fn fetch_struct( - sr_settings: &SrSettings, - table_name: &str, - is_key: bool, - ) -> Result { - let schema_result = loop { - match schema_registry_converter::async_impl::schema_registry::get_schema_by_subject( - sr_settings, - &SubjectNameStrategy::TopicNameStrategy(table_name.to_string(), is_key), - ) - .await - { - Ok(schema_result) => break schema_result, - Err(err) if err.retriable => { - const RETRY_INTERVAL: std::time::Duration = std::time::Duration::from_secs(5); - error!("schema registry fetch error {err}. retrying in {RETRY_INTERVAL:?}..."); - tokio::time::sleep(RETRY_INTERVAL).await; - continue; - } - Err(err) => return Err(KafkaError::SchemaRegistryFetchError(err)), - } - }; - - serde_json::from_str::(&schema_result.schema) - .map_err(KafkaError::JsonDecodeError) - } - - pub async fn get_schema( - table_names: Option<&[String]>, - schema_registry_url: String, - ) -> Result, KafkaError> { - let sr_settings = SrSettings::new(schema_registry_url); - match table_names { - None => Ok(vec![]), - Some(tables) => match tables.first() { - None => Ok(vec![]), - Some(table) => { - let key_result = - SchemaRegistry::fetch_struct(&sr_settings, table, true).await?; - let schema_result = - SchemaRegistry::fetch_struct(&sr_settings, table, false).await?; - - let pk_fields = key_result.fields.map_or(vec![], |fields| { - fields - .iter() - .map(|f| f.name.clone().map_or("".to_string(), |name| name)) - .collect() - }); - - let fields = schema_result.fields.map_or(vec![], |f| f); - let mut pk_keys_indexes = vec![]; - let mut fields_schema_map: HashMap = - HashMap::new(); - - let defined_fields: Result, KafkaError> = fields - .iter() - .enumerate() - .map(|(idx, f)| { - let (typ, nullable) = - Self::map_typ(f).map_err(KafkaError::KafkaSchemaError)?; - let name = f.name.clone().unwrap(); - if pk_fields.contains(&name) { - pk_keys_indexes.push(idx); - } - fields_schema_map.insert(name.clone(), f); - Ok(FieldDefinition { - name, - typ, - nullable, - source: SourceDefinition::Dynamic, - description: None, - }) - }) - .collect(); - - let schema = Schema { - fields: defined_fields?, - primary_index: pk_keys_indexes, - }; - - Ok(vec![SourceSchema::new(schema, CdcType::FullChanges)]) - } - }, - } - } -} diff --git a/dozer-ingestion/kafka/src/debezium/stream_consumer.rs b/dozer-ingestion/kafka/src/debezium/stream_consumer.rs deleted file mode 100644 index d59f8d790a..0000000000 --- a/dozer-ingestion/kafka/src/debezium/stream_consumer.rs +++ /dev/null @@ -1,213 +0,0 @@ -use crate::debezium::mapper::convert_value_to_schema; -use crate::debezium::schema::map_schema; -use crate::stream_consumer::StreamConsumer; -use crate::stream_consumer_helper::{is_network_failure, OffsetsMap, StreamConsumerHelper}; -use crate::{KafkaError, KafkaStreamError}; - -use dozer_ingestion_connector::dozer_types::node::OpIdentifier; -use dozer_ingestion_connector::TableInfo; -use dozer_ingestion_connector::{ - async_trait, - dozer_types::{ - models::ingestion_types::IngestionMessage, - serde::{Deserialize, Serialize}, - serde_json, - serde_json::Value, - types::{Operation, Record}, - }, - Ingestor, -}; -use rdkafka::{ClientConfig, Message}; - -#[derive(Debug, Serialize, Deserialize, Clone)] -#[serde(crate = "dozer_ingestion_connector::dozer_types::serde")] -#[serde(untagged)] -pub enum DebeziumFieldType { - I8(i8), - I16(i16), - I32(i32), - I64(i64), - U8(u8), - U16(u16), - U32(u32), - U64(u64), - Bool(bool), - String(String), -} - -#[derive(Debug, Serialize, Deserialize, Clone)] -#[serde(crate = "dozer_ingestion_connector::dozer_types::serde")] -pub struct DebeziumField { - pub r#type: String, - pub optional: bool, - pub default: Option, - pub field: String, -} - -#[derive(Debug, Serialize, Deserialize, PartialEq, Eq, Clone)] -#[serde(crate = "dozer_ingestion_connector::dozer_types::serde")] -pub struct DebeziumSchemaParameters { - pub scale: Option, - #[serde(rename(deserialize = "connect.decimal.precision"))] - pub precision: Option, -} - -#[derive(Debug, Serialize, Deserialize, PartialEq, Clone)] -#[serde(crate = "dozer_ingestion_connector::dozer_types::serde")] -pub struct DebeziumSchemaStruct { - pub r#type: Value, - pub fields: Option>, - pub optional: Option, - pub name: Option, - pub field: Option, - pub version: Option, - pub parameters: Option, -} - -#[derive(Debug, Serialize, Deserialize, Clone)] -#[serde(crate = "dozer_ingestion_connector::dozer_types::serde")] -pub struct DebeziumPayload { - pub before: Option, - pub after: Option, - pub op: Option, -} - -#[derive(Debug, Serialize, Deserialize)] -#[serde(crate = "dozer_ingestion_connector::dozer_types::serde")] -pub struct DebeziumMessage { - pub schema: DebeziumSchemaStruct, - pub payload: DebeziumPayload, -} - -#[derive(Default)] -pub struct DebeziumStreamConsumer {} - -impl DebeziumStreamConsumer {} - -#[async_trait] -impl StreamConsumer for DebeziumStreamConsumer { - async fn run( - &self, - client_config: ClientConfig, - ingestor: &Ingestor, - tables: Vec, - last_checkpoint: Option, - _schema_registry_url: &Option, - ) -> Result<(), KafkaError> { - assert!(last_checkpoint.is_none()); - let topics: Vec<&str> = tables.iter().map(|t| t.name.as_str()).collect(); - let mut con = StreamConsumerHelper::start(&client_config, &topics).await?; - let mut offsets = OffsetsMap::new(); - loop { - let m = match con.poll(None).unwrap() { - Ok(m) => m, - Err(err) if is_network_failure(&err) => { - con = StreamConsumerHelper::resume(&client_config, &topics, &offsets).await?; - continue; - } - Err(err) => Err(KafkaError::KafkaStreamError( - KafkaStreamError::PollingError(err), - ))?, - }; - StreamConsumerHelper::update_offsets(&mut offsets, &m); - - if let (Some(message), Some(key)) = (m.payload(), m.key()) { - let mut value_struct: DebeziumMessage = serde_json::from_str( - std::str::from_utf8(message).map_err(KafkaError::BytesConvertError)?, - ) - .map_err(KafkaError::JsonDecodeError)?; - let key_struct: DebeziumMessage = serde_json::from_str( - std::str::from_utf8(key).map_err(KafkaError::BytesConvertError)?, - ) - .map_err(KafkaError::JsonDecodeError)?; - - let (schema, fields_map) = map_schema(&value_struct.schema, &key_struct.schema) - .map_err(KafkaError::KafkaSchemaError)?; - - // When update happens before is null. - // If PK value changes, then debezium creates two events - delete and insert - if value_struct.payload.before.is_none() - && value_struct.payload.op == Some("u".to_string()) - { - value_struct.payload.before = value_struct.payload.after.clone(); - } - - match (value_struct.payload.after, value_struct.payload.before) { - (Some(new_payload), Some(old_payload)) => { - let new = convert_value_to_schema(new_payload, &schema, &fields_map) - .map_err(KafkaError::KafkaSchemaError)?; - let old = convert_value_to_schema(old_payload, &schema, &fields_map) - .map_err(KafkaError::KafkaSchemaError)?; - - if ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index: 0, - op: Operation::Update { - old: Record { - values: old, - lifetime: None, - }, - new: Record { - values: new, - lifetime: None, - }, - }, - id: None, - }) - .await - .is_err() - { - // If receiving side is closed, we should stop the stream - return Ok(()); - } - } - (None, Some(old_payload)) => { - let old = convert_value_to_schema(old_payload, &schema, &fields_map) - .map_err(KafkaError::KafkaSchemaError)?; - - if ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index: 0, - op: Operation::Delete { - old: Record { - values: old, - lifetime: None, - }, - }, - id: None, - }) - .await - .is_err() - { - // If receiving side is closed, we should stop the stream - return Ok(()); - } - } - (Some(new_payload), None) => { - let new = convert_value_to_schema(new_payload, &schema, &fields_map) - .map_err(KafkaError::KafkaSchemaError)?; - - if ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index: 0, - op: Operation::Insert { - new: Record { - values: new, - lifetime: None, - }, - }, - id: None, - }) - .await - .is_err() - { - // If receiving side is closed, we should stop the stream - return Ok(()); - } - } - (None, None) => {} - } - } - } - } -} diff --git a/dozer-ingestion/kafka/src/lib.rs b/dozer-ingestion/kafka/src/lib.rs deleted file mode 100644 index d7e9212ab8..0000000000 --- a/dozer-ingestion/kafka/src/lib.rs +++ /dev/null @@ -1,92 +0,0 @@ -use std::str::Utf8Error; - -use base64::DecodeError; -use dozer_ingestion_connector::dozer_types::{ - rust_decimal, serde_json, - thiserror::{self, Error}, -}; -use schema_registry_converter::error::SRCError; - -pub mod connector; -pub mod debezium; -pub mod no_schema_registry_basic; -pub mod schema_registry_basic; -pub mod stream_consumer; -pub mod stream_consumer_basic; -mod stream_consumer_helper; -#[cfg(any(test, feature = "debezium_bench"))] -pub mod test_utils; - -#[derive(Error, Debug)] -pub enum KafkaError { - #[error(transparent)] - KafkaSchemaError(#[from] KafkaSchemaError), - - #[error("Connection error. Error: {0}")] - KafkaConnectionError(#[from] rdkafka::error::KafkaError), - - #[error("JSON decode error. Error: {0}")] - JsonDecodeError(#[source] serde_json::Error), - - #[error("Bytes convert error")] - BytesConvertError(#[source] Utf8Error), - - #[error(transparent)] - KafkaStreamError(#[from] KafkaStreamError), - - #[error("Schema registry fetch failed. Error: {0}")] - SchemaRegistryFetchError(#[source] SRCError), - - #[error("Topic not defined")] - TopicNotDefined, -} - -#[derive(Error, Debug)] -pub enum KafkaStreamError { - #[error("Consume commit error")] - ConsumeCommitError(#[source] rdkafka::error::KafkaError), - - #[error("Message consume error")] - MessageConsumeError(#[source] rdkafka::error::KafkaError), - - #[error("Polling error")] - PollingError(#[source] rdkafka::error::KafkaError), -} - -#[derive(Error, Debug, PartialEq)] -pub enum KafkaSchemaError { - #[error("Schema definition not found")] - SchemaDefinitionNotFound, - - #[error("Unsupported \"{0}\" type")] - TypeNotSupported(String), - - #[error("Field \"{0}\" not found")] - FieldNotFound(String), - - #[error("Binary decode error")] - BinaryDecodeError(#[source] DecodeError), - - #[error("Scale not found")] - ScaleNotFound, - - #[error("Scale is invalid")] - ScaleIsInvalid, - - #[error("Decimal convert error")] - DecimalConvertError(#[source] rust_decimal::Error), - - #[error("Invalid date")] - InvalidDateError, - - #[error("Invalid json: {0}")] - InvalidJsonError(String), - - // #[error("Invalid time")] - // InvalidTimeError, - #[error("Invalid timestamp")] - InvalidTimestampError, -} - -#[cfg(test)] -mod tests; diff --git a/dozer-ingestion/kafka/src/no_schema_registry_basic.rs b/dozer-ingestion/kafka/src/no_schema_registry_basic.rs deleted file mode 100644 index 20f93a7b35..0000000000 --- a/dozer-ingestion/kafka/src/no_schema_registry_basic.rs +++ /dev/null @@ -1,46 +0,0 @@ -use dozer_ingestion_connector::{ - dozer_types::types::{FieldDefinition, FieldType, Schema, SourceDefinition}, - CdcType, SourceSchema, -}; - -use crate::KafkaError; - -pub struct NoSchemaRegistryBasic {} - -impl NoSchemaRegistryBasic { - pub fn get_single_schema() -> SourceSchema { - let schema = Schema { - fields: vec![ - FieldDefinition { - name: "key".to_string(), - typ: FieldType::String, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "message".to_string(), - typ: FieldType::String, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }, - ], - primary_index: vec![0], - }; - - SourceSchema::new(schema, CdcType::FullChanges) - } - - pub fn get_schema(table_names: Option<&[String]>) -> Result, KafkaError> { - let mut schemas = vec![]; - if let Some(tables) = table_names { - for _ in 0..tables.len() { - let schema = Self::get_single_schema(); - schemas.push(schema); - } - } - - Ok(schemas) - } -} diff --git a/dozer-ingestion/kafka/src/schema_registry_basic.rs b/dozer-ingestion/kafka/src/schema_registry_basic.rs deleted file mode 100644 index 634ca1f030..0000000000 --- a/dozer-ingestion/kafka/src/schema_registry_basic.rs +++ /dev/null @@ -1,84 +0,0 @@ -#![allow(clippy::type_complexity)] - -use dozer_ingestion_connector::{ - dozer_types::types::{FieldDefinition, Schema, SourceDefinition}, - CdcType, SourceSchema, -}; -use schema_registry_converter::async_impl::schema_registry::SrSettings; -use std::collections::HashMap; - -use crate::{ - debezium::{schema_registry::SchemaRegistry, stream_consumer::DebeziumSchemaStruct}, - KafkaError, -}; - -pub struct SchemaRegistryBasic {} - -impl SchemaRegistryBasic { - pub async fn get_single_schema( - table_name: &str, - schema_registry_url: &str, - ) -> Result<(SourceSchema, HashMap), KafkaError> { - let sr_settings = SrSettings::new(schema_registry_url.to_string()); - let key_result = SchemaRegistry::fetch_struct(&sr_settings, table_name, true).await?; - let schema_result = SchemaRegistry::fetch_struct(&sr_settings, table_name, false).await?; - - let pk_fields = key_result.fields.map_or(vec![], |fields| { - fields - .iter() - .map(|f| f.name.clone().map_or("".to_string(), |name| name)) - .collect() - }); - - let fields = schema_result.fields.map_or(vec![], |f| f); - let mut pk_keys_indexes = vec![]; - let mut fields_schema_map: HashMap = HashMap::new(); - - let defined_fields: Result, KafkaError> = fields - .iter() - .to_owned() - .enumerate() - .map(|(idx, f)| { - let (typ, nullable) = - SchemaRegistry::map_typ(f).map_err(KafkaError::KafkaSchemaError)?; - let name = f.name.clone().unwrap(); - if pk_fields.contains(&name) { - pk_keys_indexes.push(idx); - } - fields_schema_map.insert(name.clone(), f.clone()); - Ok(FieldDefinition { - name, - typ, - nullable, - source: SourceDefinition::Dynamic, - description: None, - }) - }) - .collect(); - - let schema = Schema { - fields: defined_fields?, - primary_index: pk_keys_indexes, - }; - - Ok(( - SourceSchema::new(schema, CdcType::FullChanges), - fields_schema_map, - )) - } - - pub async fn get_schema( - table_names: Option<&[String]>, - schema_registry_url: String, - ) -> Result, KafkaError> { - let mut schemas = vec![]; - if let Some(tables) = table_names { - for table_name in tables.iter() { - let (schema, _) = Self::get_single_schema(table_name, &schema_registry_url).await?; - schemas.push(schema); - } - } - - Ok(schemas) - } -} diff --git a/dozer-ingestion/kafka/src/stream_consumer.rs b/dozer-ingestion/kafka/src/stream_consumer.rs deleted file mode 100644 index 00765fd2b2..0000000000 --- a/dozer-ingestion/kafka/src/stream_consumer.rs +++ /dev/null @@ -1,18 +0,0 @@ -use crate::KafkaError; - -use dozer_ingestion_connector::{ - async_trait, dozer_types::node::OpIdentifier, Ingestor, TableInfo, -}; -use rdkafka::ClientConfig; - -#[async_trait] -pub trait StreamConsumer { - async fn run( - &self, - client_config: ClientConfig, - ingestor: &Ingestor, - tables: Vec, - last_checkpoint: Option, - schema_registry_url: &Option, - ) -> Result<(), KafkaError>; -} diff --git a/dozer-ingestion/kafka/src/stream_consumer_basic.rs b/dozer-ingestion/kafka/src/stream_consumer_basic.rs deleted file mode 100644 index fceade9724..0000000000 --- a/dozer-ingestion/kafka/src/stream_consumer_basic.rs +++ /dev/null @@ -1,176 +0,0 @@ -use std::collections::HashMap; - -use dozer_ingestion_connector::{ - async_trait, - dozer_types::{ - models::ingestion_types::IngestionMessage, - node::OpIdentifier, - serde::{Deserialize, Serialize}, - serde_json::{self, Value}, - types::{Field, Operation, Record}, - }, - Ingestor, TableInfo, -}; -use rdkafka::{ClientConfig, Message}; - -use crate::schema_registry_basic::SchemaRegistryBasic; -use crate::stream_consumer::StreamConsumer; -use crate::{debezium::mapper::convert_value_to_schema, KafkaError}; -use crate::{no_schema_registry_basic::NoSchemaRegistryBasic, KafkaStreamError}; - -use super::stream_consumer_helper::{is_network_failure, OffsetsMap, StreamConsumerHelper}; - -#[derive(Debug, Serialize, Deserialize, Clone)] -#[serde(crate = "dozer_ingestion_connector::dozer_types::serde")] -#[serde(untagged)] -pub enum FieldType { - I8(i8), - I16(i16), - I32(i32), - I64(i64), - U8(u8), - U16(u16), - U32(u32), - U64(u64), - Bool(bool), - String(String), -} - -#[derive(Debug, Serialize, Deserialize, Clone)] -#[serde(crate = "dozer_ingestion_connector::dozer_types::serde")] -pub struct KafkaField { - pub r#type: String, - pub optional: bool, - pub default: Option, - pub field: String, -} - -#[derive(Debug, Serialize, Deserialize, PartialEq, Eq, Clone)] -#[serde(crate = "dozer_ingestion_connector::dozer_types::serde")] -pub struct SchemaParameters { - pub scale: Option, - #[serde(rename(deserialize = "connect.decimal.precision"))] - pub precision: Option, -} - -#[derive(Debug, Serialize, Deserialize, PartialEq, Clone)] -#[serde(crate = "dozer_ingestion_connector::dozer_types::serde")] -pub struct SchemaStruct { - pub r#type: Value, - pub fields: Option>, - pub optional: Option, - pub name: Option, - pub field: Option, - pub version: Option, - pub parameters: Option, -} - -#[derive(Debug, Serialize, Deserialize, Clone)] -#[serde(crate = "dozer_ingestion_connector::dozer_types::serde")] -pub struct Payload { - pub before: Option, - pub after: Option, - pub op: Option, -} - -#[derive(Default)] -pub struct StreamConsumerBasic {} - -#[async_trait] -impl StreamConsumer for StreamConsumerBasic { - async fn run( - &self, - client_config: ClientConfig, - ingestor: &Ingestor, - tables: Vec, - last_checkpoint: Option, - schema_registry_url: &Option, - ) -> Result<(), KafkaError> { - assert!(last_checkpoint.is_none()); - let topics: Vec = tables.iter().map(|t| t.name.clone()).collect(); - - let mut schemas = HashMap::new(); - for (table_index, table) in tables.into_iter().enumerate() { - let schema = if let Some(url) = schema_registry_url { - SchemaRegistryBasic::get_single_schema(&table.name, url).await? - } else { - (NoSchemaRegistryBasic::get_single_schema(), HashMap::new()) - }; - - schemas.insert(table.name.clone(), (table_index, schema)); - } - - let topics: Vec<&str> = topics.iter().map(|t| t.as_str()).collect(); - let mut con = StreamConsumerHelper::start(&client_config, &topics).await?; - - let mut offsets = OffsetsMap::new(); - loop { - if let Some(result) = con.poll(None) { - if matches!(result.as_ref(), Err(err) if is_network_failure(err)) { - con = StreamConsumerHelper::resume(&client_config, &topics, &offsets).await?; - continue; - } - let m = result - .map_err(|e| KafkaError::KafkaStreamError(KafkaStreamError::PollingError(e)))?; - StreamConsumerHelper::update_offsets(&mut offsets, &m); - match schemas.get(m.topic()) { - None => return Err(KafkaError::TopicNotDefined), - Some((table_index, (schema, fields_map))) => { - if let (Some(message), Some(key)) = (m.payload(), m.key()) { - let new = match schema_registry_url { - None => { - let value = std::str::from_utf8(message) - .map_err(KafkaError::BytesConvertError)?; - let key = std::str::from_utf8(key) - .map_err(KafkaError::BytesConvertError)?; - - vec![ - Field::String(key.to_string()), - Field::String(value.to_string()), - ] - } - Some(_) => { - let value_struct: Value = serde_json::from_str( - std::str::from_utf8(message) - .map_err(KafkaError::BytesConvertError)?, - ) - .map_err(KafkaError::JsonDecodeError)?; - let _key_struct: Value = serde_json::from_str( - std::str::from_utf8(key) - .map_err(KafkaError::BytesConvertError)?, - ) - .map_err(KafkaError::JsonDecodeError)?; - - convert_value_to_schema( - value_struct, - &schema.schema, - fields_map, - ) - .map_err(KafkaError::KafkaSchemaError)? - } - }; - - if ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index: *table_index, - op: Operation::Insert { - new: Record { - values: new, - lifetime: None, - }, - }, - id: None, - }) - .await - .is_err() - { - // If receiving side is closed, we should stop the stream - return Ok(()); - } - } - } - } - } - } - } -} diff --git a/dozer-ingestion/kafka/src/stream_consumer_helper.rs b/dozer-ingestion/kafka/src/stream_consumer_helper.rs deleted file mode 100644 index c7eb31c345..0000000000 --- a/dozer-ingestion/kafka/src/stream_consumer_helper.rs +++ /dev/null @@ -1,115 +0,0 @@ -use dozer_ingestion_connector::{dozer_types, tokio}; -use rdkafka::{ - consumer::{BaseConsumer, Consumer}, - message::BorrowedMessage, - util::Timeout, - ClientConfig, Message, Offset, -}; -use std::collections::HashMap; - -use crate::KafkaError; - -pub struct StreamConsumerHelper; - -pub type OffsetsMap = HashMap; // key: topic, value: (partition, offset) - -impl StreamConsumerHelper { - pub async fn start( - client_config: &ClientConfig, - topics: &[&str], - ) -> Result { - Self::resume_impl(client_config, topics, None).await - } - - pub async fn resume( - client_config: &ClientConfig, - topics: &[&str], - offsets: &OffsetsMap, - ) -> Result { - Self::resume_impl(client_config, topics, Some(offsets)).await - } - - pub fn update_offsets(offsets: &mut OffsetsMap, message: &BorrowedMessage<'_>) { - let _ = offsets.insert( - message.topic().into(), - (message.partition(), message.offset()), - ); - } - - async fn resume_impl( - client_config: &ClientConfig, - topics: &[&str], - offsets: Option<&OffsetsMap>, - ) -> Result { - loop { - match Self::try_resume(client_config, topics, offsets).await { - Ok(con) => return Ok(con), - Err(err) if is_network_failure(&err) => { - const RETRY_INTERVAL: std::time::Duration = std::time::Duration::from_secs(5); - dozer_types::log::error!( - "stream resume error {err}. retrying in {RETRY_INTERVAL:?}..." - ); - tokio::time::sleep(RETRY_INTERVAL).await; - continue; - } - Err(err) => Err(KafkaError::KafkaConnectionError(err))?, - } - } - } - - async fn try_resume( - client_config: &ClientConfig, - topics: &[&str], - offsets: Option<&OffsetsMap>, - ) -> Result { - let con: BaseConsumer = client_config.create()?; - con.subscribe(topics.iter().as_slice())?; - - if let Some(offsets) = offsets { - for (topic, &(partition, offset)) in offsets.iter() { - con.seek(topic, partition, Offset::Offset(offset), Timeout::Never)?; - } - } - - Ok(con) - } -} - -pub fn is_network_failure(err: &rdkafka::error::KafkaError) -> bool { - use rdkafka::error::KafkaError::*; - let error_code = match err { - ConsumerCommit(error_code) => error_code, - Flush(error_code) => error_code, - Global(error_code) => error_code, - GroupListFetch(error_code) => error_code, - MessageConsumption(error_code) => error_code, - MessageProduction(error_code) => error_code, - MetadataFetch(error_code) => error_code, - OffsetFetch(error_code) => error_code, - Rebalance(error_code) => error_code, - SetPartitionOffset(error_code) => error_code, - StoreOffset(error_code) => error_code, - MockCluster(error_code) => error_code, - Transaction(rdkafka_err) => return rdkafka_err.is_retriable(), - other => { - dozer_types::log::warn!( - "unregonized kafka error error: {other}. treating as non-network error." - ); - return false; - } - }; - use rdkafka::types::RDKafkaErrorCode::*; - matches!( - error_code, - Fail | BrokerTransportFailure - | Resolve - | MessageTimedOut - | AllBrokersDown - | OperationTimedOut - | TimedOutQueue - | Retry - | PollExceeded - | RequestTimedOut - | NetworkException - ) -} diff --git a/dozer-ingestion/kafka/src/test_utils.rs b/dozer-ingestion/kafka/src/test_utils.rs deleted file mode 100644 index 954caac7bd..0000000000 --- a/dozer-ingestion/kafka/src/test_utils.rs +++ /dev/null @@ -1,118 +0,0 @@ -// use crate::connectors::postgres::connection::helper::{connect, map_connection_config}; -// use crate::connectors::{get_connector, TableInfo}; -// use crate::ingestion::{IngestionConfig, IngestionIterator, Ingestor}; -// use crate::test_util::load_config; -// use dozer_types::models::ingestion_types::KafkaConfig; -// use dozer_types::models::app_config::Config; -// use dozer_types::models::connection::ConnectionConfig; -// -// use dozer_types::serde::{Deserialize, Serialize}; -// use dozer_types::serde_yaml; -// use postgres::Client; -// use reqwest::header::{ACCEPT, CONTENT_TYPE}; -// use std::thread; -// use std::time::Duration; -// -// #[derive(Debug, Deserialize, Serialize)] -// #[serde(crate = "dozer_types::serde")] -// pub struct DebeziumTestConfig { -// pub config: Config, -// pub debezium: DebeziumConnectorConfig, -// } -// -// #[derive(Debug, Deserialize, Serialize)] -// #[serde(crate = "dozer_types::serde")] -// pub struct DebeziumConnectorConfig { -// pub postgres_source_authentication: ConnectionConfig, -// pub connector_url: String, -// } -// -// pub fn get_debezium_config(file_name: &str) -> DebeziumTestConfig { -// let config = serde_yaml::from_str::(load_config(file_name)).unwrap(); -// -// let debezium = -// serde_yaml::from_str::(load_config("test.debezium.pg.yaml")) -// .unwrap(); -// -// DebeziumTestConfig { config, debezium } -// } -// -// pub fn get_client_and_create_table(table_name: &str, auth: &ConnectionConfig) -> Client { -// let postgres_config = map_connection_config(auth).unwrap(); -// let mut client = connect(postgres_config).unwrap(); -// -// client -// .query(&format!("DROP TABLE IF EXISTS {}", &table_name), &[]) -// .unwrap(); -// -// client -// .query( -// &format!( -// "CREATE TABLE {} -// ( -// id SERIAL -// PRIMARY KEY, -// name VARCHAR(255) NOT NULL, -// description VARCHAR(512), -// weight DOUBLE PRECISION -// );", -// &table_name -// ), -// &[], -// ) -// .unwrap(); -// -// client -// } -// -// pub fn get_iterator_and_client(table_name: String) -> (IngestionIterator, Client) { -// let config = get_debezium_config("test.debezium.yaml"); -// -// let client = -// get_client_and_create_table(&table_name, &config.debezium.postgres_source_authentication); -// -// let content = load_config("test.register-postgres.json"); -// -// let connector_client = reqwest::blocking::Client::new(); -// connector_client -// .post(&config.debezium.connector_url) -// .body(content) -// .header(CONTENT_TYPE, "application/json") -// .header(ACCEPT, "application/json") -// .send() -// .unwrap() -// .text() -// .unwrap(); -// -// let (ingestor, iterator) = Ingestor::initialize_channel(IngestionConfig::default()); -// -// thread::spawn(move || { -// let tables: Vec = vec![TableInfo { -// name: table_name.clone(), -// table_name: table_name.clone(), -// id: 0, -// columns: None, -// }]; -// -// let mut connection = config.config.connections.get(0).unwrap().clone(); -// if let Some(ConnectionConfig::Kafka(KafkaConfig { -// broker, -// schema_registry_url, -// })) = connection.config -// { -// connection.config = Some(ConnectionConfig::Kafka(KafkaConfig { -// broker, -// schema_registry_url, -// })); -// }; -// -// if table_name != "products_test" { -// thread::sleep(Duration::from_secs(1)); -// } -// -// let connector = get_connector(connection).unwrap(); -// let _ = connector.start(None, &ingestor, tables); -// }); -// -// (iterator, client) -// } diff --git a/dozer-ingestion/kafka/src/tests.rs b/dozer-ingestion/kafka/src/tests.rs deleted file mode 100644 index 9790f0933d..0000000000 --- a/dozer-ingestion/kafka/src/tests.rs +++ /dev/null @@ -1,219 +0,0 @@ -// use crate::connectors::kafka::connector::KafkaConnector; -// use crate::connectors::kafka::test_utils::{ -// get_client_and_create_table, get_debezium_config, get_iterator_and_client, -// }; -// use crate::connectors::{Connector, TableInfo}; -// use dozer_types::models::connection::ConnectionConfig; -// use dozer_types::{ingestion_types::KafkaConfig, rust_decimal::Decimal, types::Operation}; -// use postgres::Client; -// use std::fmt::Write; -// use std::thread::sleep; -// -// use rand::Rng; -// use std::time::Duration; -// -// pub struct KafkaPostgres { -// client: Client, -// table_name: String, -// } -// -// impl KafkaPostgres { -// pub fn insert_rows(&mut self, count: u64) { -// let mut buf = String::new(); -// for i in 0..count { -// if i > 0 { -// buf.write_str(",").unwrap(); -// } -// buf.write_fmt(format_args!( -// "(\'Product {}\',\'Product {} description\',{})", -// i, -// i, -// Decimal::new((i * 41) as i64, 2) -// )) -// .unwrap(); -// } -// -// let query = format!( -// "insert into {}(name, description, weight) values {}", -// self.table_name, buf, -// ); -// -// self.client.query(&query, &[]).unwrap(); -// } -// -// pub fn update_rows(&mut self) { -// self.client -// .query(&format!("UPDATE {} SET weight = 5", self.table_name), &[]) -// .unwrap(); -// } -// -// pub fn delete_rows(&mut self) { -// self.client -// .query( -// &format!("DELETE FROM {} WHERE weight = 5", self.table_name), -// &[], -// ) -// .unwrap(); -// } -// -// pub fn drop_table(&mut self) { -// self.client -// .query(&format!("DROP TABLE {}", self.table_name), &[]) -// .unwrap(); -// } -// } -// -// #[ignore] -// #[test] -// // fn connector_e2e_connect_debezium_and_use_kafka_stream() { -// fn connector_disabled_test_e2e_connect_debezium_and_use_kafka_stream() { -// let mut rng = rand::thread_rng(); -// let table_name = format!("products_test_{}", rng.gen::()); -// let (mut iterator, client) = get_iterator_and_client(table_name.clone()); -// -// let mut pg_client = KafkaPostgres { client, table_name }; -// -// pg_client.insert_rows(10); -// pg_client.update_rows(); -// pg_client.delete_rows(); -// -// let mut i = 0; -// while i < 30 { -// let op = iterator.next(); -// -// if let Some((_, op)) = op { -// i += 1; -// match op { -// Operation::Insert { .. } => { -// if i > 10 { -// panic!("Unexpected operation"); -// } -// } -// Operation::Delete { .. } => { -// if i < 21 { -// panic!("Unexpected operation"); -// } -// } -// Operation::Update { .. } => { -// if !(11..=20).contains(&i) { -// panic!("Unexpected operation"); -// } -// } -// Operation::SnapshottingDone {} => (), -// } -// } -// } -// -// pg_client.drop_table(); -// -// assert_eq!(i, 30); -// } -// -// #[ignore] -// #[test] -// // fn connector_e2e_connect_debezium_json_and_get_schema() { -// fn connector_disabled_test_e2e_connect_debezium_json_and_get_schema() { -// let mut rng = rand::thread_rng(); -// let table_name = format!("products_test_{}", rng.gen::()); -// let topic = format!("dbserver1.public.{table_name}"); -// let config = get_debezium_config("test.debezium.yaml"); -// -// let client = -// get_client_and_create_table(&table_name, &config.debezium.postgres_source_authentication); -// -// let mut pg_client = KafkaPostgres { client, table_name }; -// pg_client.insert_rows(1); -// -// let broker = if let ConnectionConfig::Kafka(KafkaConfig { broker, .. }) = config -// .config -// .connections -// .get(0) -// .unwrap() -// .clone() -// .config -// .unwrap() -// { -// broker -// } else { -// todo!() -// }; -// let connector = KafkaConnector::new( -// 1, -// KafkaConfig { -// broker, -// schema_registry_url: None, -// }, -// ); -// -// sleep(Duration::from_secs(2)); -// let schemas = connector -// .get_schemas(Some(vec![TableInfo { -// name: topic.clone(), -// table_name: topic.clone(), -// id: 0, -// columns: None, -// }])) -// .unwrap(); -// -// pg_client.drop_table(); -// -// assert_eq!(topic, schemas.get(0).unwrap().name); -// assert_eq!(4, schemas.get(0).unwrap().schema.fields.len()); -// } -// -// #[ignore] -// #[test] -// // fn connector_e2e_connect_debezium_avro_and_get_schema() { -// fn connector_disabled_test_e2e_connect_debezium_avro_and_get_schema() { -// let mut rng = rand::thread_rng(); -// let table_name = format!("products_test_{}", rng.gen::()); -// let topic = format!("dbserver1.public.{table_name}"); -// let config = get_debezium_config("test.debezium-with-schema-registry.yaml"); -// -// let client = -// get_client_and_create_table(&table_name, &config.debezium.postgres_source_authentication); -// -// let mut pg_client = KafkaPostgres { client, table_name }; -// pg_client.insert_rows(1); -// -// let (broker, schema_registry_url) = -// if let dozer_types::models::connection::ConnectionConfig::Kafka(KafkaConfig { -// broker, -// schema_registry_url, -// .. -// }) = config -// .config -// .connections -// .get(0) -// .unwrap() -// .clone() -// .config -// .unwrap() -// { -// (broker, schema_registry_url) -// } else { -// todo!() -// }; -// let connector = KafkaConnector::new( -// 1, -// KafkaConfig { -// broker, -// schema_registry_url, -// }, -// ); -// -// sleep(Duration::from_secs(1)); -// let schemas = connector -// .get_schemas(Some(vec![TableInfo { -// name: topic.clone(), -// table_name: topic.clone(), -// id: 0, -// columns: None, -// }])) -// .unwrap(); -// -// pg_client.drop_table(); -// -// assert_eq!(topic, schemas.get(0).unwrap().name); -// assert_eq!(4, schemas.get(0).unwrap().schema.fields.len()); -// } diff --git a/dozer-ingestion/mongodb/Cargo.toml b/dozer-ingestion/mongodb/Cargo.toml deleted file mode 100644 index be23664b45..0000000000 --- a/dozer-ingestion/mongodb/Cargo.toml +++ /dev/null @@ -1,12 +0,0 @@ -[package] -name = "dozer-ingestion-mongodb" -version = "0.4.0" -edition = "2021" -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-ingestion-connector = { path = "../connector" } -mongodb = "2.6.1" -bson = "2.7.0" diff --git a/dozer-ingestion/mongodb/src/lib.rs b/dozer-ingestion/mongodb/src/lib.rs deleted file mode 100644 index 190be41267..0000000000 --- a/dozer-ingestion/mongodb/src/lib.rs +++ /dev/null @@ -1,703 +0,0 @@ -use std::collections::HashMap; - -use bson::{doc, Bson, Document, Timestamp}; -use dozer_ingestion_connector::{ - async_trait, - dozer_types::{ - self, - errors::{internal::BoxedError, types::DeserializationError}, - json_types::{serde_json_to_json_value, JsonValue}, - models::ingestion_types::{IngestionMessage, TransactionInfo}, - node::OpIdentifier, - thiserror::{self, Error}, - types::{Field, FieldDefinition, FieldType, Operation, Record, SourceDefinition}, - }, - futures::{stream::FuturesUnordered, StreamExt, TryFutureExt, TryStreamExt}, - tokio::{ - self, - sync::mpsc::{channel, Sender}, - }, - CdcType, Connector, Ingestor, SourceSchema, SourceSchemaResult, TableIdentifier, TableInfo, -}; -use mongodb::{ - change_stream::event::ChangeStreamEvent, - error::{CommandError, ErrorKind}, - options::{ChangeStreamOptions, ClientOptions, ConnectionString}, -}; - -pub use bson; -pub use mongodb; - -#[derive(Error, Debug)] -pub enum MongodbConnectorError { - #[error("Failed to parse connection string. {0}")] - ParseConnectionString(#[source] mongodb::error::Error), - - #[error("Server is not part of a replica set")] - NotAReplicaSet, - - #[error("Server is sharded, which is currently not supported")] - Sharded, - - #[error("Failed to connect to mongodb with the specified configuration. {0}")] - ConnectionFailure(#[source] mongodb::error::Error), - - #[error("Failed to list databases. {0}")] - ListTablesError(#[source] mongodb::error::Error), - - #[error("Failed to read collection snapshot. {0}")] - SnapshotReadError(#[source] mongodb::error::Error), - - #[error("Failed to start a change stream for collection. {0}")] - ReplicationError(#[source] mongodb::error::Error), - - #[error("Failed to parse change stream data for collection. {0}")] - ReplicationDataError(#[source] DeserializationError), - - #[error("Change stream was invalidated because the replicated collection was renamed or dropped while replicating")] - ReplicationStreamInvalidated, - - #[error("No database specified in connection string")] - NoDatabaseError, - - #[error("Capped collections cannot be used as sources. Collection: {0}")] - CappedCollection(String), - - #[error("Collection should have pre- and post-images enabled. Collection: {0}")] - NoPrePostImages(String), - - #[error("Missing permissions: {}", .0.iter().map(|(table, permissions)| format!("{table}: [{}]", permissions.join(", "))).collect::>().join(", "))] - MissingPermissions(Vec<(String, Vec)>), -} - -use MongodbConnectorError::*; - -#[derive(Debug)] -pub struct MongodbConnector { - conn_string: String, -} - -#[derive(Default, Clone, Copy)] -struct Privs { - find: bool, - watch: bool, -} - -impl std::ops::BitOrAssign for Privs { - fn bitor_assign(&mut self, rhs: Self) { - self.find |= rhs.find; - self.watch |= rhs.watch; - } -} - -impl std::ops::BitOr for Privs { - type Output = Self; - - fn bitor(self, rhs: Self) -> Self::Output { - Privs { - find: self.find || rhs.find, - watch: self.watch || rhs.watch, - } - } -} - -async fn start_session( - client: &mongodb::Client, -) -> Result { - let session_options = mongodb::options::SessionOptions::builder() - .snapshot(true) - .build(); - - // Check if we can start a session with our read and write concerns. This - // will validate that the connected server is part of a replica set - client - .start_session(Some(session_options)) - .await - .map_err(|e| match *e.kind { - ErrorKind::Command(CommandError { code: 123, .. }) => { - MongodbConnectorError::NotAReplicaSet - } - _ => MongodbConnectorError::ConnectionFailure(e), - }) -} - -async fn snapshot_collection( - client: &mongodb::Client, - db: &mongodb::Database, - collection: &str, - table_idx: usize, - tx: Sender>, -) -> Result { - let mut session = start_session(client).await?; - let collection: mongodb::Collection = db.collection(collection); - let mut documents = collection - .find_with_session(None, None, &mut session) - .await - .map_err(ConnectionFailure)?; - let timestamp = session - .operation_time() - .expect("Operation time should be `Some` after an operation"); - documents - .stream(&mut session) - .map(|doc| { - let document = doc.map_err(SnapshotReadError)?; - let id = document_id(&document)?; - let v: JsonValue = - serde_json_to_json_value(Bson::Document(document).into_relaxed_extjson()) - .expect("Could not deserialize bson into json"); - Ok(Operation::Insert { - new: Record::new(vec![Field::Json(id), Field::Json(v)]), - }) - }) - .for_each(|op| async { - tx.send(op.map(|op| (table_idx, op))).await.unwrap(); - }) - .await; - Ok(timestamp) -} - -struct ChangeEventData { - id: Field, - fields: Vec, -} - -fn change_event_fields( - event: &ChangeStreamEvent, -) -> Result { - let id = change_event_id(event)?; - let doc = event - .full_document - .as_ref() - .expect("No full document on change stream event"); - let serde_json = Bson::from(doc).into_relaxed_extjson(); - let json = serde_json_to_json_value(serde_json).map_err(ReplicationDataError)?; - Ok(ChangeEventData { - id: Field::Json(id.clone()), - fields: vec![Field::Json(id), Field::Json(json)], - }) -} - -fn change_event_id( - event: &ChangeStreamEvent, -) -> Result { - let key = event - .document_key - .as_ref() - .expect("No document key on change stream event"); - document_id(key) -} - -fn document_id(document: &Document) -> Result { - serde_json_to_json_value( - document - .get("_id") - .expect("No _id field in document key") - .clone() - .into_relaxed_extjson(), - ) - .map_err(ReplicationDataError) -} - -async fn replicate_collection( - db: &mongodb::Database, - collection: &str, - start_at: Timestamp, - table_idx: usize, - tx: Sender>, -) -> Result<(), MongodbConnectorError> { - let collection: mongodb::Collection = db.collection(collection); - let options = ChangeStreamOptions::builder() - .start_at_operation_time(Some(start_at)) - // Request the document post-image. This is required, because fine-grained - // change propagation is not supported for JSON types in dozer - .full_document(Some(mongodb::options::FullDocumentType::Required)) - .build(); - let events = collection - .watch(None, Some(options)) - .await - .map_err(ReplicationError)?; - - events - .map_err(ReplicationError) - .and_then(|event| async move { - match event.operation_type { - mongodb::change_stream::event::OperationType::Insert => { - let data = change_event_fields(&event)?; - Ok(Operation::Insert { - new: Record::new(data.fields), - }) - } - mongodb::change_stream::event::OperationType::Update - | mongodb::change_stream::event::OperationType::Replace => { - let data = change_event_fields(&event)?; - Ok(Operation::Update { - old: Record::new(vec![data.id, Field::Null]), - new: Record::new(data.fields), - }) - } - mongodb::change_stream::event::OperationType::Delete => { - let id = change_event_id(&event)?; - Ok(Operation::Delete { - old: Record::new(vec![Field::Json(id), Field::Null]), - }) - } - mongodb::change_stream::event::OperationType::Drop - | mongodb::change_stream::event::OperationType::Rename - | mongodb::change_stream::event::OperationType::DropDatabase - | mongodb::change_stream::event::OperationType::Invalidate => { - Err(ReplicationStreamInvalidated) - } - mongodb::change_stream::event::OperationType::Other(_) => todo!(), - _ => todo!(), - } - }) - .for_each(|op| async { tx.send(op.map(|op| (table_idx, op))).await.unwrap() }) - .await; - Ok(()) -} - -#[derive(Default)] -struct ServerInfo { - replset: bool, - sharded: bool, -} - -impl MongodbConnector { - pub fn new(connection_string: String) -> Result { - let _ = ConnectionString::parse(&connection_string) - .map_err(MongodbConnectorError::ParseConnectionString); - Ok(Self { - conn_string: connection_string, - }) - } - - async fn client_options( - &self, - ) -> Result { - let mut options = ClientOptions::parse(&self.conn_string) - .await - .map_err(MongodbConnectorError::ParseConnectionString)?; - options.write_concern = None; - Ok(options) - } - - async fn client(&self) -> Result { - let options = self.client_options().await?; - self.client_with_options(options).await - } - - async fn client_with_options( - &self, - options: mongodb::options::ClientOptions, - ) -> Result { - let client = mongodb::Client::with_options(options).unwrap(); - if client.default_database().is_none() { - return Err(NoDatabaseError); - } - Ok(client) - } - - fn database(&self, client: &mongodb::Client) -> mongodb::Database { - client - .default_database() - .expect("No default database specified") - } - - async fn identify_server( - &self, - client: &mongodb::Client, - ) -> Result { - let db = self.database(client); - let hello = doc! { - "hello": 1, - }; - // This command should always succeed, if we can connect. So, on error, connecting failed. - let hello_result = db - .run_command(hello, None) - .await - .map_err(MongodbConnectorError::ConnectionFailure)?; - - let mut server_info = ServerInfo::default(); - - // This field is only present if the instance is a member of a replset - if hello_result.get("setName").is_some() { - server_info.replset = true; - } - - if let Ok("isdbgrid") = hello_result.get_str("msg") { - server_info.sharded = true; - } - - Ok(server_info) - } - - async fn validate_table_privileges( - &self, - database: &mongodb::Database, - username: &str, - tables: &[TableIdentifier], - ) -> Result<(), MongodbConnectorError> { - // Users can always view their own privileges, so failure here is a connection - // error - let user_info = database - .run_command( - Document::from_iter([ - ("usersInfo".to_owned(), username.into()), - ("showPrivileges".to_owned(), true.into()), - ]), - None, - ) - .await - .map_err(ConnectionFailure)?; - let privileges = user_info.get_array("users").unwrap()[0] - .as_document() - .unwrap() - .get_array("inheritedPrivileges") - .unwrap() - .iter() - .filter_map(|privilege| privilege.as_document()); - - let mut table_privs: HashMap<&str, Privs> = HashMap::with_capacity(tables.len()); - for table in tables { - table_privs.insert( - &table.name, - Privs { - find: false, - watch: false, - }, - ); - } - - let mut db_or_global_privs = Privs { - find: false, - watch: false, - }; - // We need the `find` and `changeStream` privileges for all collections, - // or for the entire database, or for the entire server - - for privilege in privileges { - let Ok(actions) = privilege.get_array("actions") else { - continue; - }; - - let Some((db, collection)) = - privilege - .get_document("resource") - .ok() - .and_then(|resource| { - let Ok(db) = resource.get_str("db") else { - return None; - }; - let Ok(collection) = resource.get_str("collection") else { - return None; - }; - Some((db, collection)) - }) - else { - continue; - }; - - let mut privs = Privs { - find: false, - watch: false, - }; - - for action in actions { - if action.as_str() == Some("find") { - privs.find = true; - } - - if action.as_str() == Some("changeStream") { - privs.watch = true; - } - } - - if db.is_empty() || collection.is_empty() { - db_or_global_privs |= privs; - } else if db == database.name() { - if let Some(table_priv) = table_privs.get_mut(collection) { - *table_priv |= privs; - } - } - } - - if db_or_global_privs.find && db_or_global_privs.watch { - return Ok(()); - } - - let mut missing_privs = Vec::new(); - for table in tables { - let privs = table_privs - .get(table.name.as_str()) - .copied() - .unwrap_or_default() - | db_or_global_privs; - - let mut missing_table_privs = Vec::new(); - if !privs.find { - missing_table_privs.push("find".to_owned()); - } - if !privs.watch { - missing_table_privs.push("changeStream".to_owned()); - } - - if !missing_table_privs.is_empty() { - missing_privs.push((table.name.to_owned(), missing_table_privs)); - } - } - if missing_privs.is_empty() { - Ok(()) - } else { - Err(MissingPermissions(missing_privs)) - } - } -} - -#[async_trait] -impl Connector for MongodbConnector { - async fn validate_connection(&mut self) -> Result<(), BoxedError> { - let client = self.client().await?; - let server_info = self.identify_server(&client).await?; - if !server_info.replset { - return Err(NotAReplicaSet.into()); - } - if server_info.sharded { - return Err(Sharded.into()); - } - Ok(()) - } - - fn types_mapping() -> Vec<(String, Option)> { - todo!(); - } - - async fn list_columns( - &mut self, - tables: Vec, - ) -> Result, BoxedError> { - Ok(tables - .into_iter() - .map(|table| TableInfo { - schema: None, - name: table.name, - column_names: vec!["data".to_owned()], - }) - .collect()) - } - - async fn get_schemas( - &mut self, - table_infos: &[TableInfo], - ) -> Result, BoxedError> { - let _ = self.client().await?; - Ok(table_infos - .iter() - .map(|_table_info| { - Ok(SourceSchema { - schema: dozer_types::types::Schema { - fields: vec![ - FieldDefinition { - name: "_id".to_owned(), - typ: FieldType::Json, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "data".to_owned(), - typ: FieldType::Json, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - ], - primary_index: vec![0], - }, - cdc_type: CdcType::OnlyPK, - }) - }) - .collect()) - } - - async fn list_tables(&mut self) -> Result, BoxedError> { - let client = self.client().await?; - let database = self.database(&client); - let collections = database - .list_collection_names(None) - .await - .map_err(ListTablesError)?; - - dozer_types::log::debug!("Collections: {:?}", &collections); - - Ok(database - .list_collection_names(None) - .await - .map_err(ListTablesError)? - .into_iter() - .map(|collection_name| TableIdentifier::new(None, collection_name)) - .collect()) - } - - async fn validate_tables(&mut self, tables: &[TableIdentifier]) -> Result<(), BoxedError> { - let options = self.client_options().await?; - let client = self.client_with_options(options.clone()).await?; - let database = self.database(&client); - let user = options - .credential - .as_ref() - .and_then(|cred| cred.username.as_ref()); - - // If we could connect without a user, there is no access control and - // we can do whatever we want. Else, check whether we have the correct privileges - // for replication - if let Some(user) = user { - self.validate_table_privileges(&database, user, tables) - .await?; - } - - let table_names = tables - .iter() - .map(|table| Bson::String(table.name.clone())) - .collect::>(); - let collection_list = database - .list_collections(Some(doc! {"name": {"$in": table_names}}), None) - .await; - // Try to check whether the collection is capped (capped collections can't be watched), - // and whether pre- and post-images are enabled (these are currently needed to - // get the result of updates). This needs the `listCollections` privilege, - // which is included in the standard `read` role. If we don't have this privilege, - // we succeed for now, but might fail when starting replication. - if let Ok(collections) = collection_list { - collections - .map_err(MongodbConnectorError::ConnectionFailure) - .try_for_each(|collection_info| async { - let options = collection_info.options; - if options.capped.unwrap_or(false) { - return Err(CappedCollection(collection_info.name)); - } - - if !options - .change_stream_pre_and_post_images - .map(|option| option.enabled) - .unwrap_or(false) - { - return Err(NoPrePostImages(collection_info.name)); - } - - Ok::<_, MongodbConnectorError>(()) - }) - .await?; - } - Ok(()) - } - - async fn serialize_state(&self) -> Result, BoxedError> { - Ok(vec![]) - } - - async fn start( - &mut self, - ingestor: &Ingestor, - tables: Vec, - _last_checkpoint: Option, - ) -> Result<(), BoxedError> { - // Snapshot: find - // - // Replicate: changeStream - let client = self.client().await?; - let database = self.database(&client); - - let (tx, mut rx) = channel::>(100); - - let snapshots = FuturesUnordered::new(); - for (idx, table) in tables.iter().enumerate() { - let fut = snapshot_collection(&client, &database, &table.name, idx, tx.clone()) - .map_ok(move |timestamp| (idx, timestamp)); - snapshots.push(fut); - } - drop(tx); - - let snapshot_ingestor = ingestor.clone(); - let snapshot_task = tokio::spawn(async move { - if snapshot_ingestor - .handle_message(IngestionMessage::TransactionInfo( - TransactionInfo::SnapshottingStarted, - )) - .await - .is_err() - { - // If the ingestor is already closed, we don't need to do anything - return Ok::<_, MongodbConnectorError>(()); - } - while let Some(result) = rx.recv().await { - let (table_index, op) = result?; - if snapshot_ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index, - op, - id: None, - }) - .await - .is_err() - { - // If the ingestor is already closed, we don't need to do anything - return Ok(()); - } - } - if snapshot_ingestor - .handle_message(IngestionMessage::TransactionInfo( - TransactionInfo::SnapshottingDone { id: None }, - )) - .await - .is_err() - { - // If the ingestor is already closed, we don't need to do anything - return Ok(()); - }; - Ok(()) - }); - - let timestamps: Vec<(usize, Timestamp)> = snapshots.try_collect().await?; - - snapshot_task.await.unwrap()?; - - let (tx, mut rx) = channel::>(100); - - let replicators = FuturesUnordered::new(); - for (table_idx, timestamp) in timestamps { - let tx = tx.clone(); - replicators.push(replicate_collection( - &database, - &tables[table_idx].name, - timestamp, - table_idx, - tx, - )); - } - drop(tx); - - let ingestor = ingestor.clone(); - let replication_task = tokio::spawn(async move { - while let Some(result) = rx.recv().await { - let (table_index, op) = result?; - if ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index, - op, - id: None, - }) - .await - .is_err() - { - // If the ingestor is already closed, we don't need to do anything - return Ok::<_, MongodbConnectorError>(()); - } - } - Ok(()) - }); - - let _: () = replicators.try_collect().await?; - let _ = replication_task.await.unwrap()?; - Ok(()) - } -} diff --git a/dozer-ingestion/mysql/Cargo.toml b/dozer-ingestion/mysql/Cargo.toml deleted file mode 100644 index fada98eb8d..0000000000 --- a/dozer-ingestion/mysql/Cargo.toml +++ /dev/null @@ -1,28 +0,0 @@ -[package] -name = "dozer-ingestion-mysql" -version = "0.4.0" -edition = "2021" -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-ingestion-connector = { path = "../connector" } -mysql_async = { version = "0.34", default-features = false, features = [ - "default-rustls", - "binlog", -] } -mysql_common = { version = "0.32", default-features = false, features = [ - "binlog", - "chrono", - "rust_decimal", -] } -geozero = { version = "0.11.0", default-features = false, features = [ - "with-wkb", -] } -rand = "0.8.5" -sqlparser = "0.41.0" - -[dev-dependencies] -serial_test = "1.0.0" -hex = "0.4.3" diff --git a/dozer-ingestion/mysql/src/binlog.rs b/dozer-ingestion/mysql/src/binlog.rs deleted file mode 100644 index 6ff486dcca..0000000000 --- a/dozer-ingestion/mysql/src/binlog.rs +++ /dev/null @@ -1,1272 +0,0 @@ -use crate::{ - connection::is_network_failure, conversion::get_field_type_for_sql_type, schema::SchemaHelper, - BreakingSchemaChange, MySQLConnectorError, -}; - -use super::{ - connection::Conn, - conversion::{IntoField, IntoFields, IntoJsonValue}, - schema::{ColumnDefinition, TableDefinition}, -}; -use crate::state::encode_state; -use dozer_ingestion_connector::dozer_types::models::ingestion_types::TransactionInfo; -use dozer_ingestion_connector::{ - dozer_types::{ - json_types::{JsonArray, JsonObject, JsonValue}, - log::{trace, warn}, - models::ingestion_types::IngestionMessage, - types::Field, - types::{FieldType, Operation, Record}, - }, - futures::StreamExt, - Ingestor, -}; -use mysql_async::{ - binlog::{ - self, - events::{RowsEventRows, TableMapEvent}, - jsonb::{Array, ComplexValue, Object, StorageFormat}, - row::BinlogRow, - value::BinlogValue, - EventFlags, - }, - BinlogStream, Pool, Row, -}; - -use std::{ - collections::{HashMap, HashSet}, - ops::Deref, -}; - -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord)] -pub struct BinlogPosition { - pub binlog_id: u64, - pub position: u64, -} - -pub async fn get_master_binlog_position( - conn: &mut Conn, -) -> Result<(String, BinlogPosition), MySQLConnectorError> { - let (filename, position): (Vec, u64) = { - let mut row: Row = conn - .exec_first("SHOW MASTER STATUS", ()) - .await - .map_err(MySQLConnectorError::QueryExecutionError)? - .unwrap(); - (row.take(0).unwrap(), row.take(1).unwrap()) - }; - - let binlog_id_with_prefix = String::from_utf8(filename.clone()).map_err(|err| { - MySQLConnectorError::BinlogError(format!( - "Unexpected binlog filename format: {filename:?}: {err}" - )) - })?; - - let Some((prefix, suffix)) = binlog_id_with_prefix.split_once('.') else { - return Err(MySQLConnectorError::BinlogError(format!( - "Unexpected binlog filename format: {binlog_id_with_prefix:?}" - ))); - }; - - let binlog_id = suffix.parse::().map_err(|err| { - MySQLConnectorError::BinlogError(format!( - "Unexpected binlog filename format: {filename:?}: {err}" - )) - })?; - - Ok(( - prefix.to_string(), - BinlogPosition { - binlog_id, - position, - }, - )) -} - -pub async fn get_binlog_format(conn: &mut Conn) -> Result { - let mut row: Row = conn - .exec_first("SELECT @@binlog_format", ()) - .await - .map_err(MySQLConnectorError::QueryExecutionError)? - .unwrap(); - let binlog_logging_format = row.take(0).unwrap(); - Ok(binlog_logging_format) -} - -pub struct BinlogIngestor<'a, 'd, 'e> { - ingestor: &'a Ingestor, - binlog_stream: Option, - next_position: BinlogPosition, - stop_position: Option, - local_stop_position: Option, - server_id: u32, - conn_pool: &'d Pool, - conn_url: &'e String, - binlog_prefix: String, -} - -impl<'a, 'd, 'e> BinlogIngestor<'a, 'd, 'e> { - pub fn new( - ingestor: &'a Ingestor, - start_position: BinlogPosition, - stop_position: Option, - server_id: u32, - (conn_pool, conn_url): (&'d Pool, &'e String), - binlog_prefix: String, - ) -> Self { - Self { - ingestor, - binlog_stream: None, - next_position: start_position.clone(), - stop_position, - local_stop_position: None, - server_id, - conn_pool, - conn_url, - binlog_prefix, - } - } -} - -impl BinlogIngestor<'_, '_, '_> { - async fn connect(&self) -> Result { - Conn::new(self.conn_pool.clone()) - .await - .map_err(|err| MySQLConnectorError::ConnectionFailure(self.conn_url.clone(), err)) - } - - async fn open_binlog(&mut self) -> Result<(), MySQLConnectorError> { - let filename_formatted = format!( - "{}.{:0>6}", - self.binlog_prefix, self.next_position.binlog_id - ); - let filename = filename_formatted.as_bytes(); - let binlog_stream = self - .connect() - .await? - .get_binlog_stream(self.server_id, filename, self.next_position.position) - .await - .map_err(MySQLConnectorError::BinlogOpenError)?; - - self.binlog_stream = Some(binlog_stream); - - self.local_stop_position = self.stop_position.as_ref().and_then(|stop_position| { - if self.next_position.binlog_id == stop_position.binlog_id { - Some(stop_position.position) - } else { - None - } - }); - - Ok(()) - } - - pub async fn ingest( - &mut self, - tables: &mut [TableDefinition], - schema_helper: SchemaHelper<'_>, - ) -> Result<(), MySQLConnectorError> { - if self.binlog_stream.is_none() { - self.open_binlog().await?; - } - - let mut table_cache = TableManager::new(tables); - let mut schema_change_tracker = SchemaChangeTracker::new(); - - let mut transaction_pos = self.next_position.clone(); - - 'binlog_read: while let Some(result) = self.binlog_stream.as_mut().unwrap().next().await { - match self.local_stop_position { - Some(stop_position) if self.next_position.position >= stop_position => { - break 'binlog_read; - } - _ => {} - } - - let binlog_event = match result { - Ok(event) => event, - Err(err) => { - if is_network_failure(&err) { - self.open_binlog().await?; - continue 'binlog_read; - } else { - Err(MySQLConnectorError::BinlogReadError(err))? - } - } - }; - - let is_artificial = binlog_event - .header() - .flags() - .contains(EventFlags::LOG_EVENT_ARTIFICIAL_F); - - if is_artificial { - continue; - } - - self.next_position.position = binlog_event.header().log_pos().into(); - - let event_type = match binlog_event.header().event_type() { - Ok(event_type) => event_type, - _ => { - continue; - } - }; - - use mysql_async::binlog::{events::EventData::*, EventType::*}; - match event_type { - ROTATE_EVENT => { - let rotate_event = - match binlog_event.read_data().map_err(binlog_io_error)?.unwrap() { - RotateEvent(rotate_event) => rotate_event, - _ => unreachable!(), - }; - - let filename = rotate_event.name(); - let Some((prefix, suffix)) = filename.split_once('.') else { - Err(MySQLConnectorError::BinlogError(format!( - "Unexpected binlog filename format: {filename:?}" - )))? - }; - let rotated_binlog_id = suffix.parse::().map_err(|err| { - MySQLConnectorError::BinlogError(format!( - "Unexpected binlog filename format: {filename:?}: {err}" - )) - })?; - - if rotated_binlog_id != self.next_position.binlog_id - || self.binlog_prefix != prefix - { - self.next_position = BinlogPosition { - binlog_id: rotated_binlog_id, - position: rotate_event.position(), - }; - - self.binlog_prefix = prefix.to_string(); - self.open_binlog().await?; - } - - table_cache.handle_binlog_rotate(); - } - - QUERY_EVENT => { - let query_event = - match binlog_event.read_data().map_err(binlog_io_error)?.unwrap() { - QueryEvent(query_event) => query_event, - _ => unreachable!(), - }; - - let query = query_event.query_raw().trim_start(); - - if query == b"BEGIN" { - transaction_pos.binlog_id = self.next_position.binlog_id; - transaction_pos.position = (binlog_event.header().log_pos() - - binlog_event.header().event_size()) - as u64; - } else if query.starts_with_case_insensitive(b"ALTER") - || query.starts_with_case_insensitive(b"DROP") - { - if schema_change_tracker.unknown_schema_change_occured { - // An unknown schema change has occured before, so granular checks might be inaccurate. - // The pending full schema check should suffice. - } else { - let query = String::from_utf8_lossy(query); - let dialect = &sqlparser::dialect::MySqlDialect {}; - let result = sqlparser::parser::Parser::parse_sql(dialect, &query); - let schema = query_event.schema_raw(); - match result { - Err(err) => { - warn!("Failed to parse MySQL query {query:?}: {err}"); - // resort to manual schema verification - schema_change_tracker.unknown_schema_change_occured(); - } - Ok(statements) => { - for statement in statements { - use sqlparser::ast::{ - AlterTableOperation, ObjectType, Statement, - }; - match statement { - Statement::Drop { - object_type: ObjectType::Schema, - names, - .. - } => { - for name in names { - let database_name = - object_name_to_string(&name); - if table_cache - .databases() - .contains(&database_name) - { - Err(BreakingSchemaChange::DatabaseDropped( - database_name, - ))? - } - } - } - Statement::Drop { - object_type: ObjectType::Table, - names, - .. - } => { - for name in names { - if let Some(table) = table_cache - .find_table_by_object_name(&name, schema) - { - Err(BreakingSchemaChange::TableDropped( - table.to_string(), - ))? - } - } - } - Statement::AlterTable { - name, operations, .. - } => { - if let Some(table) = table_cache - .find_table_by_object_name(&name, schema) - { - let find_column = - |name: &sqlparser::ast::Ident| { - table.columns.iter().find(|cd| { - cd.name.eq_ignore_ascii_case( - &name.value, - ) - }) - }; - for operation in operations.iter() { - match operation { - AlterTableOperation::AddColumn { - .. - } => { - schema_change_tracker.column_order_changed_in(table.table_index); - } - AlterTableOperation::DropColumn { - column_name, - .. - } => { - if let Some(column) = - find_column(column_name) - { - Err(BreakingSchemaChange::ColumnDropped { table_name: table.to_string(), column_name: column.to_string()})? - } - schema_change_tracker.column_order_changed_in(table.table_index); - } - AlterTableOperation::RenameColumn { - old_column_name, - new_column_name, - } => { - if !old_column_name - .value - .eq_ignore_ascii_case( - &new_column_name.value, - ) - { - if let Some(column) = - find_column(old_column_name) - { - Err(BreakingSchemaChange::ColumnRenamed{ - table_name: table.to_string(), - old_column_name: column.to_string(), - new_column_name: new_column_name.value.clone() - })? - } - } - } - AlterTableOperation::RenameTable { - table_name, - } => { - Err(BreakingSchemaChange::TableRenamed{ - old_table_name: table.to_string(), - new_table_name: object_name_to_string(table_name), - })? - } - AlterTableOperation::ChangeColumn { - old_name, - new_name, - data_type, - options: _, // TODO: handle options changes - } => { - if let Some(column) = find_column(old_name) { - if !old_name.value.eq_ignore_ascii_case(&new_name.value) { - Err(BreakingSchemaChange::ColumnRenamed{ - table_name: table.to_string(), - old_column_name: column.to_string(), - new_column_name: new_name.value.clone(), - })? - } - let new_type = get_field_type_for_sql_type(data_type); - if new_type != column.typ { - Err(BreakingSchemaChange::ColumnDataTypeChanged{ - table_name: table.to_string(), - column_name: column.to_string(), - old_data_type: column.typ, - new_column_name: new_type, - })? - } - } - } - AlterTableOperation::AlterColumn { - column_name, - op, - } => { - if let Some(_column) = find_column(column_name) { - use sqlparser::ast::AlterColumnOperation; - match op { - AlterColumnOperation::SetDefault { .. } | - AlterColumnOperation::DropDefault => (), // TODO: handle options changes - AlterColumnOperation::SetNotNull | - AlterColumnOperation::DropNotNull | - AlterColumnOperation::SetDataType {..} => unreachable!("MySQL does not support this statement {}", operation), - } - } - } - _ => (), - } - } - } - } - _ => (), - } - } - } - } - } - } - } - - XID_EVENT => { - if self - .ingestor - .handle_message(IngestionMessage::TransactionInfo( - TransactionInfo::Commit { - id: Some(encode_state(&transaction_pos)), - source_time: None, - }, - )) - .await - .is_err() - { - return Ok(()); - } - } - - WRITE_ROWS_EVENT | UPDATE_ROWS_EVENT | DELETE_ROWS_EVENT | WRITE_ROWS_EVENT_V1 - | UPDATE_ROWS_EVENT_V1 | DELETE_ROWS_EVENT_V1 => { - let event_data = - match binlog_event.read_data().map_err(binlog_io_error)?.unwrap() { - RowsEvent(event_data) => event_data, - _ => unreachable!(), - }; - - let rows_event = BinlogRowsEvent(event_data); - - let binlog_table_id = rows_event.table_id(); - - let tme = self.get_tme(binlog_table_id)?; - - if schema_change_tracker.unknown_schema_change_occured { - table_cache.refresh_full_schema(schema_helper).await?; - schema_change_tracker.clear(); - } - - if let Some(table_index) = table_cache.get_corresponding_table_index(tme) { - if schema_change_tracker - .column_order_changed - .contains(&table_index) - { - let tables = &schema_change_tracker.column_order_changed; - table_cache - .refresh_column_ordinals(schema_helper, tables) - .await?; - schema_change_tracker.clear(); - } - - let table = table_cache.get_table_details(table_index).unwrap(); - - self.handle_rows_event(&rows_event, &table, tme).await?; - } - } - - event_type => { - trace!("other binlog event {event_type:?}"); - } - } - } - - Ok(()) - } - - async fn handle_rows_event<'a>( - &self, - rows_event: &BinlogRowsEvent<'_>, - table: &TableDetails<'_>, - tme: &TableMapEvent<'a>, - ) -> Result<(), MySQLConnectorError> { - for op in self.make_rows_operations(rows_event, table, tme) { - if self - .ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index: table.def.table_index, - op: op?, - id: Some(encode_state(&self.next_position)), - }) - .await - .is_err() - { - // If receiving side is closed, we can stop ingesting. - return Ok(()); - } - } - - Ok(()) - } - - fn get_tme(&self, binlog_table_id: u64) -> Result<&TableMapEvent<'_>, MySQLConnectorError> { - self.binlog_stream - .as_ref() - .unwrap() - .get_tme(binlog_table_id) - .ok_or_else(|| { - MySQLConnectorError::BinlogError(format!( - "Missing table-map-event for table_id: {binlog_table_id}" - )) - }) - } - - // Select the intersection between the columns present in the row and the columns in the table. - // - // Returns the index and field type of each selected column. - // - // # Parameters - // - `row_columns`: The zero-based indexes of columns present in the binlog row. - // - `table_definition`: The table definition. - fn select_columns<'a>( - &self, - row_columns: Vec, - table: &TableDetails<'a>, - ) -> Vec<(usize, &'a FieldType)> { - let columns = row_columns - .iter() - .enumerate() - .filter_map(|(i, col)| table.columns.get(col).map(|cd| (i, &cd.typ))) - .collect::>(); - - columns - } - - fn rows_iter<'a: 'r, 'b: 'r, 'c: 'r, 'd: 'r, 'r>( - &'a self, - rows_event: &'b BinlogRowsEvent<'_>, - table: &TableDetails<'c>, - tme: &'d TableMapEvent, - ) -> impl Iterator> + 'r { - let rows = rows_event.rows(tme); - let selected_columns = ( - rows_event - .columns_before_image() - .map(|rows| self.select_columns(rows.collect(), table)), - rows_event - .columns_after_image() - .map(|rows| self.select_columns(rows.collect(), table)), - ); - - rows.map(move |row| -> Result { - fn into_fields( - row: Option, - selected_columns: Option<&Vec<(usize, &FieldType)>>, - ) -> Result>, MySQLConnectorError> { - let value = if let Some(row) = row { - Some(row.into_fields(selected_columns.unwrap())?) - } else { - None - }; - Ok(value) - } - - let row = row.map_err(binlog_io_error)?; - - let old_values = into_fields(row.0, selected_columns.0.as_ref())?; - let new_values = into_fields(row.1, selected_columns.1.as_ref())?; - - Ok(RowValues { - old_values, - new_values, - }) - }) - } - - fn make_rows_operations<'a: 'r, 'b: 'r, 'c: 'r, 'd: 'r, 'r>( - &'a self, - rows_event: &'b BinlogRowsEvent<'_>, - table: &TableDetails<'c>, - tme: &'d TableMapEvent, - ) -> impl Iterator> + 'r { - if rows_event.is_write() { - OperationsIter::Iter1(self.make_insert_operations(rows_event, table, tme)) - } else if rows_event.is_update() { - OperationsIter::Iter2(self.make_update_operations(rows_event, table, tme)) - } else if rows_event.is_delete() { - OperationsIter::Iter3(self.make_delete_operations(rows_event, table, tme)) - } else { - unreachable!() - } - } - - fn make_insert_operations<'a: 'r, 'b: 'r, 'c: 'r, 'd: 'r, 'r>( - &'a self, - rows_event: &'b BinlogRowsEvent<'_>, - table: &TableDetails<'c>, - tme: &'d TableMapEvent, - ) -> impl Iterator> + 'r { - self.rows_iter(rows_event, table, tme).map(|row| { - let RowValues { new_values, .. } = row?; - - let op: Operation = Operation::Insert { - new: Record::new(new_values.unwrap()), - }; - - Ok(op) - }) - } - - fn make_update_operations<'a: 'r, 'b: 'r, 'c: 'r, 'd: 'r, 'r>( - &'a self, - rows_event: &'b BinlogRowsEvent<'_>, - table: &TableDetails<'c>, - tme: &'d TableMapEvent, - ) -> impl Iterator> + 'r { - self.rows_iter(rows_event, table, tme).map(|row| { - let RowValues { - old_values, - new_values, - } = row?; - - let op: Operation = Operation::Update { - old: Record::new(old_values.unwrap()), - new: Record::new(new_values.unwrap()), - }; - - Ok(op) - }) - } - - fn make_delete_operations<'a: 'r, 'b: 'r, 'c: 'r, 'd: 'r, 'r>( - &'a self, - rows_event: &'b BinlogRowsEvent<'_>, - table: &TableDetails<'c>, - tme: &'d TableMapEvent, - ) -> impl Iterator> + 'r { - self.rows_iter(rows_event, table, tme).map(|row| { - let RowValues { old_values, .. } = row?; - - let op: Operation = Operation::Delete { - old: Record::new(old_values.unwrap()), - }; - - Ok(op) - }) - } -} - -struct SchemaChangeTracker { - /// Table indexes for tables which had a column order change, usually caused by column addition or dropping. - pub column_order_changed: HashSet, - /// Some unkown schema change has occured. A a full schema refresh is needed and a rigourous check for breaking changes. - pub unknown_schema_change_occured: bool, -} - -impl SchemaChangeTracker { - fn new() -> Self { - Self { - column_order_changed: HashSet::new(), - unknown_schema_change_occured: false, - } - } - - fn column_order_changed_in(&mut self, table_index: usize) { - self.column_order_changed.insert(table_index); - } - - fn unknown_schema_change_occured(&mut self) { - self.unknown_schema_change_occured = true; - } - - fn clear(&mut self) { - self.column_order_changed.clear(); - self.unknown_schema_change_occured = false; - } -} - -enum OperationsIter { - Iter1(I1), - Iter2(I2), - Iter3(I3), -} - -impl Iterator for OperationsIter -where - I1: Iterator, - I2: Iterator, - I3: Iterator, -{ - type Item = I3::Item; - - fn next(&mut self) -> Option { - match self { - OperationsIter::Iter1(iter) => iter.next(), - OperationsIter::Iter2(iter) => iter.next(), - OperationsIter::Iter3(iter) => iter.next(), - } - } -} - -struct RowValues { - pub old_values: Option>, // present in Update and Delete operations - pub new_values: Option>, // present in Update and Insert operations -} - -struct ColumnDefinitionsCache { - cache: Vec>, -} - -impl ColumnDefinitionsCache { - pub fn new(table_definitions: &[TableDefinition]) -> Self { - Self { - cache: table_definitions - .iter() - .map(|td| { - td.columns - .iter() - .enumerate() - .map(|(i, c)| ((c.ordinal_position - 1) as usize, i)) - .collect() - }) - .collect(), - } - } - - pub fn get_columns_of_table<'a>( - &self, - td: &'a TableDefinition, - ) -> Option> { - self.cache.get(td.table_index).map(|hashmap| { - hashmap - .iter() - .map(|(&zero_based_ordinal, &i)| (zero_based_ordinal, &td.columns[i])) - .collect() - }) - } -} - -pub fn binlog_io_error(err: std::io::Error) -> MySQLConnectorError { - MySQLConnectorError::BinlogReadError(mysql_async::Error::Io(mysql_async::IoError::Io(err))) -} - -impl<'a> IntoFields<'a> for BinlogRow { - type Ctx = &'a [(usize, &'a FieldType)]; - - fn into_fields( - self, - selected_columns: &[(usize, &FieldType)], - ) -> Result, MySQLConnectorError> { - let mut binlog_row = self; - let mut fields = Vec::new(); - for (i, field_type) in selected_columns.iter().copied() { - let value = binlog_row.take(i); - fields.push(value.into_field(field_type)?); - } - Ok(fields) - } -} - -impl<'a, 'b> IntoField<'a> for Option> { - type Ctx = &'a FieldType; - - fn into_field(self, field_type: &FieldType) -> Result { - let binlog_value = self; - - if binlog_value.is_none() { - return Ok(Field::Null); - } - - let field = match binlog_value.unwrap() { - BinlogValue::Value(value) => value.into_field(field_type)?, - BinlogValue::Jsonb(value) => Field::Json(value.into_json_value()?), - BinlogValue::JsonDiff(_) => todo!(), - }; - - Ok(field) - } -} - -impl<'a> IntoJsonValue for binlog::jsonb::Value<'a> { - fn into_json_value(self) -> Result { - use binlog::jsonb::Value::*; - let json_value = match self { - Null => JsonValue::NULL, - Bool(v) => v.into(), - I16(v) => v.into(), - U16(v) => v.into(), - I32(v) => v.into(), - U32(v) => v.into(), - I64(v) => v.into(), - U64(v) => v.into(), - F64(v) => v.into(), - String(v) => (&*v.str()).into(), - SmallArray(v) => v.into_json_value()?, - LargeArray(v) => v.into_json_value()?, - SmallObject(v) => v.into_json_value()?, - LargeObject(v) => v.into_json_value()?, - Opaque(v) => { - let object: [(&str, JsonValue); 2] = [ - ("value_type", (v.value_type() as u8).into()), - ("data", (&*v.data()).into()), - ]; - object.into_iter().collect::().into() - } - }; - - Ok(json_value) - } -} - -impl<'a, T: StorageFormat> IntoJsonValue for ComplexValue<'a, T, Array> { - fn into_json_value(self) -> Result { - Ok(self - .iter() - .map(|value| value.map_err(binlog_io_error)?.into_json_value()) - .collect::>()? - .into()) - } -} - -impl<'a, T: StorageFormat> IntoJsonValue for ComplexValue<'a, T, Object> { - fn into_json_value(self) -> Result { - Ok(self - .iter() - .map(|entry| { - let (key, value) = entry.map_err(binlog_io_error)?; - Ok((key.value().into_owned(), value.into_json_value()?)) - }) - .collect::>()? - .into()) - } -} - -#[derive(Debug, Clone)] -pub struct TableDetails<'a> { - pub def: &'a TableDefinition, - pub columns: HashMap, -} - -pub struct TableManager<'a> { - tables: &'a mut [TableDefinition], - binlog_table_id_to_table_index_map: HashMap, - known_missing_tme_table_ids: HashSet, - column_definitions_cache: ColumnDefinitionsCache, - databases: HashSet, -} - -impl TableManager<'_> { - pub fn new(tables: &mut [TableDefinition]) -> TableManager<'_> { - let column_definitions_cache = ColumnDefinitionsCache::new(tables); - let databases = Self::get_unique_databases_set(tables); - TableManager { - tables, - binlog_table_id_to_table_index_map: HashMap::new(), - known_missing_tme_table_ids: HashSet::new(), - column_definitions_cache, - databases, - } - } - - pub fn handle_binlog_rotate(&mut self) { - // binlog table ids are not guranteed to be consistent across binlog rotates - self.binlog_table_id_to_table_index_map.clear(); - self.known_missing_tme_table_ids.clear(); - } - - /// Reload column ordinals after ALTER TABLE ADD COLUMN or DROP COLUMN. - pub async fn refresh_column_ordinals( - &mut self, - schema_helper: SchemaHelper<'_>, - table_indexes: &HashSet, - ) -> Result<(), MySQLConnectorError> { - let mut tables_to_refresh = self - .tables - .iter_mut() - .filter(|td| table_indexes.contains(&td.table_index)) - .collect::>(); - - schema_helper - .refresh_column_ordinals(tables_to_refresh.as_mut_slice()) - .await?; - - self.column_definitions_cache = ColumnDefinitionsCache::new(self.tables); - - Ok(()) - } - - // Refresh the entire schema after an ALTER TABLE. - // This is a last resort when granular schema changes are not known. - pub async fn refresh_full_schema( - &mut self, - schema_helper: SchemaHelper<'_>, - ) -> Result<(), MySQLConnectorError> { - schema_helper - .refresh_schema_and_check_for_breaking_changes(self.tables) - .await?; - - self.column_definitions_cache = ColumnDefinitionsCache::new(self.tables); - - Ok(()) - } - - pub fn get_table_details(&self, table_index: usize) -> Option { - self.tables.get(table_index).map(|td| TableDetails { - def: td, - columns: self - .column_definitions_cache - .get_columns_of_table(td) - .unwrap(), - }) - } - - pub fn get_corresponding_table_index(&mut self, tme: &TableMapEvent<'_>) -> Option { - let binlog_table_id = tme.table_id(); - if let Some(&found) = self - .binlog_table_id_to_table_index_map - .get(&binlog_table_id) - { - Some(found) - } else if self.known_missing_tme_table_ids.contains(&binlog_table_id) { - None - } else { - // see if we can find it - if let Some(&TableDefinition { table_index, .. }) = self.tables.iter().find( - |TableDefinition { - table_name, - database_name, - .. - }| { - database_name.as_bytes() == tme.database_name_raw() - && table_name.as_bytes() == tme.table_name_raw() - }, - ) { - self.binlog_table_id_to_table_index_map - .insert(binlog_table_id, table_index); - Some(table_index) - } else { - self.known_missing_tme_table_ids.insert(binlog_table_id); - None - } - } - } - - pub fn find_table_by_object_name( - &self, - object_name: &sqlparser::ast::ObjectName, - fallback_schema: &[u8], - ) -> Option<&TableDefinition> { - let object_name = &object_name.0; - if object_name.is_empty() || object_name.len() > 2 { - return None; - } - let table_name = &object_name.last().unwrap().value; - let database_name = if object_name.len() > 1 { - object_name.first().unwrap().value.as_str().into() - } else { - String::from_utf8_lossy(fallback_schema) - }; - let database_name = database_name.deref(); - self.tables.iter().find(|td| { - td.table_name.eq_ignore_ascii_case(table_name) - && td.database_name.eq_ignore_ascii_case(database_name) - }) - } - - pub fn databases(&self) -> &HashSet { - &self.databases - } - - fn get_unique_databases_set(tables: &[TableDefinition]) -> HashSet { - tables.iter().map(|td| td.database_name.clone()).collect() - } -} - -struct BinlogRowsEvent<'a>(binlog::events::RowsEventData<'a>); - -macro_rules! rows_event_apply { - ( - $event_data:expr, - event.$($op:tt)* - ) => { - { - use mysql_async::binlog::events::RowsEventData::*; - match $event_data { - WriteRowsEvent(event) => event.$($op)*, - UpdateRowsEvent(event) => event.$($op)*, - DeleteRowsEvent(event) => event.$($op)*, - WriteRowsEventV1(event) => event.$($op)*, - UpdateRowsEventV1(event) => event.$($op)*, - DeleteRowsEventV1(event) => event.$($op)*, - _ => unreachable!(), - } - } - }; -} - -impl<'a> BinlogRowsEvent<'a> { - pub fn table_id(&self) -> u64 { - rows_event_apply!(&self.0, event.table_id()) - } - - pub fn rows(&'a self, tme: &'a TableMapEvent<'a>) -> RowsEventRows<'a> { - rows_event_apply!(&self.0, event.rows(tme)) - } - - pub fn columns_before_image(&'a self) -> Option + 'a> { - use binlog::events::RowsEventData::*; - let value = match &self.0 { - UpdateRowsEventV1(event) => Some(event.columns_before_image()), - DeleteRowsEventV1(event) => Some(event.columns_before_image()), - UpdateRowsEvent(event) => Some(event.columns_before_image()), - DeleteRowsEvent(event) => Some(event.columns_before_image()), - _ => None, - }; - value.map(|bitslice| bitslice.iter_ones()) - } - - pub fn columns_after_image(&'a self) -> Option + 'a> { - use binlog::events::RowsEventData::*; - let value = match &self.0 { - UpdateRowsEventV1(event) => Some(event.columns_after_image()), - WriteRowsEventV1(event) => Some(event.columns_after_image()), - UpdateRowsEvent(event) => Some(event.columns_after_image()), - WriteRowsEvent(event) => Some(event.columns_after_image()), - _ => None, - }; - value.map(|bitslice| bitslice.iter_ones()) - } - - pub fn is_write(&self) -> bool { - use binlog::events::RowsEventData::*; - matches!(&self.0, WriteRowsEventV1(_) | WriteRowsEvent(_)) - } - - pub fn is_update(&self) -> bool { - use binlog::events::RowsEventData::*; - matches!(&self.0, UpdateRowsEventV1(_) | UpdateRowsEvent(_)) - } - - pub fn is_delete(&self) -> bool { - use binlog::events::RowsEventData::*; - matches!(&self.0, DeleteRowsEventV1(_) | DeleteRowsEvent(_)) - } -} - -trait ByteSliceExt { - fn trim_start(&self) -> &[u8]; - fn starts_with_case_insensitive(&self, prefix: &[u8]) -> bool; -} - -impl ByteSliceExt for [u8] { - fn trim_start(&self) -> &[u8] { - for i in 0..self.len() { - if !self[i].is_ascii_whitespace() { - return &self[i..]; - } - } - &[] - } - - fn starts_with_case_insensitive(&self, prefix: &[u8]) -> bool { - if self.len() < prefix.len() { - false - } else { - self[..prefix.len()].eq_ignore_ascii_case(prefix) - } - } -} - -fn object_name_to_string(object_name: &sqlparser::ast::ObjectName) -> String { - object_name - .0 - .iter() - .map(|ident| ident.value.as_str()) - .collect::>() - .join(".") -} - -#[cfg(test)] -mod tests { - - use dozer_ingestion_connector::dozer_types::{ - json_types::json, - types::{Field, FieldType}, - }; - - use mysql_async::{ - binlog::{ - jsonb::{self, JsonbString, JsonbType, OpaqueValue}, - value::BinlogValue, - }, - consts::ColumnType, - Value, - }; - use mysql_common::io::ParseBuf; - - use crate::conversion::IntoField; - - #[test] - fn test_field_conversion() { - use jsonb::Value as JValue; - - assert_eq!( - Field::UInt(0), - Some(BinlogValue::Value(Value::UInt(0))) - .into_field(&FieldType::UInt) - .unwrap() - ); - - assert_eq!( - Field::Json(json!(null)), - Some(BinlogValue::Jsonb(JValue::Null)) - .into_field(&FieldType::Json) - .unwrap() - ); - - assert_eq!( - Field::Json(json!(2.0)), - Some(BinlogValue::Jsonb(JValue::I16(2))) - .into_field(&FieldType::Json) - .unwrap() - ); - - assert_eq!( - Field::Json(json!(3.0)), - Some(BinlogValue::Jsonb(JValue::I32(3))) - .into_field(&FieldType::Json) - .unwrap() - ); - - assert_eq!( - Field::Json(json!(4.0)), - Some(BinlogValue::Jsonb(JValue::U16(4))) - .into_field(&FieldType::Json) - .unwrap() - ); - - assert_eq!( - Field::Json(json!(5.0)), - Some(BinlogValue::Jsonb(JValue::U32(5))) - .into_field(&FieldType::Json) - .unwrap() - ); - - assert_eq!( - Field::Json(json!(6.0)), - Some(BinlogValue::Jsonb(JValue::I64(6))) - .into_field(&FieldType::Json) - .unwrap() - ); - - assert_eq!( - Field::Json(json!(7.0)), - Some(BinlogValue::Jsonb(JValue::U64(7))) - .into_field(&FieldType::Json) - .unwrap() - ); - - assert_eq!( - Field::Json(json!(8.0)), - Some(BinlogValue::Jsonb(JValue::F64(8.0))) - .into_field(&FieldType::Json) - .unwrap() - ); - - assert_eq!( - Field::Json(json!("9")), - Some(BinlogValue::Jsonb(JValue::String(JsonbString::new(vec![ - b'9' - ])))) - .into_field(&FieldType::Json) - .unwrap() - ); - - assert_eq!( - Field::Json(json!([10.0])), - Some(BinlogValue::Jsonb(JValue::SmallArray( - ParseBuf(&[1, 0, 7, 0, JsonbType::JSONB_TYPE_INT16 as u8, 10, 0]) - .parse(()) - .unwrap(), - ))) - .into_field(&FieldType::Json) - .unwrap() - ); - - assert_eq!( - Field::Json(json!([])), - Some(BinlogValue::Jsonb(JValue::LargeArray( - ParseBuf(&[0, 0, 0, 0, 8, 0, 0, 0]).parse(()).unwrap(), - ))) - .into_field(&FieldType::Json) - .unwrap() - ); - - assert_eq!( - Field::Json(json!({"k": 12.0})), - Some(BinlogValue::Jsonb(JValue::SmallObject( - ParseBuf(&[ - 1, - 0, - 12, - 0, - 11, - 0, - 1, - 0, - JsonbType::JSONB_TYPE_INT16 as u8, - 12, - 0, - b'k', - ]) - .parse(()) - .unwrap(), - ))) - .into_field(&FieldType::Json) - .unwrap() - ); - - assert_eq!( - Field::Json(json!({})), - Some(BinlogValue::Jsonb(JValue::LargeObject( - ParseBuf(&[0, 0, 0, 0, 8, 0, 0, 0]).parse(()).unwrap(), - ))) - .into_field(&FieldType::Json) - .unwrap() - ); - - assert_eq!( - Field::Json(json!({"value_type": ColumnType::MYSQL_TYPE_TINY as u8, "data": "a"})), - Some(BinlogValue::Jsonb(JValue::Opaque(OpaqueValue::new( - ColumnType::MYSQL_TYPE_TINY, - vec![b'a'] - )))) - .into_field(&FieldType::Json) - .unwrap() - ); - - assert_eq!( - Field::Json(json!(true)), - Some(BinlogValue::Jsonb(JValue::Bool(true))) - .into_field(&FieldType::Json) - .unwrap() - ); - - assert_eq!( - Field::Null, - None::>.into_field(&FieldType::Int).unwrap(), - ); - } -} diff --git a/dozer-ingestion/mysql/src/connection.rs b/dozer-ingestion/mysql/src/connection.rs deleted file mode 100644 index db444ed973..0000000000 --- a/dozer-ingestion/mysql/src/connection.rs +++ /dev/null @@ -1,210 +0,0 @@ -use dozer_ingestion_connector::{ - dozer_types, retry_on_network_failure, - tokio::{self, sync::mpsc::Receiver}, -}; -use mysql_async::{prelude::*, BinlogStream, Params, Pool, Row}; - -#[derive(Debug)] -pub struct Conn { - pool: mysql_async::Pool, - inner: mysql_async::Conn, -} - -impl Conn { - pub async fn new(pool: mysql_async::Pool) -> Result { - let conn = new_mysql_connection(&pool).await?; - Ok(Conn { pool, inner: conn }) - } - - pub async fn exec_first<'a: 'b, 'b, T, P>( - &'a mut self, - query: &str, - params: P, - ) -> Result, mysql_async::Error> - where - P: Into + Send + Copy + 'b, - T: FromRow + Send + 'static, - { - retry_on_network_failure!( - "query", - self.inner.exec_first(query, params).await, - is_network_failure, - self.reconnect().await? - ) - } - - pub fn exec_iter(&mut self, query: String, params: Vec) -> QueryResult { - exec_iter_impl(self.pool.clone(), query, params) - } - - #[allow(unused)] - pub async fn exec_drop<'a: 'b, 'b, P>( - &'a mut self, - query: &str, - params: P, - ) -> Result<(), mysql_async::Error> - where - P: Into + Send + Copy + 'b, - { - retry_on_network_failure!( - "query", - self.inner.exec_drop(query, params).await, - is_network_failure, - self.reconnect().await? - ) - } - - pub async fn query_drop(&mut self, query: &str) -> Result<(), mysql_async::Error> { - retry_on_network_failure!( - "query", - self.inner.query_drop(query).await, - is_network_failure, - self.reconnect().await? - ) - } - - pub async fn get_binlog_stream( - self, - server_id: u32, - filename: &[u8], - pos: u64, - ) -> Result { - let mut inner = self.inner; - retry_on_network_failure!( - "get_binlog_stream", - { - let request = mysql_async::BinlogStreamRequest::new(server_id) - .with_filename(filename) - .with_pos(pos); - inner.get_binlog_stream(request).await - }, - is_network_failure, - inner = new_mysql_connection(&self.pool).await? - ) - } - - async fn reconnect(&mut self) -> Result<(), mysql_async::Error> { - self.inner = new_mysql_connection(&self.pool).await?; - Ok(()) - } -} - -async fn new_mysql_connection(pool: &Pool) -> Result { - retry_on_network_failure!("connect", pool.get_conn().await, is_network_failure) -} - -pub fn is_network_failure(err: &mysql_async::Error) -> bool { - use mysql_async::DriverError::*; - use mysql_async::Error::*; - matches!(err, Driver(ConnectionClosed) | Io(_)) -} - -fn add_query_offset(query: &str, offset: u64) -> String { - assert!([(7, "SELECT ".to_string()), (5, "SHOW ".to_string())] - .iter() - .any(|(len, prefix)| { - query - .trim_start() - .get(0..*len) - .map(|s| s.to_uppercase() == *prefix) - .unwrap_or(false) - })); - - if offset == 0 { - query.into() - } else { - format!("{query} LIMIT {offset},18446744073709551615") - } -} - -fn exec_iter_impl(pool: Pool, query: String, params: Vec) -> QueryResult { - // this is basically a generator/coroutine using a channel to communicate the results - let (sender, receiver) = tokio::sync::mpsc::channel(10); - - tokio::spawn(async move { - let mut cursor_position: u64 = 0; - 'main: loop { - let mut conn = match new_mysql_connection(&pool).await { - Ok(conn) => conn, - Err(err) => { - let _ = sender.send(Err(err)).await; - break; - } - }; - let mut rows = match retry_on_network_failure!( - "query", - conn.exec_iter(add_query_offset(&query, cursor_position), ¶ms) - .await, - is_network_failure, - continue 'main - ) { - Ok(rows) => rows, - Err(err) => { - let _ = sender.send(Err(err)).await; - break; - } - }; - loop { - let result = retry_on_network_failure!( - "query", - rows.next().await, - is_network_failure, - continue 'main - ); - let stop = result.is_err() || result.as_ref().unwrap().is_none(); - if sender.send(result).await.is_err() { - break; - } - if stop { - break; - } - cursor_position += 1; - } - break; - } - }); - - QueryResult::new(receiver) -} - -pub struct QueryResult { - receiver: Receiver, mysql_async::Error>>, -} - -impl QueryResult { - pub fn new(receiver: Receiver, mysql_async::Error>>) -> Self { - Self { receiver } - } - - pub async fn next(&mut self) -> Option> { - self.receiver.recv().await?.transpose() - } - - pub async fn map(&mut self, mut fun: F) -> Result, mysql_async::Error> - where - F: FnMut(Row) -> U, - { - let mut acc = Vec::new(); - while let Some(result) = self.next().await { - let row = result?; - acc.push(fun(mysql_async::from_row(row))); - } - Ok(acc) - } - - pub async fn reduce( - &mut self, - mut init: U, - mut fun: F, - ) -> Result - where - F: FnMut(U, T) -> U, - T: FromRow + Send + 'static, - { - while let Some(result) = self.next().await { - let row = result?; - init = fun(init, mysql_async::from_row(row)); - } - Ok(init) - } -} diff --git a/dozer-ingestion/mysql/src/connector.rs b/dozer-ingestion/mysql/src/connector.rs deleted file mode 100644 index 3229335cc7..0000000000 --- a/dozer-ingestion/mysql/src/connector.rs +++ /dev/null @@ -1,917 +0,0 @@ -use crate::MySQLConnectorError; - -use super::{ - binlog::{get_binlog_format, get_master_binlog_position, BinlogIngestor, BinlogPosition}, - connection::Conn, - conversion::IntoFields, - helpers::{escape_identifier, qualify_table_name}, - schema::{ColumnDefinition, SchemaHelper, TableDefinition}, -}; -use crate::MySQLConnectorError::BinlogQueryError; -use dozer_ingestion_connector::{ - async_trait, - dozer_types::{ - errors::internal::BoxedError, - log::info, - models::ingestion_types::IngestionMessage, - models::ingestion_types::TransactionInfo, - node::OpIdentifier, - types::{FieldDefinition, FieldType, Operation, Record, Schema, SourceDefinition}, - }, - utils::TableNotFound, - CdcType, Connector, Ingestor, SourceSchema, SourceSchemaResult, TableIdentifier, TableInfo, -}; -use mysql_async::{Opts, Pool}; -use mysql_common::Row; -use rand::Rng; - -#[derive(Debug)] -pub struct MySQLConnector { - conn_url: String, - conn_pool: Pool, - server_id: Option, -} - -pub fn mysql_connection_opts_from_url(url: &str) -> Result { - Opts::from_url(url).map_err(MySQLConnectorError::InvalidConnectionURLError) -} - -impl MySQLConnector { - pub fn new(conn_url: String, opts: Opts, server_id: Option) -> MySQLConnector { - MySQLConnector { - conn_url, - conn_pool: Pool::new(opts), - server_id, - } - } -} - -#[async_trait] -impl Connector for MySQLConnector { - fn types_mapping() -> Vec<(String, Option)> - where - Self: Sized, - { - vec![ - ("decimal".into(), Some(FieldType::Decimal)), - ("tinyint unsigned".into(), Some(FieldType::UInt)), - ("tinyint".into(), Some(FieldType::Int)), - ("smallint unsigned".into(), Some(FieldType::UInt)), - ("smallint".into(), Some(FieldType::Int)), - ("mediumint unsigned".into(), Some(FieldType::UInt)), - ("mediumint".into(), Some(FieldType::Int)), - ("int unsigned".into(), Some(FieldType::UInt)), - ("int".into(), Some(FieldType::Int)), - ("bigint unsigned".into(), Some(FieldType::UInt)), - ("bigint".into(), Some(FieldType::Int)), - ("float".into(), Some(FieldType::Float)), - ("double".into(), Some(FieldType::Float)), - ("timestamp".into(), Some(FieldType::Timestamp)), - ("time".into(), Some(FieldType::Duration)), - ("year".into(), Some(FieldType::Int)), - ("date".into(), Some(FieldType::Date)), - ("datetime".into(), Some(FieldType::Timestamp)), - ("varchar".into(), Some(FieldType::Text)), - ("varbinary".into(), Some(FieldType::Binary)), - ("char".into(), Some(FieldType::String)), - ("binary".into(), Some(FieldType::Binary)), - ("tinyblob".into(), Some(FieldType::Binary)), - ("blob".into(), Some(FieldType::Binary)), - ("mediumblob".into(), Some(FieldType::Binary)), - ("longblob".into(), Some(FieldType::Binary)), - ("tinytext".into(), Some(FieldType::Text)), - ("text".into(), Some(FieldType::Text)), - ("mediumtext".into(), Some(FieldType::Text)), - ("longtext".into(), Some(FieldType::Text)), - ("point".into(), Some(FieldType::Point)), - ("json".into(), Some(FieldType::Json)), - ("bit".into(), Some(FieldType::Int)), - ("enum".into(), Some(FieldType::String)), - ("set".into(), Some(FieldType::String)), - ("null".into(), None), - ("linestring".into(), None), - ("polygon".into(), None), - ("multipoint".into(), None), - ("multilinestring".into(), None), - ("multipolygon".into(), None), - ("geomcollection".into(), None), - ("geometry".into(), None), - ] - } - - async fn validate_connection(&mut self) -> Result<(), BoxedError> { - let _ = self.connect().await?; - - Ok(()) - } - - async fn list_tables(&mut self) -> Result, BoxedError> { - let tables = self.schema_helper().list_tables().await?; - Ok(tables) - } - - async fn validate_tables(&mut self, tables: &[TableIdentifier]) -> Result<(), BoxedError> { - let existing_tables = self.list_tables().await?; - for table in tables { - if !existing_tables.contains(table) { - Err(TableNotFound { - schema: table.schema.clone(), - name: table.name.clone(), - })?; - } - } - - Ok(()) - } - - async fn list_columns( - &mut self, - tables: Vec, - ) -> Result, BoxedError> { - let tables_infos = self.schema_helper().list_columns(tables).await?; - Ok(tables_infos) - } - - async fn get_schemas( - &mut self, - table_infos: &[TableInfo], - ) -> Result, BoxedError> { - if table_infos.is_empty() { - return Ok(Vec::new()); - } - - let table_definitions = self - .schema_helper() - .get_table_definitions(table_infos) - .await?; - - let binlog_logging_format: String = get_binlog_format(&mut self.connect().await?).await?; - let cdc_type = if binlog_logging_format == "ROW" { - CdcType::FullChanges - } else { - CdcType::Nothing - }; - - let schemas = table_definitions - .into_iter() - .map(|TableDefinition { columns, .. }| { - let primary_index = columns - .iter() - .enumerate() - .filter(|(_, ColumnDefinition { primary_key, .. })| *primary_key) - .map(|(i, _)| i) - .collect(); - Ok(SourceSchema { - schema: Schema { - fields: columns - .into_iter() - .map( - |ColumnDefinition { - name, - typ, - nullable, - .. - }| { - FieldDefinition { - name, - typ, - nullable, - source: SourceDefinition::Dynamic, - description: None, - } - }, - ) - .collect(), - primary_index, - }, - cdc_type, - }) - }) - .collect(); - - Ok(schemas) - } - - async fn serialize_state(&self) -> Result, BoxedError> { - Ok(vec![]) - } - - async fn start( - &mut self, - ingestor: &Ingestor, - tables: Vec, - last_checkpoint: Option, - ) -> Result<(), BoxedError> { - self.replicate(ingestor, tables, last_checkpoint) - .await - .map_err(Into::into) - } -} - -impl MySQLConnector { - async fn connect(&self) -> Result { - Conn::new(self.conn_pool.clone()) - .await - .map_err(|err| MySQLConnectorError::ConnectionFailure(self.conn_url.clone(), err)) - } - - fn schema_helper(&self) -> SchemaHelper<'_> { - SchemaHelper::new(&self.conn_url, &self.conn_pool) - } - - async fn replicate( - &self, - ingestor: &Ingestor, - table_infos: Vec, - last_checkpoint: Option, - ) -> Result<(), MySQLConnectorError> { - let mut table_definitions = self - .schema_helper() - .get_table_definitions( - table_infos - .iter() - .map(|table| TableInfo { - schema: table.schema.clone(), - name: table.name.clone(), - column_names: table.column_names.clone(), - }) - .collect::>() - .as_slice(), - ) - .await?; - - let binlog_position = last_checkpoint - .map(crate::binlog::BinlogPosition::try_from) - .transpose()?; - - let binlog_positions = self - .replicate_tables(ingestor, &table_definitions, binlog_position) - .await?; - - let binlog_position = self.sync_with_binlog(ingestor, binlog_positions).await?; - - let prefix = self.get_prefix(binlog_position.binlog_id).await?; - - info!("Ingestion starting at {:?}", binlog_position); - self.ingest_binlog( - ingestor, - &mut table_definitions, - binlog_position, - None, - prefix, - ) - .await?; - - Ok(()) - } - - async fn get_prefix(&self, suffix: u64) -> Result { - let suffix_formatted = format!("{:0>6}", suffix); - let mut conn = self.connect().await?; - let mut rows = conn.exec_iter("SHOW BINARY LOGS".to_string(), vec![]); - - let mut prefix = None; - while let Some(result) = rows.next().await { - let binlog_id = result - .map_err(MySQLConnectorError::QueryResultError)? - .get::(0) - .ok_or(BinlogQueryError)?; - - let row_binlog_suffix = &binlog_id[binlog_id.len() - 6..]; - - if row_binlog_suffix == suffix_formatted { - if prefix.is_some() { - return Err(MySQLConnectorError::MultipleBinlogsWithSameSuffix); - } - - prefix = Some(binlog_id[..(binlog_id.len() - 7)].to_string()); - } - } - - if let Some(prefix) = prefix { - Ok(prefix) - } else { - Err(MySQLConnectorError::BinlogNotFound) - } - } - - async fn replicate_tables( - &self, - ingestor: &Ingestor, - table_definitions: &[TableDefinition], - binlog_position: Option, - ) -> Result, MySQLConnectorError> { - let mut binlog_position_per_table = Vec::new(); - - let mut conn = self.connect().await?; - - let mut snapshot_started = false; - for (table_index, td) in table_definitions.iter().enumerate() { - let position = match &binlog_position { - Some(position) => position.clone(), - _ => { - if !snapshot_started { - if ingestor - .handle_message(IngestionMessage::TransactionInfo( - TransactionInfo::SnapshottingStarted, - )) - .await - .is_err() - { - // If receiving end is closed, we should stop the replication - break; - } - snapshot_started = true; - } - - conn.query_drop(&format!( - "LOCK TABLES {} READ", - qualify_table_name(Some(&td.database_name), &td.table_name) - )) - .await - .map_err(MySQLConnectorError::QueryExecutionError)?; - - let row_count = { - let mut row: Row = conn - .exec_first( - &format!( - "SELECT COUNT(*) from {}", - qualify_table_name(Some(&td.database_name), &td.table_name) - ), - (), - ) - .await - .map_err(MySQLConnectorError::QueryExecutionError)? - .unwrap(); - let count: u64 = row.take(0).unwrap(); - count - }; - - if row_count == 0 { - conn.query_drop("UNLOCK TABLES") - .await - .map_err(MySQLConnectorError::QueryExecutionError)?; - continue; - } - - let mut rows = conn.exec_iter( - format!( - "SELECT {} from {}", - td.columns - .iter() - .map(|ColumnDefinition { name, .. }| escape_identifier(name)) - .collect::>() - .join(", "), - qualify_table_name(Some(&td.database_name), &td.table_name) - ), - vec![], - ); - - let field_types: Vec = td - .columns - .iter() - .map(|ColumnDefinition { typ, .. }| *typ) - .collect(); - - while let Some(result) = rows.next().await { - let row = result.map_err(MySQLConnectorError::QueryResultError)?; - let op: Operation = Operation::Insert { - new: Record::new(row.into_fields(&field_types)?), - }; - - if ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index, - op, - id: None, - }) - .await - .is_err() - { - // If receiving end is closed, we should stop the replication - break; - } - } - - let (_prefix, binlog_position) = get_master_binlog_position(&mut conn).await?; - - conn.query_drop("UNLOCK TABLES") - .await - .map_err(MySQLConnectorError::QueryExecutionError)?; - - binlog_position - } - }; - - binlog_position_per_table.push((td.clone(), position)); - } - - if snapshot_started - && ingestor - .handle_message(IngestionMessage::TransactionInfo( - TransactionInfo::SnapshottingDone { id: None }, - )) - .await - .is_err() - { - return Err(MySQLConnectorError::SnapshotIngestionMessageError); - } - - Ok(binlog_position_per_table) - } - - async fn sync_with_binlog( - &self, - ingestor: &Ingestor, - binlog_positions: Vec<(TableDefinition, BinlogPosition)>, - ) -> Result { - assert!(!binlog_positions.is_empty()); - - let position = { - let mut last_position: Option = None; - let mut synced_tables = Vec::new(); - - for (table, position) in binlog_positions.into_iter() { - synced_tables.push(table); - - if let Some(start_position) = last_position { - let end_position = position.clone(); - - let prefix = self.get_prefix(start_position.binlog_id).await?; - - self.ingest_binlog( - ingestor, - &mut synced_tables, - start_position, - Some(end_position), - prefix, - ) - .await?; - } - - last_position = Some(position); - } - - last_position.unwrap() - }; - - Ok(position) - } - - async fn ingest_binlog( - &self, - ingestor: &Ingestor, - tables: &mut [TableDefinition], - start_position: BinlogPosition, - stop_position: Option, - binlog_prefix: String, - ) -> Result<(), MySQLConnectorError> { - let server_id = self.server_id.unwrap_or_else(|| rand::thread_rng().gen()); - - let mut binlog_ingestor = BinlogIngestor::new( - ingestor, - start_position, - stop_position, - server_id, - (&self.conn_pool, &self.conn_url), - binlog_prefix, - ); - - binlog_ingestor.ingest(tables, self.schema_helper()).await - } -} - -#[cfg(test)] -mod tests { - use crate::{ - connection::Conn, - tests::{create_test_table, mariadb_test_config, mysql_test_config, TestConfig}, - }; - - use super::MySQLConnector; - use dozer_ingestion_connector::{ - dozer_types::{ - models::ingestion_types::{IngestionMessage, TransactionInfo}, - types::{ - Field, FieldDefinition, FieldType, Operation::*, Record, Schema, SourceDefinition, - }, - }, - tokio, CdcType, Connector, IngestionIterator, Ingestor, SourceSchema, TableIdentifier, - }; - use serial_test::serial; - use std::time::Duration; - - struct TestCtx { - pub connector: MySQLConnector, - pub ingestor: Ingestor, - pub iterator: IngestionIterator, - } - - impl TestCtx { - async fn setup(config: &TestConfig) -> Self { - let url = config.url.clone(); - let opts = config.opts.clone(); - - let (ingestor, iterator) = Ingestor::initialize_channel(Default::default()); - let connector = MySQLConnector::new(url, opts, Some(10)); - - Self { - connector, - ingestor, - iterator, - } - } - } - - async fn check_ingestion_messages( - iterator: &mut IngestionIterator, - expected_ingestion_messages: Vec, - ) { - let mut actual_ingestion_messages = take_timeout( - iterator, - expected_ingestion_messages.len(), - Duration::from_secs(5), - ) - .await; - - // We are not checking state. - for actual in actual_ingestion_messages.iter_mut() { - match actual { - IngestionMessage::OperationEvent { id, .. } => { - *id = None; - } - IngestionMessage::TransactionInfo(TransactionInfo::Commit { - id, - source_time: None, - }) => { - *id = None; - } - _ => {} - } - } - - for (i, (actual, expected)) in std::iter::zip( - actual_ingestion_messages.iter(), - expected_ingestion_messages.iter(), - ) - .enumerate() - { - assert_eq!( - expected, actual, - "The {i}th message didn't match. Expected {expected:?}; Found {actual:?}\nThe actual message queue is {actual_ingestion_messages:?}" - ); - } - } - - async fn take_timeout( - iterator: &mut IngestionIterator, - n: usize, - timeout: Duration, - ) -> Vec { - let mut vec = Vec::new(); - for _ in 0..n { - let msg = iterator.next_timeout(timeout).await.unwrap(); - vec.push(msg); - } - vec - } - - async fn test_connector_simple_table_replication(config: TestConfig) { - // setup - let TestCtx { - connector, - ingestor, - mut iterator, - .. - } = TestCtx::setup(&config).await; - - let table_info = create_test_table("test1", &config).await; - let table_definitions = connector - .schema_helper() - .get_table_definitions(&[table_info]) - .await - .unwrap(); - - let mut conn = Conn::new(connector.conn_pool.clone()).await.unwrap(); - - // test - conn.exec_drop( - " - REPLACE INTO test1 - VALUES - (1, 'a', 1.0), - (2, 'b', 2.0), - (3, 'c', 3.0) - ", - (), - ) - .await - .unwrap(); - - let result = connector - .replicate_tables(&ingestor, &table_definitions, None) - .await; - assert!(result.is_ok(), "unexpected error: {result:?}"); - - let expected_ingestion_messages = vec![ - IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingStarted), - IngestionMessage::OperationEvent { - table_index: 0, - op: Insert { - new: Record::new(vec![ - Field::Int(1), - Field::Text("a".into()), - Field::Float(1.0.into()), - ]), - }, - id: None, - }, - IngestionMessage::OperationEvent { - table_index: 0, - op: Insert { - new: Record::new(vec![ - Field::Int(2), - Field::Text("b".into()), - Field::Float(2.0.into()), - ]), - }, - id: None, - }, - IngestionMessage::OperationEvent { - table_index: 0, - op: Insert { - new: Record::new(vec![ - Field::Int(3), - Field::Text("c".into()), - Field::Float(3.0.into()), - ]), - }, - id: None, - }, - IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingDone { id: None }), - ]; - - check_ingestion_messages(&mut iterator, expected_ingestion_messages).await; - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_simple_table_replication_mysql() { - test_connector_simple_table_replication(mysql_test_config()).await; - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_simple_table_replication_mariadb() { - test_connector_simple_table_replication(mariadb_test_config()).await; - } - - async fn test_connector_cdc(config: TestConfig) { - // setup - let TestCtx { - connector, - ingestor, - mut iterator, - .. - } = TestCtx::setup(&config).await; - - let mut table_infos = Vec::new(); - table_infos.push(create_test_table("test3", &config).await); - table_infos.push(create_test_table("test2", &config).await); - - let mut conn = Conn::new(connector.conn_pool.clone()).await.unwrap(); - - conn.exec_drop("DELETE FROM test2", ()).await.unwrap(); - conn.exec_drop("DELETE FROM test3", ()).await.unwrap(); - - let _handle = std::thread::spawn(move || { - tokio::runtime::Builder::new_current_thread() - .enable_io() - .build() - .unwrap() - .block_on(async move { - let _ = connector.replicate(&ingestor, table_infos, None).await; - }); - }); - - // test insert - conn.exec_drop("REPLACE INTO test3 VALUES (4, 4.0)", ()) - .await - .unwrap(); - - conn.exec_drop("REPLACE INTO test2 VALUES (1, 'true')", ()) - .await - .unwrap(); - - let expected_ingestion_messages = vec![ - IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingStarted), - IngestionMessage::OperationEvent { - table_index: 0, - op: Insert { - new: Record::new(vec![Field::Int(4), Field::Float(4.0.into())]), - }, - id: None, - }, - IngestionMessage::OperationEvent { - table_index: 1, - op: Insert { - new: Record::new(vec![Field::Int(1), Field::Json(true.into())]), - }, - id: None, - }, - IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingDone { id: None }), - ]; - - check_ingestion_messages(&mut iterator, expected_ingestion_messages).await; - - // test update - conn.exec_drop("UPDATE test3 SET b = 5.0 WHERE a = 4", ()) - .await - .unwrap(); - - let expected_ingestion_messages = vec![ - IngestionMessage::OperationEvent { - table_index: 0, - op: Update { - old: Record::new(vec![Field::Int(4), Field::Float(4.0.into())]), - new: Record::new(vec![Field::Int(4), Field::Float(5.0.into())]), - }, - id: None, - }, - IngestionMessage::TransactionInfo(TransactionInfo::Commit { - id: None, - source_time: None, - }), - ]; - - check_ingestion_messages(&mut iterator, expected_ingestion_messages).await; - - // test delete - conn.exec_drop("DELETE FROM test3 WHERE a = 4", ()) - .await - .unwrap(); - - let expected_ingestion_messages = vec![ - IngestionMessage::OperationEvent { - table_index: 0, - op: Delete { - old: Record::new(vec![Field::Int(4), Field::Float(5.0.into())]), - }, - id: None, - }, - IngestionMessage::TransactionInfo(TransactionInfo::Commit { - id: None, - source_time: None, - }), - ]; - - check_ingestion_messages(&mut iterator, expected_ingestion_messages).await; - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_cdc_mysql() { - test_connector_cdc(mysql_test_config()).await; - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_cdc_mariadb() { - test_connector_cdc(mariadb_test_config()).await; - } - - async fn test_connector_schemas(config: TestConfig) { - // setup - let TestCtx { mut connector, .. } = TestCtx::setup(&config).await; - - let mut expected_table_infos = Vec::new(); - expected_table_infos.push(create_test_table("test1", &config).await); - expected_table_infos.push(create_test_table("test2", &config).await); - - let expected_table_identifiers = vec![ - TableIdentifier { - schema: Some("test".into()), - name: "test1".into(), - }, - TableIdentifier { - schema: Some("test".into()), - name: "test2".into(), - }, - ]; - - // test list_tables - let result = connector.list_tables().await; - assert!(result.is_ok(), "unexpected error: {result:?}"); - let tables = result.unwrap(); - for table_identifier in expected_table_identifiers.iter() { - assert!( - tables.contains(table_identifier), - "missing {table_identifier:?} from list {tables:?}" - ); - } - - // test list_columns - let result = connector.list_columns(expected_table_identifiers).await; - assert!(result.is_ok(), "unexpected error: {result:?}"); - let table_infos = result.unwrap(); - for (i, (expected, actual)) in - std::iter::zip(expected_table_infos.iter(), table_infos.iter()).enumerate() - { - assert_eq!( - expected, actual, - "The {i}th table doesn't match! Expected {expected:?}; Found {actual:?}" - ); - } - - // test get_schemas - let result = connector.get_schemas(&table_infos).await; - assert!(result.is_ok(), "unexpected error: {result:?}"); - let source_schema_results = result.unwrap(); - - let expected_source_schema_results = [ - SourceSchema { - schema: Schema { - fields: vec![ - FieldDefinition { - name: "c1".into(), - typ: FieldType::Int, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "c2".into(), - typ: FieldType::Text, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "c3".into(), - typ: FieldType::Float, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }, - ], - primary_index: vec![0], - }, - cdc_type: CdcType::FullChanges, - }, - SourceSchema { - schema: Schema { - fields: vec![ - FieldDefinition { - name: "id".into(), - typ: FieldType::Int, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - FieldDefinition { - name: "value".into(), - typ: FieldType::Json, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }, - ], - primary_index: vec![0], - }, - cdc_type: CdcType::FullChanges, - }, - ]; - - for (i, (expected, actual)) in std::iter::zip( - expected_source_schema_results.iter(), - source_schema_results.iter(), - ) - .enumerate() - { - assert!(actual.is_ok(), "unexpected error: {actual:?}"); - let actual = actual.as_ref().unwrap(); - assert_eq!( - expected, actual, - "The {i}th table doesn't match! Expected {expected:?}; Found {actual:?}" - ); - } - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_schemas_mysql() { - test_connector_schemas(mysql_test_config()).await; - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_schemas_mariadb() { - test_connector_schemas(mariadb_test_config()).await; - } -} diff --git a/dozer-ingestion/mysql/src/conversion.rs b/dozer-ingestion/mysql/src/conversion.rs deleted file mode 100644 index 98ef8c3415..0000000000 --- a/dozer-ingestion/mysql/src/conversion.rs +++ /dev/null @@ -1,436 +0,0 @@ -use dozer_ingestion_connector::dozer_types::{ - chrono::{DateTime, NaiveDate, NaiveDateTime, Offset, Utc}, - json_types::{serde_json_to_json_value, JsonValue}, - rust_decimal::Decimal, - serde_json, - types::{DozerDuration, DozerPoint, Field, FieldType, TimeUnit}, -}; -use geozero::{wkb, GeomProcessor}; -use mysql_async::{Row, Value}; -use std::time::Duration; - -use crate::MySQLConnectorError; - -pub fn get_field_type_for_mysql_column_type( - column_type: &str, -) -> Result { - let data_type = { - let space = column_type.find(' '); - let parenthesis = column_type.find('('); - let end = { - let max = column_type.len(); - std::cmp::min(space.unwrap_or(max), parenthesis.unwrap_or(max)) - }; - &column_type[0..end] - }; - let is_array = column_type.ends_with(" array"); - let is_unsigned = column_type.contains(" unsigned"); - - if is_array { - return Err(MySQLConnectorError::UnsupportedFieldType(data_type.into())); - } - - let field_type = match data_type { - "decimal" => FieldType::Decimal, - "int" | "tinyint" | "smallint" | "mediumint" | "bigint" => { - if is_unsigned { - FieldType::UInt - } else { - FieldType::Int - } - } - "float" | "double" => FieldType::Float, - "timestamp" => FieldType::Timestamp, - "time" => FieldType::Duration, - "year" => FieldType::Int, - "date" => FieldType::Date, - "datetime" => FieldType::Timestamp, - "varchar" => FieldType::Text, - "varbinary" => FieldType::Binary, - "char" => FieldType::String, - "binary" => FieldType::Binary, - "tinyblob" => FieldType::Binary, - "blob" => FieldType::Binary, - "mediumblob" => FieldType::Binary, - "longblob" => FieldType::Binary, - "tinytext" => FieldType::Text, - "text" => FieldType::Text, - "mediumtext" => FieldType::Text, - "longtext" => FieldType::Text, - "point" => FieldType::Point, - "json" => FieldType::Json, - "bit" => FieldType::Int, - "enum" => FieldType::String, - "set" => FieldType::String, - "null" | "linestring" | "polygon" | "multipoint" | "multilinestring" | "multipolygon" - | "geomcollection" | "geometry" => { - Err(MySQLConnectorError::UnsupportedFieldType(data_type.into()))? - } - _ => Err(MySQLConnectorError::UnsupportedFieldType(data_type.into()))?, - }; - - Ok(field_type) -} - -pub fn get_field_type_for_sql_type(sql_data_type: &sqlparser::ast::DataType) -> FieldType { - use sqlparser::ast::DataType; - match sql_data_type { - DataType::Character(_) - | DataType::Char(_) - | DataType::String(_) - | DataType::Enum(_) - | DataType::Set(_) => FieldType::String, - DataType::CharacterVarying(_) - | DataType::CharVarying(_) - | DataType::Varchar(_) - | DataType::Nvarchar(_) - | DataType::Text => FieldType::Text, - DataType::Uuid - | DataType::CharacterLargeObject(_) - | DataType::CharLargeObject(_) - | DataType::Clob(_) - | DataType::Regclass - | DataType::Custom(_, _) - | DataType::Array(_) - | DataType::Struct(_) => unreachable!("MySQL does not support this type: {sql_data_type}"), - DataType::Binary(_) - | DataType::Varbinary(_) - | DataType::Blob(_) - | DataType::Bytes(_) - | DataType::Bytea => FieldType::Binary, - DataType::Numeric(_) - | DataType::Decimal(_) - | DataType::BigNumeric(_) - | DataType::BigDecimal(_) - | DataType::Dec(_) => FieldType::Decimal, - DataType::Float(_) - | DataType::Float4 - | DataType::Float64 - | DataType::Real - | DataType::Float8 - | DataType::Double - | DataType::DoublePrecision => FieldType::Float, - DataType::TinyInt(_) - | DataType::Int2(_) - | DataType::SmallInt(_) - | DataType::MediumInt(_) - | DataType::Int(_) - | DataType::Int4(_) - | DataType::Int64 - | DataType::Integer(_) - | DataType::BigInt(_) - | DataType::Int8(_) => FieldType::Int, - DataType::UnsignedTinyInt(_) - | DataType::UnsignedInt2(_) - | DataType::UnsignedSmallInt(_) - | DataType::UnsignedMediumInt(_) - | DataType::UnsignedInt(_) - | DataType::UnsignedInt4(_) - | DataType::UnsignedInteger(_) - | DataType::UnsignedBigInt(_) - | DataType::UnsignedInt8(_) => FieldType::UInt, - DataType::Bool | DataType::Boolean => FieldType::Boolean, - DataType::Date => FieldType::Date, - DataType::Time(_, _) | DataType::Interval => FieldType::Duration, - DataType::Datetime(_) | DataType::Timestamp(_, _) => FieldType::Timestamp, - DataType::JSON => FieldType::Json, - } -} - -pub trait IntoFields<'a> { - type Ctx: 'a; - - fn into_fields(self, ctx: Self::Ctx) -> Result, MySQLConnectorError>; -} - -impl<'a> IntoFields<'a> for Row { - type Ctx = &'a [FieldType]; - - fn into_fields(self, field_types: &[FieldType]) -> Result, MySQLConnectorError> { - let mut row = self; - let mut fields = Vec::new(); - for i in 0..row.len() { - let field = row - .take::(i) - .into_field(field_types.get(i).unwrap())?; - fields.push(field); - } - Ok(fields) - } -} - -pub trait IntoField<'a> { - type Ctx: 'a; - - fn into_field(self, ctx: Self::Ctx) -> Result; -} - -impl<'a> IntoField<'a> for Option { - type Ctx = &'a FieldType; - - fn into_field(self, field_type: &FieldType) -> Result { - if let Some(value) = self { - value.into_field(field_type) - } else { - Ok(Field::Null) - } - } -} - -impl<'a> IntoField<'a> for Value { - type Ctx = &'a FieldType; - - fn into_field(self, field_type: &FieldType) -> Result { - use mysql_common::value::convert::from_value_opt; - use Value::*; - - let value = self; - - let field = if let NULL = value { - Field::Null - } else { - match field_type { - FieldType::UInt => Field::UInt(from_value_opt::(value)?), - FieldType::U128 => Field::U128(from_value_opt::(value)?), - FieldType::Int => Field::Int(from_value_opt::(value)?), - FieldType::Int8 => Field::Int8(from_value_opt::(value)?), - FieldType::I128 => Field::I128(from_value_opt::(value)?), - FieldType::Float => Field::Float(from_value_opt::(value)?.into()), - FieldType::Boolean => Field::Boolean(from_value_opt::(value)?), - FieldType::String => Field::String(from_value_opt::(value)?), - FieldType::Text => Field::Text(from_value_opt::(value)?), - FieldType::Binary => Field::Binary(from_value_opt::>(value)?), - FieldType::Decimal => Field::Decimal(from_value_opt::(value)?), - FieldType::Timestamp => { - let date_time = from_value_opt::(value)?; - Field::Timestamp(DateTime::from_naive_utc_and_offset(date_time, Utc.fix())) - } - FieldType::Date => Field::Date(from_value_opt::(value)?), - FieldType::Json => { - let json = - serde_json_to_json_value(from_value_opt::(value)?)?; - Field::Json(json) - } - FieldType::Point => { - let bytes = from_value_opt::>(value)?; - let (_srid, mut wkb_point) = bytes.as_slice().split_at(4); - let mut wkb_processor = PointProcessor::new(); - wkb::process_wkb_geom(&mut wkb_point, &mut wkb_processor)?; - let point = wkb_processor.point.unwrap(); - Field::Point(point) - } - FieldType::Duration => Field::Duration(DozerDuration( - from_value_opt::(value)?, - TimeUnit::Microseconds, - )), - } - }; - - Ok(field) - } -} - -pub trait IntoJsonValue { - fn into_json_value(self) -> Result; -} - -struct PointProcessor { - point: Option, -} - -impl PointProcessor { - pub fn new() -> Self { - Self { point: None } - } -} - -impl GeomProcessor for PointProcessor { - fn dimensions(&self) -> geozero::CoordDimensions { - geozero::CoordDimensions::xy() - } - - fn xy(&mut self, x: f64, y: f64, _idx: usize) -> geozero::error::Result<()> { - self.point = Some((x, y).into()); - Ok(()) - } -} - -#[cfg(test)] -mod tests { - use super::*; - use mysql_common::{chrono::NaiveTime, Value}; - use std::time::Duration; - - #[test] - fn test_field_types() { - let supported_types = [ - // (mysql type, dozer type) - ("int", FieldType::Int), - ("int(8) unsigned zerofill", FieldType::UInt), - ("tinyint", FieldType::Int), - ("smallint(4)", FieldType::Int), - ("mediumint", FieldType::Int), - ("bigint unsigned", FieldType::UInt), - ("year", FieldType::Int), - ("decimal(5, 2)", FieldType::Decimal), - ("decimal(10, 0) unsigned", FieldType::Decimal), - ("float", FieldType::Float), - ("double", FieldType::Float), - ("timestamp(4)", FieldType::Timestamp), - ("time(3)", FieldType::Duration), - ("date", FieldType::Date), - ("datetime(6)", FieldType::Timestamp), - ("year", FieldType::Int), - ("char", FieldType::String), - ("varchar(20)", FieldType::Text), - ("text", FieldType::Text), - ("tinytext", FieldType::Text), - ("mediumtext", FieldType::Text), - ("longtext", FieldType::Text), - ("binary(5)", FieldType::Binary), - ("varbinary(10)", FieldType::Binary), - ("blob", FieldType::Binary), - ("tinyblob", FieldType::Binary), - ("mediumblob", FieldType::Binary), - ("longblob", FieldType::Binary), - ("enum('a','b','c')", FieldType::String), - ("set('1','2','3')", FieldType::String), - ("json", FieldType::Json), - ("point", FieldType::Point), - ]; - - let unsupported_types = [ - "null", - "linestring", - "polygon", - "multipoint", - "multilinestring", - "multipolygon", - "geomcollection", - "geometry", - "some fictional type", - "int array", - ]; - - for (mysql_type, expected_field_type) in supported_types { - let result = get_field_type_for_mysql_column_type(mysql_type); - assert!(result.is_ok(), "unexpected error {result:?}"); - let actual_field_type = result.unwrap(); - assert_eq!( - actual_field_type, expected_field_type, - "expected {expected_field_type:?}; got {actual_field_type:?} for mysql type {mysql_type}" - ); - } - - for mysql_type in unsupported_types { - let result = get_field_type_for_mysql_column_type(mysql_type); - assert!(result.is_err(), "expected an error; got {result:?}"); - } - } - - #[test] - fn test_field_conversion() { - assert_eq!( - Field::UInt(0), - Value::UInt(0).into_field(&FieldType::UInt).unwrap() - ); - assert_eq!( - Field::U128(1), - Value::UInt(1).into_field(&FieldType::U128).unwrap() - ); - assert_eq!( - Field::Int(2), - Value::Int(2).into_field(&FieldType::Int).unwrap() - ); - assert_eq!( - Field::I128(3), - Value::Int(3).into_field(&FieldType::I128).unwrap() - ); - assert_eq!( - Field::Float(4.0.into()), - Value::Float(4.0).into_field(&FieldType::Float).unwrap() - ); - assert_eq!( - Field::Boolean(true), - Value::Int(1).into_field(&FieldType::Boolean).unwrap() - ); - assert_eq!( - Field::String("6".into()), - Value::Bytes(b"6".as_slice().into()) - .into_field(&FieldType::String) - .unwrap() - ); - assert_eq!( - Field::Text("7".into()), - Value::Bytes(b"7".as_slice().into()) - .into_field(&FieldType::Text) - .unwrap() - ); - assert_eq!( - Field::Binary(b"8".as_slice().into()), - Value::Bytes(b"8".as_slice().into()) - .into_field(&FieldType::Binary) - .unwrap() - ); - assert_eq!( - Field::Decimal(9.into()), - Value::Int(9).into_field(&FieldType::Decimal).unwrap() - ); - assert_eq!( - Field::Timestamp(DateTime::from_naive_utc_and_offset( - NaiveDateTime::new( - NaiveDate::from_ymd_opt(2023, 8, 17).unwrap(), - NaiveTime::from_hms_micro_opt(10, 30, 0, 0).unwrap(), - ), - Utc.fix(), - )), - Value::Date(2023, 8, 17, 10, 30, 0, 0) - .into_field(&FieldType::Timestamp) - .unwrap() - ); - assert_eq!( - Field::Date(NaiveDate::from_ymd_opt(2023, 8, 11).unwrap()), - Value::Date(2023, 8, 11, 0, 0, 0, 0) - .into_field(&FieldType::Date) - .unwrap() - ); - assert_eq!( - Field::Json(12.0.into()), - Value::Bytes(b"12".as_slice().into()) - .into_field(&FieldType::Json) - .unwrap() - ); - assert_eq!( - Field::Point((13.0, 0.0).into()), - Value::Bytes( - hex::decode("0000000001010000000000000000002a400000000000000000").unwrap(), - ) - .into_field(&FieldType::Point) - .unwrap() - ); - assert_eq!( - Field::Duration(DozerDuration( - Duration::from_micros(14), - TimeUnit::Microseconds, - )), - Value::Time(false, 0, 0, 0, 0, 14) - .into_field(&FieldType::Duration) - .unwrap() - ); - - assert_eq!( - Field::Null, - Value::NULL.into_field(&FieldType::Int).unwrap() - ); - assert_eq!( - Field::Null, - None::.into_field(&FieldType::Int).unwrap() - ); - - assert!( - Value::Bytes(hex::decode("0000000001010000000000000000002a40").unwrap()) - .into_field(&FieldType::Point) - .is_err() - ); - } -} diff --git a/dozer-ingestion/mysql/src/helpers.rs b/dozer-ingestion/mysql/src/helpers.rs deleted file mode 100644 index df3b2fd415..0000000000 --- a/dozer-ingestion/mysql/src/helpers.rs +++ /dev/null @@ -1,26 +0,0 @@ -pub fn escape_identifier(identifier: &str) -> String { - format!("`{}`", identifier.replace('`', "``")) -} - -pub fn qualify_table_name(schema: Option<&str>, name: &str) -> String { - if let Some(schema) = schema { - format!("{}.{}", escape_identifier(schema), escape_identifier(name)) - } else { - escape_identifier(name) - } -} - -#[cfg(test)] -mod tests { - use crate::helpers::{escape_identifier, qualify_table_name}; - - #[test] - fn test_identifiers() { - assert_eq!(escape_identifier("test"), String::from("`test`")); - assert_eq!( - qualify_table_name(Some("db"), "test"), - String::from("`db`.`test`") - ); - assert_eq!(qualify_table_name(None, "test"), String::from("`test`")); - } -} diff --git a/dozer-ingestion/mysql/src/lib.rs b/dozer-ingestion/mysql/src/lib.rs deleted file mode 100644 index 10ba9a2447..0000000000 --- a/dozer-ingestion/mysql/src/lib.rs +++ /dev/null @@ -1,121 +0,0 @@ -use dozer_ingestion_connector::dozer_types::{ - errors::types::DeserializationError, - thiserror::{self, Error}, - types::FieldType, -}; -use geozero::error::GeozeroError; - -mod binlog; -mod connection; -pub mod connector; -mod conversion; -pub(crate) mod helpers; -mod schema; -mod state; -#[cfg(test)] -mod tests; - -#[derive(Error, Debug)] -pub enum MySQLConnectorError { - #[error("Invalid connection URL: {0:?}")] - InvalidConnectionURLError(#[source] mysql_async::UrlError), - - #[error("Failed to connect to mysql with the specified url {0}. {1}")] - ConnectionFailure(String, #[source] mysql_async::Error), - - #[error("Unsupported field type: {0}")] - UnsupportedFieldType(String), - - #[error("Invalid field value. {0}")] - InvalidFieldValue(#[from] mysql_common::FromValueError), - - #[error("Invalid json value. {0}")] - JsonDeserializationError(#[from] DeserializationError), - - #[error("Invalid geometric value. {0}")] - InvalidGeometricValue(#[from] GeozeroError), - - #[error("Failed to open binlog. {0}")] - BinlogOpenError(#[source] mysql_async::Error), - - #[error("Failed to read binlog. {0}")] - BinlogReadError(#[source] mysql_async::Error), - - #[error("Binlog error: {0}")] - BinlogError(String), - - #[error("Query failed. {0}")] - QueryExecutionError(#[source] mysql_async::Error), - - #[error("Failed to fetch query result. {0}")] - QueryResultError(#[source] mysql_async::Error), - - #[error("Schema had a breaking change: {0}")] - BreakingSchemaChange(#[from] BreakingSchemaChange), - - #[error("Failed to send snapshot completed ingestion message")] - SnapshotIngestionMessageError, - - #[error("State error: {0}")] - State(#[from] MysqlStateError), - - #[error("Binlog not found")] - BinlogNotFound, - - #[error("Fetch of binlog query failed")] - BinlogQueryError, - - #[error("Multiple binlogs with the same suffix")] - MultipleBinlogsWithSameSuffix, -} - -#[derive(Error, Debug)] -pub enum BreakingSchemaChange { - #[error("Database \"{0}\" was dropped")] - DatabaseDropped(String), - #[error("Table \"{0}\" was dropped")] - TableDropped(String), - #[error("Table \"{0}\" has been dropped or renamed")] - TableDroppedOrRenamed(String), - #[error("Multiple tables have been dropped or renamed: {}", .0.join(", "))] - MultipleTablesDroppedOrRenamed(Vec), - #[error("Table \"{old_table_name}\" was renamed to \"{new_table_name}\"")] - TableRenamed { - old_table_name: String, - new_table_name: String, - }, - #[error("Column \"{column_name}\" from table \"{table_name}\" was dropped")] - ColumnDropped { - table_name: String, - column_name: String, - }, - #[error("Column \"{old_column_name}\" from table \"{table_name}\" was renamed to \"{new_column_name}\"")] - ColumnRenamed { - table_name: String, - old_column_name: String, - new_column_name: String, - }, - #[error("Column \"{column_name}\" from table \"{table_name}\" has been dropped or renamed")] - ColumnDroppedOrRenamed { - table_name: String, - column_name: String, - }, - #[error("Multiple columns from table \"{table_name}\" have been dropped or renamed: {}", .columns.join(", "))] - MultipleColumnsDroppedOrRenamed { - table_name: String, - columns: Vec, - }, - #[error("Column \"{column_name}\" from table \"{table_name}\" changed data type from \"{old_data_type}\" to \"{new_column_name}\"")] - ColumnDataTypeChanged { - table_name: String, - column_name: String, - old_data_type: FieldType, - new_column_name: FieldType, - }, -} - -#[derive(Error, Debug)] -pub enum MysqlStateError { - #[error("Failed to read binlog position from state. Error: {0}")] - TrySliceError(#[from] std::array::TryFromSliceError), -} diff --git a/dozer-ingestion/mysql/src/schema.rs b/dozer-ingestion/mysql/src/schema.rs deleted file mode 100644 index dbe4de1ab9..0000000000 --- a/dozer-ingestion/mysql/src/schema.rs +++ /dev/null @@ -1,667 +0,0 @@ -use crate::{helpers::escape_identifier, BreakingSchemaChange, MySQLConnectorError}; - -use super::{ - connection::{Conn, QueryResult}, - conversion::get_field_type_for_mysql_column_type, -}; -use dozer_ingestion_connector::{dozer_types::types::FieldType, TableIdentifier, TableInfo}; -use mysql_async::{from_row, Pool}; -use mysql_common::Value; - -#[derive(Clone, Debug, Eq, PartialEq)] -pub struct TableDefinition { - pub table_index: usize, - pub table_name: String, - pub database_name: String, - pub columns: Vec, -} - -#[derive(Clone, Debug, Eq, PartialEq)] -pub struct ColumnDefinition { - pub ordinal_position: u32, - pub name: String, - pub typ: FieldType, - pub nullable: bool, - pub primary_key: bool, -} - -#[derive(Debug, Clone, Copy)] -pub struct SchemaHelper<'a> { - conn_url: &'a String, - conn_pool: &'a Pool, -} - -impl TableDefinition { - pub fn qualified_name(&self) -> String { - format!("{}.{}", self.database_name, self.table_name) - } -} - -impl std::fmt::Display for TableDefinition { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.write_str(self.qualified_name().as_str()) - } -} - -impl std::fmt::Display for ColumnDefinition { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.write_str(self.name.as_str()) - } -} - -impl SchemaHelper<'_> { - pub fn new<'a>(conn_url: &'a String, conn_pool: &'a Pool) -> SchemaHelper<'a> { - SchemaHelper { - conn_url, - conn_pool, - } - } - - pub async fn list_tables(&self) -> Result, MySQLConnectorError> { - let mut conn = self.connect().await?; - - let mut rows = conn.exec_iter( - " - SELECT table_name, table_schema FROM information_schema.tables - WHERE table_schema = DATABASE() - AND table_type = 'BASE TABLE' - " - .into(), - vec![], - ); - - let tables = rows - .map(|row| { - let (table_name, table_schema) = from_row(row); - - TableIdentifier { - schema: Some(table_schema), - name: table_name, - } - }) - .await - .map_err(MySQLConnectorError::QueryResultError)?; - - Ok(tables) - } - - pub async fn list_columns( - &self, - tables: Vec, - ) -> Result, MySQLConnectorError> { - if tables.is_empty() { - return Ok(Vec::new()); - } - - let mut conn = self.connect().await?; - let mut rows = Self::query_information_schema_columns( - &mut conn, - &tables, - &["table_name", "table_schema", "column_name"], - &[], - ) - .await?; - - let table_infos = rows - .reduce( - Vec::new(), - |mut table_infos, row: (String, String, String)| { - let (table_name, table_schema, column_name) = row; - - let last_table = table_infos.last_mut(); - match last_table { - Some(TableInfo { - name, - schema: Some(schema), - .. - }) if *name == table_name && *schema == table_schema => { - last_table.unwrap().column_names.push(column_name); - } - _ => table_infos.push(TableInfo { - schema: Some(table_schema), - name: table_name, - column_names: vec![column_name], - }), - } - - table_infos - }, - ) - .await - .map_err(MySQLConnectorError::QueryResultError)?; - - Ok(table_infos) - } - - pub async fn get_table_definitions<'a, T>( - &self, - table_infos: &'a [T], - ) -> Result, MySQLConnectorError> - where - TableInfoRef<'a>: From<&'a T>, - { - if table_infos.is_empty() { - return Ok(Vec::new()); - } - - let mut conn = self.connect().await?; - let mut rows = Self::query_information_schema_columns( - &mut conn, - table_infos, - &[ - "table_name", - "table_schema", - "column_name", - "column_type", - "is_nullable", - "column_key", - "ordinal_position", - ], - &[Self::MARIADB_JSON_CHECK], - ) - .await?; - - let mut table_definitions: Vec = Vec::new(); - { - while let Some(result) = rows.next().await { - let row = result.map_err(MySQLConnectorError::QueryResultError)?; - let ( - table_name, - table_schema, - column_name, - column_type, - is_nullable, - column_key, - ordinal_position, - is_mariadb_json, - ) = from_row::<(String, String, String, String, String, String, u32, bool)>(row); - - let nullable = is_nullable == "YES"; - let primary_key = column_key == "PRI"; - let column_definiton = ColumnDefinition { - ordinal_position, - name: column_name.clone(), - typ: if is_mariadb_json { - FieldType::Json - } else { - get_field_type_for_mysql_column_type(&column_type)? - }, - nullable, - primary_key, - }; - - match table_definitions.last_mut() { - Some(td) if td.table_name == table_name && td.database_name == table_schema => { - td.columns.push(column_definiton); - } - _ => { - let table_index = table_definitions.len(); - let td = TableDefinition { - table_index, - table_name, - database_name: table_schema, - columns: vec![column_definiton], - }; - table_definitions.push(td); - } - } - } - } - - Ok(table_definitions) - } - - pub async fn refresh_column_ordinals( - &self, - tables: &mut [&mut TableDefinition], - ) -> Result<(), MySQLConnectorError> { - if tables.is_empty() { - return Ok(()); - } - - let mut conn = self.connect().await?; - - let mut query_params: Vec = Vec::new(); - let query = format!( - " - SELECT {} AS table_index, {} as column_index, ordinal_position - FROM information_schema.columns - WHERE {} - ", - SqlHelper::table_index_case_expression(tables, &mut query_params), - SqlHelper::column_index_case_expression(tables, &mut query_params), - SqlHelper::select_columns_filter_predicate(tables, &mut query_params), - ); - - let mut rows = conn.exec_iter(query, query_params); - - while let Some(result) = rows.next().await { - let row = result.map_err(MySQLConnectorError::QueryResultError)?; - let (table_index, column_index, ordinal_position) = - from_row::<(usize, usize, u32)>(row); - - let table = &mut tables[table_index]; - let column = &mut table.columns[column_index]; - column.ordinal_position = ordinal_position; - } - - tables.iter_mut().for_each(|table| { - table - .columns - .sort_unstable_by_key(|column| column.ordinal_position) - }); - - Ok(()) - } - - pub async fn refresh_schema_and_check_for_breaking_changes( - &self, - tables: &mut [TableDefinition], - ) -> Result<(), MySQLConnectorError> { - let new_schema = self.get_table_definitions(tables).await?; - - // Check missing tables - if new_schema.len() != tables.len() { - let missing = tables - .iter() - .filter( - |TableDefinition { - table_name, - database_name, - .. - }| { - !new_schema.iter().any(|td| { - td.table_name.eq(table_name) && td.database_name.eq(database_name) - }) - }, - ) - .collect::>(); - if missing.len() == 1 { - Err(BreakingSchemaChange::TableDroppedOrRenamed( - missing[0].to_string(), - ))? - } else { - Err(BreakingSchemaChange::MultipleTablesDroppedOrRenamed( - missing.iter().map(|td| td.to_string()).collect::>(), - ))? - } - } - - // Check missing columns and data type changes - for (old, new) in tables.iter_mut().zip(new_schema) { - debug_assert_eq!(old.table_index, new.table_index); - debug_assert_eq!(old.table_name, new.table_name); - debug_assert_eq!(old.database_name, new.database_name); - - // Check missing columns - if old.columns.len() != new.columns.len() { - let missing = old - .columns - .iter() - .filter( - |ColumnDefinition { - name: column_name, .. - }| { - !new.columns.iter().any(|cd| cd.name.eq(column_name)) - }, - ) - .collect::>(); - if missing.len() == 1 { - Err(BreakingSchemaChange::ColumnDroppedOrRenamed { - column_name: missing[0].to_string(), - table_name: old.to_string(), - })? - } else { - Err(BreakingSchemaChange::MultipleColumnsDroppedOrRenamed { - table_name: old.to_string(), - columns: missing.iter().map(|cd| cd.to_string()).collect::>(), - })? - } - } - - // Check data type change - for old_column in old.columns.iter() { - let new_column = new - .columns - .iter() - .find(|cd| cd.name == old_column.name) - .unwrap(); - - if old_column.typ != new_column.typ { - Err(BreakingSchemaChange::ColumnDataTypeChanged { - table_name: old.to_string(), - column_name: old_column.to_string(), - old_data_type: old_column.typ, - new_column_name: new_column.typ, - })? - } - } - - // TODO: check nullable and primary key change - - // Checks passed; update schema - *old = new; - } - - Ok(()) - } - - const MARIADB_JSON_CHECK: &'static str = "(column_type = 'longtext' - AND ( - SELECT COUNT(*) > 0 - FROM information_schema.check_constraints - WHERE ( - constraint_schema = table_schema - AND table_name = table_name - AND constraint_name = column_name - AND check_clause LIKE CONCAT('%json_valid(`', column_name, '`)%') - ) - ) - ) as is_mariadb_json"; - - async fn query_information_schema_columns<'a, 'b, T>( - conn: &'b mut Conn, - table_infos: &'a [T], - select_columns: &[&str], - additional_select_expressions: &[&str], - ) -> Result - where - TableInfoRef<'a>: From<&'a T>, - { - let mut query_params: Vec = Vec::new(); - let query = format!( - " - SELECT {} - FROM information_schema.columns - WHERE {} - ORDER BY {}, ordinal_position ASC - ", - { - let mut select = select_columns - .iter() - .copied() - .map(escape_identifier) - .collect::>() - .join(", "); - - if !additional_select_expressions.is_empty() { - select.push_str(", "); - select.push_str(additional_select_expressions.join(", ").as_str()); - } - - select - }, - // where clause: filter by table name and table schema and column names - SqlHelper::select_columns_filter_predicate(table_infos, &mut query_params), - // order by clause: preserve the order of the input tables in the output - { - let table_index_expression = - SqlHelper::table_index_case_expression(table_infos, &mut query_params); - let order_by = format!("{} ASC", table_index_expression); - order_by - } - ); - - let rows = conn.exec_iter(query, query_params); - - Ok(rows) - } - - async fn connect(&self) -> Result { - Conn::new(self.conn_pool.clone()) - .await - .map_err(|err| MySQLConnectorError::ConnectionFailure(self.conn_url.clone(), err)) - } -} - -struct SqlHelper; - -impl SqlHelper { - fn table_index_case_expression<'a, T>(tables: &'a [T], query_params: &mut Vec) -> String - where - TableInfoRef<'a>: From<&'a T>, - { - let mut case = String::from("CASE "); - - tables.iter().map(Into::into).enumerate().for_each( - |(i, TableInfoRef { schema, name, .. })| { - case.push_str("WHEN (table_name = ? AND table_schema = "); - query_params.push(name.into()); - - if let Some(schema) = schema { - query_params.push(schema.into()); - case.push('?'); - } else { - case.push_str("DATABASE()"); - } - case.push(')'); - case.push_str(" THEN ? "); - query_params.push(i.into()); - }, - ); - - case.push_str("END"); - case - } - - fn column_index_case_expression<'a, T>(tables: &'a [T], query_params: &mut Vec) -> String - where - TableInfoRef<'a>: From<&'a T>, - { - let mut case = String::from("CASE "); - - tables.iter().map(Into::into).for_each( - |TableInfoRef { - column_names, - schema, - name: table_name, - }| { - column_names - .iter() - .enumerate() - .for_each(|(i, &column_name)| { - case.push_str("WHEN (table_name = ? AND table_schema = "); - query_params.push(table_name.into()); - - if let Some(schema) = schema { - query_params.push(schema.into()); - case.push('?'); - } else { - case.push_str("DATABASE()"); - } - case.push_str(" AND column_name = ?"); - query_params.push(column_name.into()); - case.push(')'); - - case.push_str(" THEN ? "); - query_params.push(i.into()); - }) - }, - ); - - case.push_str("END"); - case - } - - fn select_columns_filter_predicate<'a, T>( - tables: &'a [T], - query_params: &mut Vec, - ) -> String - where - TableInfoRef<'a>: From<&'a T>, - { - let predicate = tables - .iter() - .map(Into::into) - .map( - |TableInfoRef { - schema, - name, - column_names, - }| { - query_params.push(name.into()); - let mut condition = "(table_name = ? AND table_schema = ".to_string(); - if let Some(schema) = schema { - query_params.push(schema.into()); - condition.push('?'); - } else { - condition.push_str("DATABASE()"); - } - if !column_names.is_empty() { - condition.push_str(" AND column_name IN ("); - for (i, column) in column_names.iter().enumerate() { - query_params.push(column.into()); - if i == 0 { - condition.push('?'); - } else { - condition.push_str(", ?"); - } - } - condition.push(')'); - } - condition.push(')'); - condition - }, - ) - .collect::>() - .join(" OR "); - predicate - } -} - -pub struct TableInfoRef<'a> { - pub schema: Option<&'a str>, - pub name: &'a str, - pub column_names: Vec<&'a String>, -} - -impl<'a> From<&'a TableIdentifier> for TableInfoRef<'a> { - fn from(value: &'a TableIdentifier) -> Self { - Self { - schema: value.schema.as_deref(), - name: &value.name, - column_names: vec![], - } - } -} - -impl<'a> From<&'a TableInfo> for TableInfoRef<'a> { - fn from(value: &'a TableInfo) -> Self { - Self { - schema: value.schema.as_deref(), - name: &value.name, - column_names: value.column_names.iter().collect(), - } - } -} - -impl<'a> From<&'a TableDefinition> for TableInfoRef<'a> { - fn from(value: &'a TableDefinition) -> Self { - Self { - schema: Some(value.database_name.as_str()), - name: &value.table_name, - column_names: value.columns.iter().map(|cd| &cd.name).collect(), - } - } -} - -impl<'a> From<&'a &mut TableDefinition> for TableInfoRef<'a> { - fn from(value: &'a &mut TableDefinition) -> Self { - Self { - schema: Some(value.database_name.as_str()), - name: &value.table_name, - column_names: value.columns.iter().map(|cd| &cd.name).collect(), - } - } -} - -#[cfg(test)] -mod tests { - use super::{ColumnDefinition, SchemaHelper, TableDefinition}; - use crate::tests::{create_test_table, mariadb_test_config, mysql_test_config, TestConfig}; - use dozer_ingestion_connector::{ - dozer_types::types::FieldType, tokio, TableIdentifier, TableInfo, - }; - use serial_test::serial; - - async fn test_connector_schemas(config: TestConfig) { - // setup - let url = &config.url; - let pool = &config.pool; - - let schema_helper = SchemaHelper::new(url, pool); - - let _ = create_test_table("test1", &config).await; - - // test - let tables = schema_helper.list_tables().await.unwrap(); - let expected_table = TableIdentifier { - name: "test1".into(), - schema: Some("test".into()), - }; - assert!( - tables.contains(&expected_table), - "Missing test table. Existing tables list is {tables:?}" - ); - - let columns = schema_helper - .list_columns(vec![expected_table]) - .await - .unwrap(); - assert_eq!( - columns, - vec![TableInfo { - schema: Some("test".into()), - name: "test1".into(), - column_names: vec!["c1".into(), "c2".into(), "c3".into()] - }] - ); - - let schemas = schema_helper.get_table_definitions(&columns).await.unwrap(); - assert_eq!( - schemas, - vec![TableDefinition { - table_index: 0, - table_name: "test1".into(), - database_name: "test".into(), - columns: vec![ - ColumnDefinition { - ordinal_position: 1, - name: "c1".into(), - typ: FieldType::Int, - nullable: false, - primary_key: true, - }, - ColumnDefinition { - ordinal_position: 2, - name: "c2".into(), - typ: FieldType::Text, - nullable: true, - primary_key: false, - }, - ColumnDefinition { - ordinal_position: 3, - name: "c3".into(), - typ: FieldType::Float, - nullable: true, - primary_key: false, - }, - ] - }] - ); - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_schemas_mysql() { - test_connector_schemas(mysql_test_config()).await; - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_schemas_mariadb() { - test_connector_schemas(mariadb_test_config()).await; - } -} diff --git a/dozer-ingestion/mysql/src/state.rs b/dozer-ingestion/mysql/src/state.rs deleted file mode 100644 index d9798b89f3..0000000000 --- a/dozer-ingestion/mysql/src/state.rs +++ /dev/null @@ -1,44 +0,0 @@ -use dozer_ingestion_connector::dozer_types::node::OpIdentifier; - -use crate::binlog::BinlogPosition; -use crate::MysqlStateError; - -pub fn encode_state(pos: &BinlogPosition) -> OpIdentifier { - let lsn = (pos.binlog_id << 32) | pos.position; - - OpIdentifier { - txid: lsn, - seq_in_tx: 0, - } -} - -impl TryFrom for BinlogPosition { - type Error = MysqlStateError; - - fn try_from(state: OpIdentifier) -> Result { - let binlog_id = state.txid >> 32; - let position = state.txid & 0x00000000ffffffff; - - Ok(BinlogPosition { - binlog_id, - position, - }) - } -} - -#[cfg(test)] -mod tests { - #[test] - fn test_decode_encode() { - use super::*; - let pos = BinlogPosition { - binlog_id: 123, - position: 456, - }; - - let state = encode_state(&pos); - let pos2 = BinlogPosition::try_from(state).unwrap(); - - assert_eq!(pos, pos2); - } -} diff --git a/dozer-ingestion/mysql/src/tests.rs b/dozer-ingestion/mysql/src/tests.rs deleted file mode 100644 index 8b5155167d..0000000000 --- a/dozer-ingestion/mysql/src/tests.rs +++ /dev/null @@ -1,81 +0,0 @@ -use dozer_ingestion_connector::TableInfo; -use mysql_async::{prelude::Queryable, Opts, Pool}; - -pub struct TestConfig { - pub url: String, - pub opts: Opts, - pub pool: Pool, -} - -impl TestConfig { - pub fn new(url: String) -> Self { - let opts = Opts::from_url(url.as_str()).unwrap(); - let pool = Pool::new(opts.clone()); - Self { url, opts, pool } - } -} - -pub fn mysql_test_config() -> TestConfig { - TestConfig::new("mysql://root:mysql@localhost:3306/test".into()) -} - -pub fn mariadb_test_config() -> TestConfig { - TestConfig::new("mysql://root:mariadb@localhost:3307/test".into()) -} - -pub async fn create_test_table(name: &str, config: &TestConfig) -> TableInfo { - let TestTable { - create_table_sql, - table_info, - .. - } = test_tables().into_iter().find(|t| t.name == name).unwrap(); - - config - .pool - .get_conn() - .await - .unwrap() - .exec_drop(create_table_sql, ()) - .await - .unwrap(); - - table_info -} - -pub fn test_tables() -> Vec { - vec![ - TestTable { - name: "test1", - create_table_sql: "CREATE TABLE IF NOT EXISTS test1 (c1 INT NOT NULL, c2 VARCHAR(20), c3 DOUBLE, PRIMARY KEY (c1))", - table_info: TableInfo { - schema: Some("test".into()), - name: "test1".into(), - column_names: vec!["c1".into(), "c2".into(), "c3".into()], - } - }, - TestTable { - name: "test2", - create_table_sql: "CREATE TABLE IF NOT EXISTS test2 (id INT NOT NULL, value JSON, PRIMARY KEY (id))", - table_info: TableInfo { - schema: Some("test".into()), - name: "test2".into(), - column_names: vec!["id".into(), "value".into()], - } - }, - TestTable { - name: "test3", - create_table_sql: "CREATE TABLE IF NOT EXISTS test3 (a INT NOT NULL, b FLOAT, PRIMARY KEY (a))", - table_info: TableInfo { - schema: Some("test".into()), - name: "test3".into(), - column_names: vec!["a".into(), "b".into()], - } - }, - ] -} - -pub struct TestTable { - pub name: &'static str, - pub create_table_sql: &'static str, - pub table_info: TableInfo, -} diff --git a/dozer-ingestion/object-store/Cargo.toml b/dozer-ingestion/object-store/Cargo.toml deleted file mode 100644 index be8a8653e3..0000000000 --- a/dozer-ingestion/object-store/Cargo.toml +++ /dev/null @@ -1,13 +0,0 @@ -[package] -name = "dozer-ingestion-object-store" -version = "0.4.0" -edition = "2021" -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-ingestion-connector = { path = "../connector" } -object_store = { version = "0.9.0", features = ["aws"] } -url = "2.4.1" -datafusion = { version = "35.0.0" } diff --git a/dozer-ingestion/object-store/src/adapters.rs b/dozer-ingestion/object-store/src/adapters.rs deleted file mode 100644 index 17ea1d9e32..0000000000 --- a/dozer-ingestion/object-store/src/adapters.rs +++ /dev/null @@ -1,130 +0,0 @@ -use dozer_ingestion_connector::dozer_types::models::ingestion_types::{ - LocalStorage, S3Storage, Table, TableConfig, -}; -use object_store::aws::{AmazonS3, AmazonS3Builder}; -use object_store::local::LocalFileSystem; -use object_store::{BackoffConfig, ObjectStore, RetryConfig}; -use std::fmt::Debug; -use url::Url; - -use crate::{ObjectStoreConnectorError, ObjectStoreObjectError}; - -pub trait DozerObjectStore: Clone + Send + Sync + Debug + 'static { - type ObjectStore: ObjectStore; - - fn table_params( - &self, - table_name: &str, - ) -> Result, ObjectStoreConnectorError> { - let table = self - .tables() - .iter() - .find(|table| table.name == table_name) - .ok_or(ObjectStoreConnectorError::DataFusionStorageObjectError( - ObjectStoreObjectError::TableDefinitionNotFound, - ))?; - - self.store_params(table) - } - - fn store_params( - &self, - table: &Table, - ) -> Result, ObjectStoreConnectorError>; - - fn tables(&self) -> &[Table]; -} - -pub struct DozerObjectStoreParams { - pub url: Url, - pub object_store: T, - - pub table_path: String, - pub folder: String, - pub data_fusion_table: Table, - - // todo: refactor this datastructure - pub aws_region: Option, - pub aws_access_key_id: Option, - pub aws_secret_access_key: Option, -} - -impl DozerObjectStore for S3Storage { - type ObjectStore = AmazonS3; - - fn store_params( - &self, - table: &Table, - ) -> Result, ObjectStoreConnectorError> { - let details = &self.details; - - let retry_config = RetryConfig { - backoff: BackoffConfig::default(), - max_retries: usize::max_value(), - retry_timeout: std::time::Duration::from_secs(u64::MAX), - }; - - let object_store = AmazonS3Builder::new() - .with_bucket_name(&details.bucket_name) - .with_region(&details.region) - .with_access_key_id(&details.access_key_id) - .with_secret_access_key(&details.secret_access_key) - .with_retry(retry_config) - .build()?; - - let folder = match &table.config { - TableConfig::CSV(csv_config) => csv_config.path.clone(), - TableConfig::Parquet(parquet_config) => parquet_config.path.clone(), - }; - - Ok(DozerObjectStoreParams { - url: Url::parse(&format!("s3://{}", details.bucket_name)).expect("Must be valid url"), - object_store, - table_path: format!("s3://{}/{folder}/", details.bucket_name), - folder, - data_fusion_table: table.clone(), - - aws_region: Some(details.region.clone()), - aws_access_key_id: Some(details.access_key_id.clone()), - aws_secret_access_key: Some(details.secret_access_key.clone()), - }) - } - - fn tables(&self) -> &[Table] { - &self.tables - } -} - -impl DozerObjectStore for LocalStorage { - type ObjectStore = LocalFileSystem; - - fn store_params( - &self, - table: &Table, - ) -> Result, ObjectStoreConnectorError> { - let path = &self.details.path.as_str(); - - let object_store = LocalFileSystem::new_with_prefix(path)?; - - let folder = match &table.config { - TableConfig::CSV(csv_config) => csv_config.path.clone(), - TableConfig::Parquet(parquet_config) => parquet_config.path.clone(), - }; - - Ok(DozerObjectStoreParams { - url: Url::parse(&format!("local://{}", path)).expect("Must be valid url"), - object_store, - table_path: format!("{path}/{folder}/"), - folder, - data_fusion_table: table.clone(), - - aws_region: None, - aws_access_key_id: None, - aws_secret_access_key: None, - }) - } - - fn tables(&self) -> &[Table] { - &self.tables - } -} diff --git a/dozer-ingestion/object-store/src/connection.rs b/dozer-ingestion/object-store/src/connection.rs deleted file mode 100644 index fa199f2462..0000000000 --- a/dozer-ingestion/object-store/src/connection.rs +++ /dev/null @@ -1 +0,0 @@ -pub mod validator; diff --git a/dozer-ingestion/object-store/src/connection/validator.rs b/dozer-ingestion/object-store/src/connection/validator.rs deleted file mode 100644 index 34ff20767f..0000000000 --- a/dozer-ingestion/object-store/src/connection/validator.rs +++ /dev/null @@ -1,55 +0,0 @@ -use datafusion::datasource::listing::ListingTableUrl; -use dozer_ingestion_connector::{ - dozer_types::indicatif::{ProgressBar, ProgressStyle}, - TableIdentifier, -}; - -use crate::{adapters::DozerObjectStore, ObjectStoreConnectorError}; - -pub enum Validations { - Permissions, -} - -pub fn validate_connection( - name: &str, - tables: Option<&[TableIdentifier]>, - config: T, -) -> Result<(), ObjectStoreConnectorError> { - let validations_order: Vec = vec![Validations::Permissions]; - let pb = ProgressBar::new(validations_order.len() as u64); - pb.set_style( - ProgressStyle::with_template(&format!( - "[{}] {}", - name, "{spinner:.green} {wide_msg} {bar}" - )) - .unwrap(), - ); - pb.set_message("Validating connection to source"); - - for validation_type in validations_order { - match validation_type { - Validations::Permissions => validate_permissions(tables, config.clone())?, - } - - pb.inc(1); - } - - pb.finish_and_clear(); - - Ok(()) -} - -fn validate_permissions( - tables: Option<&[TableIdentifier]>, - config: T, -) -> Result<(), ObjectStoreConnectorError> { - if let Some(tables) = tables { - for table in tables.iter() { - let params = config.table_params(&table.name)?; - ListingTableUrl::parse(¶ms.table_path) - .map_err(ObjectStoreConnectorError::InternalDataFusionError)?; - } - } - - Ok(()) -} diff --git a/dozer-ingestion/object-store/src/connector.rs b/dozer-ingestion/object-store/src/connector.rs deleted file mode 100644 index 20d7e44601..0000000000 --- a/dozer-ingestion/object-store/src/connector.rs +++ /dev/null @@ -1,245 +0,0 @@ -use dozer_ingestion_connector::dozer_types::errors::internal::BoxedError; -use dozer_ingestion_connector::dozer_types::log::error; -use dozer_ingestion_connector::dozer_types::models::ingestion_types::{ - IngestionMessage, TransactionInfo, -}; -use dozer_ingestion_connector::dozer_types::node::OpIdentifier; -use dozer_ingestion_connector::dozer_types::types::FieldType; -use dozer_ingestion_connector::futures::future::try_join_all; -use dozer_ingestion_connector::tokio::sync::mpsc::channel; -use dozer_ingestion_connector::tokio::task::JoinSet; -use dozer_ingestion_connector::utils::{ListOrFilterColumns, TableNotFound}; -use dozer_ingestion_connector::{ - async_trait, tokio, Connector, Ingestor, SourceSchemaResult, TableIdentifier, TableInfo, -}; - -use crate::adapters::DozerObjectStore; -use crate::table::ObjectStoreTable; -use crate::{schema_mapper, ObjectStoreConnectorError}; - -use super::connection::validator::validate_connection; - -#[derive(Debug)] -pub struct ObjectStoreConnector { - config: T, -} - -impl ObjectStoreConnector { - pub fn new(config: T) -> Self { - Self { config } - } -} - -#[async_trait] -impl Connector for ObjectStoreConnector { - fn types_mapping() -> Vec<(String, Option)> - where - Self: Sized, - { - todo!() - } - - async fn validate_connection(&mut self) -> Result<(), BoxedError> { - validate_connection("object_store", None, self.config.clone()).map_err(Into::into) - } - - async fn list_tables(&mut self) -> Result, BoxedError> { - Ok(self - .config - .tables() - .iter() - .map(|table| TableIdentifier::from_table_name(table.name.clone())) - .collect()) - } - - async fn validate_tables(&mut self, tables: &[TableIdentifier]) -> Result<(), BoxedError> { - validate_connection("object_store", Some(tables), self.config.clone()).map_err(Into::into) - } - - async fn list_columns( - &mut self, - tables: Vec, - ) -> Result, BoxedError> { - let schemas = get_schema_from_tables(&self.config, &tables).await; - let mut result = vec![]; - for (table, schema) in tables.into_iter().zip(schemas) { - let schema = schema?; - let column_names = schema - .schema - .fields - .into_iter() - .map(|field| field.name) - .collect(); - let table_info = TableInfo { - schema: table.schema, - name: table.name, - column_names, - }; - result.push(table_info); - } - Ok(result) - } - - async fn get_schemas( - &mut self, - table_infos: &[TableInfo], - ) -> Result, BoxedError> { - let list_or_filter_columns = table_infos - .iter() - .map(|table_info| ListOrFilterColumns { - schema: table_info.schema.clone(), - name: table_info.name.clone(), - columns: Some(table_info.column_names.clone()), - }) - .collect::>(); - Ok(schema_mapper::get_schema(&self.config, &list_or_filter_columns).await) - } - - async fn serialize_state(&self) -> Result, BoxedError> { - Ok(vec![]) - } - - async fn start( - &mut self, - ingestor: &Ingestor, - tables: Vec, - last_checkpoint: Option, - ) -> Result<(), BoxedError> { - assert!(last_checkpoint.is_none()); - let (sender, mut receiver) = - channel::, ObjectStoreConnectorError>>(100); // todo: increase buffer siz - let ingestor_clone = ingestor.clone(); - - // Ingestor loop - generating operation message out - tokio::spawn(async move { - loop { - let message = receiver - .recv() - .await - .ok_or(ObjectStoreConnectorError::RecvError)?; - match message { - Ok(Some(evt)) => ingestor_clone.handle_message(evt).await?, - Ok(None) => { - break; - } - Err(ObjectStoreConnectorError::TableReaderError(e)) => error!("{e}"), - Err(e) => { - return Err(e.into()); - } - } - } - Ok::<_, BoxedError>(()) - }); - - // sender sending out message for pipeline - sender - .send(Ok(Some(IngestionMessage::TransactionInfo( - TransactionInfo::SnapshottingStarted, - )))) - .await - .unwrap(); - - let mut handles = vec![]; - - for (table_index, table_info) in tables.iter().enumerate() { - let table_info = TableInfo { - schema: table_info.schema.clone(), - name: table_info.name.clone(), - column_names: table_info.column_names.clone(), - }; - - let mut found = false; - for table_config in self.config.tables() { - if table_info.name == table_config.name { - let table = ObjectStoreTable::new( - table_config.config.clone(), - self.config.clone(), - Default::default(), - ); - let table_info = table_info.clone(); - let sender = sender.clone(); - handles.push(tokio::spawn(async move { - table - .snapshot(table_index, &table_info, sender) - .await - .unwrap() - })); - found = true; - break; - } - } - - if !found { - return Err(TableNotFound { - schema: table_info.schema, - name: table_info.name, - } - .into()); - } - } - - let updated_state = try_join_all(handles).await.unwrap(); - - sender - .send(Ok(Some(IngestionMessage::TransactionInfo( - TransactionInfo::SnapshottingDone { id: None }, - )))) - .await - .unwrap(); - sender - .send(Ok(Some(IngestionMessage::TransactionInfo( - TransactionInfo::Commit { - id: None, - source_time: None, - }, - )))) - .await - .unwrap(); - - let mut joinset = JoinSet::new(); - for (table_index, table_info) in tables.into_iter().enumerate() { - let table_info = TableInfo { - schema: table_info.schema, - name: table_info.name, - column_names: table_info.column_names, - }; - - for table in self.config.tables() { - if table_info.name == table.name { - let (state, schema) = updated_state[table_index].clone(); - let table = - ObjectStoreTable::new(table.config.clone(), self.config.clone(), state); - let sender = sender.clone(); - joinset.spawn(async move { - table - .watch(table_index, &table_info, sender, schema.as_ref()) - .await - }); - break; - } - } - } - while let Some(result) = joinset.join_next().await { - // Unwrap to propagate a panic in a task, then return - // short-circuit on any errors in connectors. The JoinSet - // will abort all other tasks when it is dropped - result.unwrap()?; - } - Ok(()) - } -} - -async fn get_schema_from_tables( - config: &impl DozerObjectStore, - tables: &[TableIdentifier], -) -> Vec { - let table_infos = tables - .iter() - .map(|table| ListOrFilterColumns { - schema: table.schema.clone(), - name: table.name.clone(), - columns: None, - }) - .collect::>(); - schema_mapper::get_schema(config, &table_infos).await -} diff --git a/dozer-ingestion/object-store/src/helper.rs b/dozer-ingestion/object-store/src/helper.rs deleted file mode 100644 index 69476f720f..0000000000 --- a/dozer-ingestion/object-store/src/helper.rs +++ /dev/null @@ -1,43 +0,0 @@ -use datafusion::datasource::{ - file_format::{csv::CsvFormat, parquet::ParquetFormat}, - listing::ListingOptions, -}; -use dozer_ingestion_connector::dozer_types::models::ingestion_types::{Table, TableConfig}; -use std::sync::Arc; - -use crate::{table_watcher::FileInfo, ObjectStoreObjectError}; - -pub fn map_listing_options( - data_fusion_table: &Table, -) -> Result { - match &data_fusion_table.config { - TableConfig::CSV(csv) => { - let format = CsvFormat::default(); - Ok(ListingOptions::new(Arc::new(format)).with_file_extension(csv.extension.clone())) - } - TableConfig::Parquet(parquet) => { - let format = ParquetFormat::new(); - Ok( - ListingOptions::new(Arc::new(format)) - .with_file_extension(parquet.extension.clone()), - ) - } - } -} - -pub fn is_marker_file_exist(marker_files: Vec, info: &FileInfo) -> bool { - for marker_file in marker_files { - let marker_file_name = match marker_file.name.rsplit_once('.') { - None => "", - Some(n) => n.0, - }; - let file_name = match info.name.rsplit_once('.') { - None => "", - Some(n) => n.0, - }; - if !file_name.is_empty() && marker_file_name == file_name { - return true; - } - } - false -} diff --git a/dozer-ingestion/object-store/src/lib.rs b/dozer-ingestion/object-store/src/lib.rs deleted file mode 100644 index 69acbf8a39..0000000000 --- a/dozer-ingestion/object-store/src/lib.rs +++ /dev/null @@ -1,95 +0,0 @@ -use datafusion::{datasource::listing::ListingTableUrl, error::DataFusionError}; -use dozer_ingestion_connector::dozer_types::{ - arrow_types::errors::FromArrowError, - thiserror::{self, Error}, -}; - -mod adapters; -mod connection; -pub mod connector; -mod helper; -mod schema_helper; -pub mod schema_mapper; -mod table; -mod table_reader; -pub(crate) mod table_watcher; -#[cfg(test)] -mod tests; - -#[derive(Error, Debug)] -pub enum ObjectStoreConnectorError { - #[error("object store error: {0}")] - ObjectStore(#[from] object_store::Error), - - #[error(transparent)] - DataFusionSchemaError(#[from] ObjectStoreSchemaError), - - #[error(transparent)] - DataFusionStorageObjectError(#[from] ObjectStoreObjectError), - - #[error("Internal data fusion error")] - InternalDataFusionError(#[source] DataFusionError), - - #[error(transparent)] - TableReaderError(#[from] ObjectStoreTableReaderError), - - #[error(transparent)] - FromArrowError(#[from] FromArrowError), - - #[error("Failed to send message on data read channel")] - SendError, - - #[error("Failed to receive message on data read channel")] - RecvError, -} - -#[derive(Error, Debug, PartialEq)] -pub enum ObjectStoreSchemaError { - #[error("Unsupported type of \"{0}\" field")] - FieldTypeNotSupported(String), - - #[error("Date time conversion failed")] - DateTimeConversionError, - - #[error("Date conversion failed")] - DateConversionError, - - #[error("Time conversion failed")] - TimeConversionError, - - #[error("Duration conversion failed")] - DurationConversionError, -} - -#[derive(Error, Debug)] -pub enum ObjectStoreObjectError { - #[error("Missing storage details")] - MissingStorageDetails, - - #[error("Table definition not found")] - TableDefinitionNotFound, - - #[error("Listing path {0} parsing error: {1}")] - ListingPathParsingError(String, #[source] DataFusionError), - - #[error("File format unsupported: {0}")] - FileFormatUnsupportedError(String), - - #[error("Listing path {0} error: {1}")] - ListingPathError(String, #[source] DataFusionError), -} - -#[derive(Error, Debug)] -pub enum ObjectStoreTableReaderError { - #[error("Table read failed: {0}")] - TableReadFailed(DataFusionError), - - #[error("Columns select failed: {0}")] - ColumnsSelectFailed(DataFusionError), - - #[error("Stream execution failed: {0}")] - StreamExecutionError(DataFusionError), - - #[error("File {0} has a conflicting schema")] - ConflictingSchema(ListingTableUrl), -} diff --git a/dozer-ingestion/object-store/src/readme.md b/dozer-ingestion/object-store/src/readme.md deleted file mode 100644 index d0caa4f982..0000000000 --- a/dozer-ingestion/object-store/src/readme.md +++ /dev/null @@ -1,36 +0,0 @@ -## Object store connector - -This connector uses local or cloud file system to ingest data, which are stored in files. -At the moment connector supports only append-only data changes. Also, current implementation only supports csv and parquet files stored locally or in s3 bucket. - -Depending on storage type configuration of connection is slightly different. -Example configuration: -```yaml -connections: - - db_type: ObjectStore - name: data_s3 - authentication: !S3Storage - details: - access_key_id: {{ ACCESS_ID }} - secret_access_key: {{ ACCESS_KEY }} - region: ap-southeast-1 - bucket_name: {{ BUCKET }} - tables: - - !Table - name: userdata - config: !Parquet - path: userdata_parquet - extension: .parquet #optional - - - db_type: ObjectStore - name: data_local - authentication: !LocalStorage - details: - path: "/Users/user/data" - tables: - - !Table - name: taxi_data - config: !CSV - path: taxi_data - extension: .csv -``` diff --git a/dozer-ingestion/object-store/src/schema_helper.rs b/dozer-ingestion/object-store/src/schema_helper.rs deleted file mode 100644 index 2f18df30cc..0000000000 --- a/dozer-ingestion/object-store/src/schema_helper.rs +++ /dev/null @@ -1,59 +0,0 @@ -use std::sync::Arc; - -use datafusion::arrow::datatypes::{DataType, Field}; -use dozer_ingestion_connector::dozer_types::types::{FieldDefinition, FieldType, SourceDefinition}; - -use crate::ObjectStoreSchemaError; - -pub fn map_schema_to_dozer<'a, I: Iterator>>( - fields_list: I, -) -> Result, ObjectStoreSchemaError> { - fields_list - .map(|field| { - let mapped_field_type = match field.data_type() { - DataType::Boolean => FieldType::Boolean, - DataType::Duration(_) - | DataType::Int8 - | DataType::Int16 - | DataType::Int32 - | DataType::Int64 => FieldType::Int, - DataType::UInt8 - | DataType::UInt16 - | DataType::UInt32 - | DataType::UInt64 - | DataType::Time32(_) - | DataType::Time64(_) => FieldType::UInt, - DataType::Float16 | DataType::Float32 | DataType::Float64 => FieldType::Float, - DataType::Timestamp(_, _) => FieldType::Timestamp, - DataType::Date32 | DataType::Date64 => FieldType::Date, - DataType::Binary | DataType::FixedSizeBinary(_) | DataType::LargeBinary => { - FieldType::Binary - } - DataType::Utf8 => FieldType::String, - DataType::LargeUtf8 => FieldType::Text, - // DataType::List(_) => {} - // DataType::FixedSizeList(_, _) => {} - // DataType::LargeList(_) => {} - // DataType::Struct(_) => {} - // DataType::Union(_, _, _) => {} - // DataType::Dictionary(_, _) => {} - // DataType::Decimal128(_, _) => {} - // DataType::Decimal256(_, _) => {} - // DataType::Map(_, _) => {} - _ => { - return Err(ObjectStoreSchemaError::FieldTypeNotSupported( - field.name().clone(), - )) - } - }; - - Ok(FieldDefinition { - name: field.name().clone(), - typ: mapped_field_type, - nullable: field.is_nullable(), - source: SourceDefinition::Dynamic, - description: None, - }) - }) - .collect() -} diff --git a/dozer-ingestion/object-store/src/schema_mapper.rs b/dozer-ingestion/object-store/src/schema_mapper.rs deleted file mode 100644 index 31f531e035..0000000000 --- a/dozer-ingestion/object-store/src/schema_mapper.rs +++ /dev/null @@ -1,101 +0,0 @@ -use datafusion::arrow::datatypes::SchemaRef; -use datafusion::datasource::file_format::csv::CsvFormat; -use datafusion::datasource::file_format::parquet::ParquetFormat; -use datafusion::datasource::listing::{ListingOptions, ListingTableUrl}; -use datafusion::prelude::SessionContext; -use dozer_ingestion_connector::dozer_types::log::error; -use dozer_ingestion_connector::dozer_types::models::ingestion_types::TableConfig; -use dozer_ingestion_connector::dozer_types::types::Schema; -use dozer_ingestion_connector::utils::ListOrFilterColumns; -use dozer_ingestion_connector::{CdcType, SourceSchema, SourceSchemaResult}; -use std::sync::Arc; - -use crate::adapters::DozerObjectStore; -use crate::schema_helper::map_schema_to_dozer; -use crate::{ObjectStoreConnectorError, ObjectStoreObjectError, ObjectStoreSchemaError}; - -pub fn map_schema( - resolved_schema: SchemaRef, - table: &ListOrFilterColumns, -) -> Result { - let fields_list = resolved_schema.fields().iter(); - - let fields = match &table.columns { - Some(columns) if !columns.is_empty() => { - let fields_list = fields_list.filter(|f| columns.iter().any(|c| c == f.name())); - - map_schema_to_dozer(fields_list) - } - _ => map_schema_to_dozer(fields_list), - }; - - Ok(Schema { - fields: fields?, - primary_index: vec![], - }) -} - -pub async fn get_schema( - config: &impl DozerObjectStore, - tables: &[ListOrFilterColumns], -) -> Vec { - let mut result = vec![]; - for table in tables.iter() { - result.push(get_table_schema(config, table).await); - } - result -} - -async fn get_table_schema( - config: &impl DozerObjectStore, - table: &ListOrFilterColumns, -) -> SourceSchemaResult { - let params = &config.table_params(&table.name)?; - - match ¶ms.data_fusion_table.config { - TableConfig::CSV(table_config) => { - let format = CsvFormat::default(); - let listing_options = ListingOptions::new(Arc::new(format)) - .with_file_extension(table_config.extension.clone()); - get_object_schema(table, config, listing_options).await - } - TableConfig::Parquet(table_config) => { - let format = ParquetFormat::default(); - let listing_options = ListingOptions::new(Arc::new(format)) - .with_file_extension(table_config.extension.clone()); - - get_object_schema(table, config, listing_options).await - } - } -} - -async fn get_object_schema( - table: &ListOrFilterColumns, - store_config: &impl DozerObjectStore, - listing_options: ListingOptions, -) -> SourceSchemaResult { - let params = store_config.table_params(&table.name)?; - - let table_path = ListingTableUrl::parse(¶ms.table_path).map_err(|e| { - ObjectStoreConnectorError::DataFusionStorageObjectError( - ObjectStoreObjectError::ListingPathParsingError(params.table_path.clone(), e), - ) - })?; - - let ctx = SessionContext::new(); - - ctx.runtime_env() - .register_object_store(¶ms.url, Arc::new(params.object_store)); - - let resolved_schema = listing_options - .infer_schema(&ctx.state(), &table_path) - .await - .map_err(|e| { - error!("{:?}", e); - ObjectStoreConnectorError::InternalDataFusionError(e) - })?; - - let schema = map_schema(resolved_schema, table)?; - - Ok(SourceSchema::new(schema, CdcType::Nothing)) -} diff --git a/dozer-ingestion/object-store/src/table.rs b/dozer-ingestion/object-store/src/table.rs deleted file mode 100644 index 202eca59ba..0000000000 --- a/dozer-ingestion/object-store/src/table.rs +++ /dev/null @@ -1,384 +0,0 @@ -use std::{collections::HashMap, sync::Arc, time::Duration}; - -use datafusion::{common::DFSchema, datasource::listing::ListingTableUrl, prelude::SessionContext}; -use dozer_ingestion_connector::{ - dozer_types::{ - chrono::{DateTime, Utc}, - log::info, - models::ingestion_types::{self, CsvConfig, IngestionMessage, ParquetConfig}, - }, - futures::StreamExt, - tokio::{self, sync::mpsc::Sender}, - TableInfo, -}; -use object_store::path::Path; -use object_store::ObjectStore; - -use crate::{ - adapters::DozerObjectStore, - helper::{is_marker_file_exist, map_listing_options}, - table_reader, - table_watcher::FileInfo, - ObjectStoreConnectorError, ObjectStoreObjectError, -}; - -pub trait TableConfig { - fn path(&self) -> &str; - fn extension(&self) -> &str; - fn marker_extension(&self) -> Option<&str>; -} - -pub struct ObjectStoreTable { - table_config: C, - store: O, - update_state: HashMap>, -} - -impl ObjectStoreTable { - pub fn new(table_config: C, store: O, update_state: HashMap>) -> Self { - Self { - table_config, - store, - update_state, - } - } - - pub async fn snapshot( - &self, - table_index: usize, - table_info: &TableInfo, - sender: Sender, ObjectStoreConnectorError>>, - ) -> Result<(HashMap>, Option), ObjectStoreConnectorError> { - let params = self.store.table_params(&table_info.name)?; - let store = Arc::new(params.object_store); - - let listing_options = map_listing_options(¶ms.data_fusion_table) - .map_err(ObjectStoreConnectorError::DataFusionStorageObjectError)?; - - let ctx = SessionContext::new(); - - ctx.runtime_env() - .register_object_store(¶ms.url, store.clone()); - - // Get the table state after snapshot - let mut update_state = self.update_state.clone(); - - // List objects in the S3 bucket with the specified prefix - let mut stream = store.list(Some(&Path::from(params.folder.clone()))); - - let mut new_files = vec![]; - let mut new_marker_files = vec![]; - - while let Some(item) = stream.next().await { - // Check if any objects have been added or modified - let object = item.unwrap(); - - if let Some(last_modified) = update_state.get_mut(&object.location) { - // Scenario 1: Update on existing file - if *last_modified < object.last_modified { - info!( - "Source Object has been modified: {:?}, {:?}", - object.location, object.last_modified - ); - } - } else { - // Scenario 2: New file added - info!( - "Source Object has been added: {:?}, {:?}", - object.location, object.last_modified - ); - - let file_path = object.location.to_string(); - // Skip the source folder - if file_path == params.folder { - continue; - } - - if file_path.ends_with(self.table_config.extension()) { - // Scenario 2: New file added - info!( - "Source Object has been added: {:?}, {:?}", - object.location, object.last_modified - ); - - // Remove base folder from relative path - let path = std::path::Path::new(&file_path); - let new_path = path - .strip_prefix(path.components().next().unwrap()) - .unwrap(); - let new_path_str = new_path.to_str().unwrap(); - - new_files.push(FileInfo { - name: params.table_path.clone() + new_path_str, - last_modified: object.last_modified.timestamp(), - }); - if self.table_config.marker_extension().is_none() { - update_state.insert(object.location, object.last_modified); - } - } else if let Some(marker_extension) = self.table_config.marker_extension() { - if file_path.ends_with(marker_extension) { - // Scenario 3: New marker file added - info!( - "Source Object Marker has been added: {:?}, {:?}", - object.location, object.last_modified - ); - - // Remove base folder from relative path - let path = std::path::Path::new(&file_path); - let new_path = path - .strip_prefix(path.components().next().unwrap()) - .unwrap(); - let new_path_str = new_path.to_str().unwrap(); - - new_marker_files.push(FileInfo { - name: params.table_path.clone() + new_path_str, - last_modified: object.last_modified.timestamp(), - }); - - update_state.insert(object.location, object.last_modified); - } else { - continue; - } - } else { - // Skip files that do not match the extension nor marker extension - continue; - } - } - } - - let mut schema = None; - new_files.sort(); - for file in &new_files { - let marker_file_exist = is_marker_file_exist(new_marker_files.clone(), file); - if !marker_file_exist && self.table_config.marker_extension().is_some() { - continue; - } else { - let file_path = ListingTableUrl::parse(&file.name) - .map_err(|e| { - ObjectStoreConnectorError::DataFusionStorageObjectError( - ObjectStoreObjectError::ListingPathParsingError(file.name.clone(), e), - ) - }) - .unwrap(); - - let result = table_reader::read( - table_index, - ctx.clone(), - file_path, - listing_options.clone(), - table_info, - sender.clone(), - schema.as_ref(), - ) - .await; - match result { - Ok(s) => { - schema = Some(s); - } - Err(e) => sender.send(Err(e)).await.unwrap(), - } - } - } - - Ok((update_state, schema)) - } - - pub async fn watch( - &self, - table_index: usize, - table_info: &TableInfo, - sender: Sender, ObjectStoreConnectorError>>, - schema: Option<&DFSchema>, - ) -> Result<(), ObjectStoreConnectorError> { - let params = self.store.table_params(&table_info.name)?; - let store = Arc::new(params.object_store); - - let source_folder = params.folder.to_string(); - let base_path = params.table_path; - - let listing_options = map_listing_options(¶ms.data_fusion_table) - .map_err(ObjectStoreConnectorError::DataFusionStorageObjectError)?; - - let ctx = SessionContext::new(); - - ctx.runtime_env() - .register_object_store(¶ms.url, store.clone()); - - // Get the table state after snapshot - let mut update_state = self.update_state.clone(); - - loop { - // List objects in the S3 bucket with the specified prefix - let mut stream = store.list(Some(&Path::from(source_folder.to_owned()))); - - let mut new_files = vec![]; - let mut new_marker_files = vec![]; - - while let Some(item) = stream.next().await { - // Check if any objects have been added or modified - let object = item.unwrap(); - - if let Some(last_modified) = update_state.get_mut(&object.location) { - // Scenario 1: Update on existing file - if *last_modified < object.last_modified { - info!( - "Source Object has been modified: {:?}, {:?}", - object.location, object.last_modified - ); - } - } else { - // Scenario 2: New file added - info!( - "Source Object has been added: {:?}, {:?}", - object.location, object.last_modified - ); - - let file_path = object.location.to_string(); - // Skip the source folder - if file_path == source_folder { - continue; - } - - if file_path.ends_with(self.table_config.extension()) { - // Scenario 2: New file added - info!( - "Source Object has been added: {:?}, {:?}", - object.location, object.last_modified - ); - - // Remove base folder from relative path - let path = std::path::Path::new(&file_path); - let new_path = path - .strip_prefix(path.components().next().unwrap()) - .unwrap(); - let new_path_str = new_path.to_str().unwrap(); - - new_files.push(FileInfo { - name: base_path.clone() + new_path_str, - last_modified: object.last_modified.timestamp(), - }); - if self.table_config.marker_extension().is_none() { - update_state.insert(object.location, object.last_modified); - } - } else if let Some(marker_extension) = self.table_config.marker_extension() { - if file_path.ends_with(marker_extension) { - // Scenario 3: New marker file added - info!( - "Source Object Marker has been added: {:?}, {:?}", - object.location, object.last_modified - ); - - // Remove base folder from relative path - let path = std::path::Path::new(&file_path); - let new_path = path - .strip_prefix(path.components().next().unwrap()) - .unwrap(); - let new_path_str = new_path.to_str().unwrap(); - - new_marker_files.push(FileInfo { - name: base_path.clone() + new_path_str, - last_modified: object.last_modified.timestamp(), - }); - update_state.insert(object.location, object.last_modified); - } else { - continue; - } - } else { - // Skip files that do not match the extension nor marker extension - continue; - } - } - } - - new_files.sort(); - for file in &new_files { - let marker_file_exist = is_marker_file_exist(new_marker_files.clone(), file); - if !marker_file_exist && self.table_config.marker_extension().is_some() { - continue; - } else { - let file_path = ListingTableUrl::parse(&file.name) - .map_err(|e| { - ObjectStoreConnectorError::DataFusionStorageObjectError( - ObjectStoreObjectError::ListingPathParsingError( - file.name.clone(), - e, - ), - ) - }) - .unwrap(); - - let result = table_reader::read( - table_index, - ctx.clone(), - file_path, - listing_options.clone(), - table_info, - sender.clone(), - schema, - ) - .await; - if let Err(e) = result { - sender.send(Err(e)).await.unwrap(); - } - } - } - - // Wait for 10 seconds before checking again - const WATCHER_INTERVAL: Duration = Duration::from_secs(1); - tokio::time::sleep(WATCHER_INTERVAL).await; - } - } -} - -impl TableConfig for CsvConfig { - fn path(&self) -> &str { - &self.path - } - - fn extension(&self) -> &str { - &self.extension - } - - fn marker_extension(&self) -> Option<&str> { - self.marker_extension.as_deref() - } -} - -impl TableConfig for ParquetConfig { - fn path(&self) -> &str { - &self.path - } - - fn extension(&self) -> &str { - &self.extension - } - - fn marker_extension(&self) -> Option<&str> { - self.marker_extension.as_deref() - } -} - -impl TableConfig for ingestion_types::TableConfig { - fn path(&self) -> &str { - match self { - ingestion_types::TableConfig::CSV(csv_config) => csv_config.path(), - ingestion_types::TableConfig::Parquet(parquet_config) => parquet_config.path(), - } - } - - fn extension(&self) -> &str { - match self { - ingestion_types::TableConfig::CSV(csv_config) => csv_config.extension(), - ingestion_types::TableConfig::Parquet(parquet_config) => parquet_config.extension(), - } - } - - fn marker_extension(&self) -> Option<&str> { - match self { - ingestion_types::TableConfig::CSV(csv_config) => csv_config.marker_extension(), - ingestion_types::TableConfig::Parquet(parquet_config) => { - parquet_config.marker_extension() - } - } - } -} diff --git a/dozer-ingestion/object-store/src/table_reader.rs b/dozer-ingestion/object-store/src/table_reader.rs deleted file mode 100644 index 4cd990cfd5..0000000000 --- a/dozer-ingestion/object-store/src/table_reader.rs +++ /dev/null @@ -1,130 +0,0 @@ -use datafusion::common::DFSchema; -use datafusion::datasource::listing::{ - ListingOptions, ListingTable, ListingTableConfig, ListingTableUrl, -}; -use datafusion::prelude::SessionContext; -use dozer_ingestion_connector::dozer_types::arrow_types::from_arrow::{ - map_schema_to_dozer, map_value_to_dozer_field, -}; -use dozer_ingestion_connector::dozer_types::log::error; -use dozer_ingestion_connector::dozer_types::models::ingestion_types::IngestionMessage; -use dozer_ingestion_connector::dozer_types::types::{Operation, Record}; -use dozer_ingestion_connector::futures::StreamExt; -use dozer_ingestion_connector::tokio::sync::mpsc::Sender; -use dozer_ingestion_connector::{tokio, TableInfo}; -use std::sync::Arc; - -use crate::{ObjectStoreConnectorError, ObjectStoreTableReaderError}; - -pub async fn read( - table_index: usize, - ctx: SessionContext, - table_path: ListingTableUrl, - listing_options: ListingOptions, - table: &TableInfo, - sender: Sender, ObjectStoreConnectorError>>, - schema: Option<&DFSchema>, -) -> Result { - let resolved_schema = listing_options - .infer_schema(&ctx.state(), &table_path) - .await - .map_err(ObjectStoreConnectorError::InternalDataFusionError)?; - - let fields = resolved_schema.all_fields(); - - let config = ListingTableConfig::new(table_path.clone()) - .with_listing_options(listing_options) - .with_schema(resolved_schema.clone()); - - let provider = Arc::new( - ListingTable::try_new(config) - .map_err(ObjectStoreConnectorError::InternalDataFusionError)?, - ); - - let cols: Vec<&str> = if table.column_names.is_empty() { - fields.iter().map(|f| f.name().as_str()).collect() - } else { - table.column_names.iter().map(|c| c.as_str()).collect() - }; - let dataframe = ctx - .read_table(provider.clone()) - .map_err(|e| { - ObjectStoreConnectorError::TableReaderError( - ObjectStoreTableReaderError::TableReadFailed(e), - ) - })? - .select_columns(&cols) - .map_err(|e| { - ObjectStoreConnectorError::TableReaderError( - ObjectStoreTableReaderError::ColumnsSelectFailed(e), - ) - })?; - - let this_schema = dataframe.schema().to_owned(); - if let Some(schema) = schema { - if schema != &this_schema { - return Err(ObjectStoreConnectorError::TableReaderError( - ObjectStoreTableReaderError::ConflictingSchema(table_path), - )); - } - } - let data = dataframe.execute_stream().await.map_err(|e| { - ObjectStoreConnectorError::TableReaderError( - ObjectStoreTableReaderError::StreamExecutionError(e), - ) - })?; - - tokio::pin!(data); - - while let Some(batch) = data.next().await { - let batch = match batch { - Ok(batch) => batch, - Err(e) => { - error!("Error reading record batch from {table_path:?}: {e}"); - continue; - } - }; - - let batch_schema = batch.schema(); - let dozer_schema = map_schema_to_dozer(batch_schema.as_ref())?; - - for row in 0..batch.num_rows() { - let fields = batch - .columns() - .iter() - .enumerate() - .map(|(col, column)| { - map_value_to_dozer_field( - column, - row, - resolved_schema.field(col).name(), - &dozer_schema, - ) - }) - .collect::, _>>()?; - - let evt = Operation::Insert { - new: Record { - values: fields, - lifetime: None, - }, - }; - - if sender - .send(Ok(Some(IngestionMessage::OperationEvent { - table_index, - op: evt, - id: None, - }))) - .await - .is_err() - { - break; - } - } - } - - // sender.send(Ok(None)).await.unwrap(); - - Ok(this_schema.to_owned()) -} diff --git a/dozer-ingestion/object-store/src/table_watcher.rs b/dozer-ingestion/object-store/src/table_watcher.rs deleted file mode 100644 index a45472c230..0000000000 --- a/dozer-ingestion/object-store/src/table_watcher.rs +++ /dev/null @@ -1,57 +0,0 @@ -use std::collections::HashMap; - -use dozer_ingestion_connector::{ - async_trait, - dozer_types::{ - chrono::{DateTime, Utc}, - models::ingestion_types::IngestionMessage, - }, - tokio::{sync::mpsc::Sender, task::JoinHandle}, - TableInfo, -}; - -use crate::ObjectStoreConnectorError; - -#[derive(Debug, Eq, Clone)] -pub struct FileInfo { - pub name: String, - pub last_modified: i64, -} - -impl Ord for FileInfo { - fn cmp(&self, other: &Self) -> std::cmp::Ordering { - self.last_modified.cmp(&other.last_modified) - } -} - -impl PartialOrd for FileInfo { - fn partial_cmp(&self, other: &Self) -> Option { - Some(self.cmp(other)) - } -} - -impl PartialEq for FileInfo { - fn eq(&self, other: &Self) -> bool { - self.last_modified == other.last_modified - } -} - -#[async_trait] -pub trait TableWatcher { - async fn watch( - &self, - table_index: usize, - table: &TableInfo, - sender: Sender, ObjectStoreConnectorError>>, - ) -> Result<(), ObjectStoreConnectorError>; - - async fn snapshot( - &self, - table_index: usize, - table: &TableInfo, - sender: Sender, ObjectStoreConnectorError>>, - ) -> Result< - JoinHandle<(usize, HashMap>)>, - ObjectStoreConnectorError, - >; -} diff --git a/dozer-ingestion/object-store/src/tests/files/all_types_csv/new_sample.csv b/dozer-ingestion/object-store/src/tests/files/all_types_csv/new_sample.csv deleted file mode 100644 index 85656e55ba..0000000000 --- a/dozer-ingestion/object-store/src/tests/files/all_types_csv/new_sample.csv +++ /dev/null @@ -1,11 +0,0 @@ -id,item,name,quantity,profit,margin,unit_price,location,category,amount -11,"Eldon Base for stackable storage shelf, platinum",Muhammed MacIntyre,3,-213.25,38.94,35,Nunavut,Storage & Organization, -12,"1.7 Cubic Foot Compact ""Cube"" Office Refrigerators",Barry French,293,457.81,208.16,68.02,Nunavut,Appliances,0.58 -13,"Cardinal Slant-D Ring Binder, Heavy Gauge Vinyl",Barry French,293,46.71,8.69,2.99,Nunavut,Binders and Binder Accessories, -14,R380,Clay Rozendal,483,1198.97,195.99,3.99,Nunavut,Telephones and Communication, -15,Holmes HEPA Air Purifier,Carlos Soltero,515,30.94,21.78,5.94,Nunavut,Appliances, -16,G.E. Longer-Life Indoor Recessed Floodlight Bulbs,Carlos Soltero,515,4.43,6.64,4.95,Nunavut,Office Furnishings, -17,"Angle-D Binders with Locking Rings, Label Holders",Carl Jackson,613,-54.04,7.3,7.72,Nunavut,Binders and Binder Accessorie, -18,"SAFCO Mobile Desk Side File, Wire Frame",Carl Jackson,613,127.70,42.76,6.22,Nunavut,Storage & Organization, -19,"SAFCO Commercial Wire Shelving, Black",Monica Federle,643,-695.26,138.14,35,Nunavut,Storage & Organization, -20,Xerox 198,Dorothy Badders,678,-226.36,4.98,8.33,Nunavut,Paper, diff --git a/dozer-ingestion/object-store/src/tests/files/all_types_csv/sample.csv b/dozer-ingestion/object-store/src/tests/files/all_types_csv/sample.csv deleted file mode 100644 index 302d93a7d6..0000000000 --- a/dozer-ingestion/object-store/src/tests/files/all_types_csv/sample.csv +++ /dev/null @@ -1,11 +0,0 @@ -id,item,name,quantity,profit,margin,unit_price,location,category,amount -1,"Eldon Base for stackable storage shelf, platinum",Muhammed MacIntyre,3,-213.25,38.94,35,Nunavut,Storage & Organization, -2,"1.7 Cubic Foot Compact ""Cube"" Office Refrigerators",Barry French,293,457.81,208.16,68.02,Nunavut,Appliances,0.58 -3,"Cardinal Slant-D Ring Binder, Heavy Gauge Vinyl",Barry French,293,46.71,8.69,2.99,Nunavut,Binders and Binder Accessories, -4,R380,Clay Rozendal,483,1198.97,195.99,3.99,Nunavut,Telephones and Communication, -5,Holmes HEPA Air Purifier,Carlos Soltero,515,30.94,21.78,5.94,Nunavut,Appliances, -6,G.E. Longer-Life Indoor Recessed Floodlight Bulbs,Carlos Soltero,515,4.43,6.64,4.95,Nunavut,Office Furnishings, -7,"Angle-D Binders with Locking Rings, Label Holders",Carl Jackson,613,-54.04,7.3,7.72,Nunavut,Binders and Binder Accessorie, -8,"SAFCO Mobile Desk Side File, Wire Frame",Carl Jackson,613,127.70,42.76,6.22,Nunavut,Storage & Organization, -9,"SAFCO Commercial Wire Shelving, Black",Monica Federle,643,-695.26,138.14,35,Nunavut,Storage & Organization, -10,Xerox 198,Dorothy Badders,678,-226.36,4.98,8.33,Nunavut,Paper, diff --git a/dozer-ingestion/object-store/src/tests/files/all_types_parquet/alltypes_plain.parquet b/dozer-ingestion/object-store/src/tests/files/all_types_parquet/alltypes_plain.parquet deleted file mode 100644 index a63f5dca7c..0000000000 Binary files a/dozer-ingestion/object-store/src/tests/files/all_types_parquet/alltypes_plain.parquet and /dev/null differ diff --git a/dozer-ingestion/object-store/src/tests/files/marker_csv/marker_sample.csv b/dozer-ingestion/object-store/src/tests/files/marker_csv/marker_sample.csv deleted file mode 100644 index 302d93a7d6..0000000000 --- a/dozer-ingestion/object-store/src/tests/files/marker_csv/marker_sample.csv +++ /dev/null @@ -1,11 +0,0 @@ -id,item,name,quantity,profit,margin,unit_price,location,category,amount -1,"Eldon Base for stackable storage shelf, platinum",Muhammed MacIntyre,3,-213.25,38.94,35,Nunavut,Storage & Organization, -2,"1.7 Cubic Foot Compact ""Cube"" Office Refrigerators",Barry French,293,457.81,208.16,68.02,Nunavut,Appliances,0.58 -3,"Cardinal Slant-D Ring Binder, Heavy Gauge Vinyl",Barry French,293,46.71,8.69,2.99,Nunavut,Binders and Binder Accessories, -4,R380,Clay Rozendal,483,1198.97,195.99,3.99,Nunavut,Telephones and Communication, -5,Holmes HEPA Air Purifier,Carlos Soltero,515,30.94,21.78,5.94,Nunavut,Appliances, -6,G.E. Longer-Life Indoor Recessed Floodlight Bulbs,Carlos Soltero,515,4.43,6.64,4.95,Nunavut,Office Furnishings, -7,"Angle-D Binders with Locking Rings, Label Holders",Carl Jackson,613,-54.04,7.3,7.72,Nunavut,Binders and Binder Accessorie, -8,"SAFCO Mobile Desk Side File, Wire Frame",Carl Jackson,613,127.70,42.76,6.22,Nunavut,Storage & Organization, -9,"SAFCO Commercial Wire Shelving, Black",Monica Federle,643,-695.26,138.14,35,Nunavut,Storage & Organization, -10,Xerox 198,Dorothy Badders,678,-226.36,4.98,8.33,Nunavut,Paper, diff --git a/dozer-ingestion/object-store/src/tests/files/marker_csv/marker_sample.marker b/dozer-ingestion/object-store/src/tests/files/marker_csv/marker_sample.marker deleted file mode 100644 index 302d93a7d6..0000000000 --- a/dozer-ingestion/object-store/src/tests/files/marker_csv/marker_sample.marker +++ /dev/null @@ -1,11 +0,0 @@ -id,item,name,quantity,profit,margin,unit_price,location,category,amount -1,"Eldon Base for stackable storage shelf, platinum",Muhammed MacIntyre,3,-213.25,38.94,35,Nunavut,Storage & Organization, -2,"1.7 Cubic Foot Compact ""Cube"" Office Refrigerators",Barry French,293,457.81,208.16,68.02,Nunavut,Appliances,0.58 -3,"Cardinal Slant-D Ring Binder, Heavy Gauge Vinyl",Barry French,293,46.71,8.69,2.99,Nunavut,Binders and Binder Accessories, -4,R380,Clay Rozendal,483,1198.97,195.99,3.99,Nunavut,Telephones and Communication, -5,Holmes HEPA Air Purifier,Carlos Soltero,515,30.94,21.78,5.94,Nunavut,Appliances, -6,G.E. Longer-Life Indoor Recessed Floodlight Bulbs,Carlos Soltero,515,4.43,6.64,4.95,Nunavut,Office Furnishings, -7,"Angle-D Binders with Locking Rings, Label Holders",Carl Jackson,613,-54.04,7.3,7.72,Nunavut,Binders and Binder Accessorie, -8,"SAFCO Mobile Desk Side File, Wire Frame",Carl Jackson,613,127.70,42.76,6.22,Nunavut,Storage & Organization, -9,"SAFCO Commercial Wire Shelving, Black",Monica Federle,643,-695.26,138.14,35,Nunavut,Storage & Organization, -10,Xerox 198,Dorothy Badders,678,-226.36,4.98,8.33,Nunavut,Paper, diff --git a/dozer-ingestion/object-store/src/tests/files/marker_only_one_csv/maker_new_sample.csv b/dozer-ingestion/object-store/src/tests/files/marker_only_one_csv/maker_new_sample.csv deleted file mode 100644 index 85656e55ba..0000000000 --- a/dozer-ingestion/object-store/src/tests/files/marker_only_one_csv/maker_new_sample.csv +++ /dev/null @@ -1,11 +0,0 @@ -id,item,name,quantity,profit,margin,unit_price,location,category,amount -11,"Eldon Base for stackable storage shelf, platinum",Muhammed MacIntyre,3,-213.25,38.94,35,Nunavut,Storage & Organization, -12,"1.7 Cubic Foot Compact ""Cube"" Office Refrigerators",Barry French,293,457.81,208.16,68.02,Nunavut,Appliances,0.58 -13,"Cardinal Slant-D Ring Binder, Heavy Gauge Vinyl",Barry French,293,46.71,8.69,2.99,Nunavut,Binders and Binder Accessories, -14,R380,Clay Rozendal,483,1198.97,195.99,3.99,Nunavut,Telephones and Communication, -15,Holmes HEPA Air Purifier,Carlos Soltero,515,30.94,21.78,5.94,Nunavut,Appliances, -16,G.E. Longer-Life Indoor Recessed Floodlight Bulbs,Carlos Soltero,515,4.43,6.64,4.95,Nunavut,Office Furnishings, -17,"Angle-D Binders with Locking Rings, Label Holders",Carl Jackson,613,-54.04,7.3,7.72,Nunavut,Binders and Binder Accessorie, -18,"SAFCO Mobile Desk Side File, Wire Frame",Carl Jackson,613,127.70,42.76,6.22,Nunavut,Storage & Organization, -19,"SAFCO Commercial Wire Shelving, Black",Monica Federle,643,-695.26,138.14,35,Nunavut,Storage & Organization, -20,Xerox 198,Dorothy Badders,678,-226.36,4.98,8.33,Nunavut,Paper, diff --git a/dozer-ingestion/object-store/src/tests/files/marker_only_one_csv/marker_sample.csv b/dozer-ingestion/object-store/src/tests/files/marker_only_one_csv/marker_sample.csv deleted file mode 100644 index 302d93a7d6..0000000000 --- a/dozer-ingestion/object-store/src/tests/files/marker_only_one_csv/marker_sample.csv +++ /dev/null @@ -1,11 +0,0 @@ -id,item,name,quantity,profit,margin,unit_price,location,category,amount -1,"Eldon Base for stackable storage shelf, platinum",Muhammed MacIntyre,3,-213.25,38.94,35,Nunavut,Storage & Organization, -2,"1.7 Cubic Foot Compact ""Cube"" Office Refrigerators",Barry French,293,457.81,208.16,68.02,Nunavut,Appliances,0.58 -3,"Cardinal Slant-D Ring Binder, Heavy Gauge Vinyl",Barry French,293,46.71,8.69,2.99,Nunavut,Binders and Binder Accessories, -4,R380,Clay Rozendal,483,1198.97,195.99,3.99,Nunavut,Telephones and Communication, -5,Holmes HEPA Air Purifier,Carlos Soltero,515,30.94,21.78,5.94,Nunavut,Appliances, -6,G.E. Longer-Life Indoor Recessed Floodlight Bulbs,Carlos Soltero,515,4.43,6.64,4.95,Nunavut,Office Furnishings, -7,"Angle-D Binders with Locking Rings, Label Holders",Carl Jackson,613,-54.04,7.3,7.72,Nunavut,Binders and Binder Accessorie, -8,"SAFCO Mobile Desk Side File, Wire Frame",Carl Jackson,613,127.70,42.76,6.22,Nunavut,Storage & Organization, -9,"SAFCO Commercial Wire Shelving, Black",Monica Federle,643,-695.26,138.14,35,Nunavut,Storage & Organization, -10,Xerox 198,Dorothy Badders,678,-226.36,4.98,8.33,Nunavut,Paper, diff --git a/dozer-ingestion/object-store/src/tests/files/marker_only_one_csv/marker_sample.marker b/dozer-ingestion/object-store/src/tests/files/marker_only_one_csv/marker_sample.marker deleted file mode 100644 index 302d93a7d6..0000000000 --- a/dozer-ingestion/object-store/src/tests/files/marker_only_one_csv/marker_sample.marker +++ /dev/null @@ -1,11 +0,0 @@ -id,item,name,quantity,profit,margin,unit_price,location,category,amount -1,"Eldon Base for stackable storage shelf, platinum",Muhammed MacIntyre,3,-213.25,38.94,35,Nunavut,Storage & Organization, -2,"1.7 Cubic Foot Compact ""Cube"" Office Refrigerators",Barry French,293,457.81,208.16,68.02,Nunavut,Appliances,0.58 -3,"Cardinal Slant-D Ring Binder, Heavy Gauge Vinyl",Barry French,293,46.71,8.69,2.99,Nunavut,Binders and Binder Accessories, -4,R380,Clay Rozendal,483,1198.97,195.99,3.99,Nunavut,Telephones and Communication, -5,Holmes HEPA Air Purifier,Carlos Soltero,515,30.94,21.78,5.94,Nunavut,Appliances, -6,G.E. Longer-Life Indoor Recessed Floodlight Bulbs,Carlos Soltero,515,4.43,6.64,4.95,Nunavut,Office Furnishings, -7,"Angle-D Binders with Locking Rings, Label Holders",Carl Jackson,613,-54.04,7.3,7.72,Nunavut,Binders and Binder Accessorie, -8,"SAFCO Mobile Desk Side File, Wire Frame",Carl Jackson,613,127.70,42.76,6.22,Nunavut,Storage & Organization, -9,"SAFCO Commercial Wire Shelving, Black",Monica Federle,643,-695.26,138.14,35,Nunavut,Storage & Organization, -10,Xerox 198,Dorothy Badders,678,-226.36,4.98,8.33,Nunavut,Paper, diff --git a/dozer-ingestion/object-store/src/tests/files/marker_parquet/marker_plain.marker b/dozer-ingestion/object-store/src/tests/files/marker_parquet/marker_plain.marker deleted file mode 100644 index a63f5dca7c..0000000000 Binary files a/dozer-ingestion/object-store/src/tests/files/marker_parquet/marker_plain.marker and /dev/null differ diff --git a/dozer-ingestion/object-store/src/tests/files/marker_parquet/marker_plain.parquet b/dozer-ingestion/object-store/src/tests/files/marker_parquet/marker_plain.parquet deleted file mode 100644 index a63f5dca7c..0000000000 Binary files a/dozer-ingestion/object-store/src/tests/files/marker_parquet/marker_plain.parquet and /dev/null differ diff --git a/dozer-ingestion/object-store/src/tests/files/no_marker_csv/marker_sample.csv b/dozer-ingestion/object-store/src/tests/files/no_marker_csv/marker_sample.csv deleted file mode 100644 index 302d93a7d6..0000000000 --- a/dozer-ingestion/object-store/src/tests/files/no_marker_csv/marker_sample.csv +++ /dev/null @@ -1,11 +0,0 @@ -id,item,name,quantity,profit,margin,unit_price,location,category,amount -1,"Eldon Base for stackable storage shelf, platinum",Muhammed MacIntyre,3,-213.25,38.94,35,Nunavut,Storage & Organization, -2,"1.7 Cubic Foot Compact ""Cube"" Office Refrigerators",Barry French,293,457.81,208.16,68.02,Nunavut,Appliances,0.58 -3,"Cardinal Slant-D Ring Binder, Heavy Gauge Vinyl",Barry French,293,46.71,8.69,2.99,Nunavut,Binders and Binder Accessories, -4,R380,Clay Rozendal,483,1198.97,195.99,3.99,Nunavut,Telephones and Communication, -5,Holmes HEPA Air Purifier,Carlos Soltero,515,30.94,21.78,5.94,Nunavut,Appliances, -6,G.E. Longer-Life Indoor Recessed Floodlight Bulbs,Carlos Soltero,515,4.43,6.64,4.95,Nunavut,Office Furnishings, -7,"Angle-D Binders with Locking Rings, Label Holders",Carl Jackson,613,-54.04,7.3,7.72,Nunavut,Binders and Binder Accessorie, -8,"SAFCO Mobile Desk Side File, Wire Frame",Carl Jackson,613,127.70,42.76,6.22,Nunavut,Storage & Organization, -9,"SAFCO Commercial Wire Shelving, Black",Monica Federle,643,-695.26,138.14,35,Nunavut,Storage & Organization, -10,Xerox 198,Dorothy Badders,678,-226.36,4.98,8.33,Nunavut,Paper, diff --git a/dozer-ingestion/object-store/src/tests/files/no_marker_parquet/marker_plain.parquet b/dozer-ingestion/object-store/src/tests/files/no_marker_parquet/marker_plain.parquet deleted file mode 100644 index a63f5dca7c..0000000000 Binary files a/dozer-ingestion/object-store/src/tests/files/no_marker_parquet/marker_plain.parquet and /dev/null differ diff --git a/dozer-ingestion/object-store/src/tests/local_storage_tests.rs b/dozer-ingestion/object-store/src/tests/local_storage_tests.rs deleted file mode 100644 index 779ddafc6a..0000000000 --- a/dozer-ingestion/object-store/src/tests/local_storage_tests.rs +++ /dev/null @@ -1,372 +0,0 @@ -use dozer_ingestion_connector::{ - dozer_types::{ - models::ingestion_types::{IngestionMessage, TransactionInfo}, - types::{Field, FieldType, Operation}, - }, - test_util::create_runtime_and_spawn_connector_all_tables, - tokio, Connector, -}; - -use crate::{connector::ObjectStoreConnector, tests::test_utils::get_local_storage_config}; - -#[macro_export] -macro_rules! test_type_conversion { - ($a:expr,$b:expr,$c:pat) => { - let value = $a.get($b).unwrap(); - assert!(matches!(value, $c)); - }; -} - -#[tokio::test] -async fn test_get_schema_of_parquet() { - let local_storage = get_local_storage_config("parquet", ""); - - let mut connector = ObjectStoreConnector::new(local_storage); - let (_, schemas) = connector.list_all_schemas().await.unwrap(); - let schema = schemas.first().unwrap(); - - let fields = schema.schema.fields.clone(); - assert_eq!(fields.first().unwrap().typ, FieldType::Int); - assert_eq!(fields.get(1).unwrap().typ, FieldType::Boolean); - assert_eq!(fields.get(2).unwrap().typ, FieldType::Int); - assert_eq!(fields.get(3).unwrap().typ, FieldType::Int); - assert_eq!(fields.get(4).unwrap().typ, FieldType::Int); - assert_eq!(fields.get(5).unwrap().typ, FieldType::Int); - assert_eq!(fields.get(6).unwrap().typ, FieldType::Float); - assert_eq!(fields.get(7).unwrap().typ, FieldType::Float); - assert_eq!(fields.get(8).unwrap().typ, FieldType::Binary); - assert_eq!(fields.get(9).unwrap().typ, FieldType::Binary); - assert_eq!(fields.get(10).unwrap().typ, FieldType::Timestamp); -} - -#[tokio::test] -async fn test_get_schema_of_csv() { - let local_storage = get_local_storage_config("csv", ""); - - let mut connector = ObjectStoreConnector::new(local_storage); - let (_, schemas) = connector.list_all_schemas().await.unwrap(); - let schema = schemas.first().unwrap(); - - let fields = schema.schema.fields.clone(); - assert_eq!(fields.first().unwrap().typ, FieldType::Int); - assert_eq!(fields.get(1).unwrap().typ, FieldType::String); - assert_eq!(fields.get(2).unwrap().typ, FieldType::String); - assert_eq!(fields.get(3).unwrap().typ, FieldType::Int); - assert_eq!(fields.get(4).unwrap().typ, FieldType::Float); - assert_eq!(fields.get(5).unwrap().typ, FieldType::Float); - assert_eq!(fields.get(6).unwrap().typ, FieldType::Float); - assert_eq!(fields.get(7).unwrap().typ, FieldType::String); - assert_eq!(fields.get(8).unwrap().typ, FieldType::String); -} - -#[test] -fn test_read_parquet_file() { - let local_storage = get_local_storage_config("parquet", ""); - - let connector = ObjectStoreConnector::new(local_storage); - - let (mut iterator, _) = create_runtime_and_spawn_connector_all_tables(connector); - - let row = iterator.next(); - if let Some(IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingStarted)) = row { - } else { - panic!("Unexpected message"); - } - - let mut i = 1; - while i < 9 { - let row = iterator.next(); - if let Some(IngestionMessage::OperationEvent { - op: Operation::Insert { new }, - .. - }) = row - { - let values = new.values; - - test_type_conversion!(values, 0, Field::Int(_)); - test_type_conversion!(values, 1, Field::Boolean(_)); - test_type_conversion!(values, 2, Field::Int(_)); - test_type_conversion!(values, 3, Field::Int(_)); - test_type_conversion!(values, 4, Field::Int(_)); - test_type_conversion!(values, 5, Field::Int(_)); - test_type_conversion!(values, 6, Field::Float(_)); - test_type_conversion!(values, 7, Field::Float(_)); - test_type_conversion!(values, 8, Field::Binary(_)); - test_type_conversion!(values, 9, Field::Binary(_)); - test_type_conversion!(values, 10, Field::Timestamp(_)); - } else { - panic!("Unexpected message"); - } - - i += 1; - } - - let row = iterator.next(); - if let Some(IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingDone { .. })) = row { - } else { - panic!("Unexpected message"); - } -} - -#[test] -fn test_read_parquet_file_marker() { - let local_storage = get_local_storage_config("parquet", "marker"); - - let connector = ObjectStoreConnector::new(local_storage); - - let (mut iterator, _) = create_runtime_and_spawn_connector_all_tables(connector); - - let row = iterator.next(); - if let Some(IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingStarted)) = row { - } else { - panic!("Unexpected message"); - } - - let mut i = 1; - while i < 9 { - let row = iterator.next(); - if let Some(IngestionMessage::OperationEvent { - op: Operation::Insert { new }, - .. - }) = row - { - let values = new.values; - - test_type_conversion!(values, 0, Field::Int(_)); - test_type_conversion!(values, 1, Field::Boolean(_)); - test_type_conversion!(values, 2, Field::Int(_)); - test_type_conversion!(values, 3, Field::Int(_)); - test_type_conversion!(values, 4, Field::Int(_)); - test_type_conversion!(values, 5, Field::Int(_)); - test_type_conversion!(values, 6, Field::Float(_)); - test_type_conversion!(values, 7, Field::Float(_)); - test_type_conversion!(values, 8, Field::Binary(_)); - test_type_conversion!(values, 9, Field::Binary(_)); - test_type_conversion!(values, 10, Field::Timestamp(_)); - } else { - panic!("Unexpected message"); - } - - i += 1; - } - - let row = iterator.next(); - if let Some(IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingDone { .. })) = row { - } else { - panic!("Unexpected message"); - } -} - -#[test] -fn test_read_parquet_file_no_marker() { - let local_storage = get_local_storage_config("parquet", "no_marker"); - - let connector = ObjectStoreConnector::new(local_storage); - - let (mut iterator, _) = create_runtime_and_spawn_connector_all_tables(connector); - - let row = iterator.next(); - if let Some(IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingStarted)) = row { - } else { - panic!("Unexpected message"); - } - - let row = iterator.next(); - if let Some(IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingDone { .. })) = row { - } else { - panic!("Unexpected message"); - } -} - -#[test] -fn test_csv_read() { - let local_storage = get_local_storage_config("csv", ""); - - let connector = ObjectStoreConnector::new(local_storage); - - let (mut iterator, _) = create_runtime_and_spawn_connector_all_tables(connector); - - let row = iterator.next(); - if let Some(IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingStarted)) = row { - } else { - panic!("Unexpected message"); - } - - let mut i = 1; - while i < 21 { - // no. of row in the csv data - let row = iterator.next(); - if let Some(IngestionMessage::OperationEvent { - op: Operation::Insert { new }, - .. - }) = row - { - let values = new.values; - - test_type_conversion!(values, 0, Field::Int(_)); - test_type_conversion!(values, 1, Field::String(_)); - test_type_conversion!(values, 2, Field::String(_)); - test_type_conversion!(values, 3, Field::Int(_)); - test_type_conversion!(values, 4, Field::Float(_)); - test_type_conversion!(values, 5, Field::Float(_)); - test_type_conversion!(values, 6, Field::Float(_)); - test_type_conversion!(values, 7, Field::String(_)); - test_type_conversion!(values, 8, Field::String(_)); - - if let Field::Int(id) = values.first().unwrap() { - if *id == 2 || *id == 12 { - test_type_conversion!(values, 9, Field::Float(_)); - } else { - test_type_conversion!(values, 9, Field::Null); - } - } - } else { - panic!("Unexpected message"); - } - - i += 1; - } - - let row = iterator.next(); - if let Some(IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingDone { .. })) = row { - } else { - panic!("Unexpected message"); - } -} - -#[test] -fn test_csv_read_marker() { - let local_storage = get_local_storage_config("csv", "marker"); - - let connector = ObjectStoreConnector::new(local_storage); - - let (mut iterator, _) = create_runtime_and_spawn_connector_all_tables(connector); - - let row = iterator.next(); - if let Some(IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingStarted)) = row { - } else { - panic!("Unexpected message"); - } - - let mut i = 1; - while i < 11 { - // no. of row in the csv data - let row = iterator.next(); - if let Some(IngestionMessage::OperationEvent { - op: Operation::Insert { new }, - .. - }) = row - { - let values = new.values; - - test_type_conversion!(values, 0, Field::Int(_)); - test_type_conversion!(values, 1, Field::String(_)); - test_type_conversion!(values, 2, Field::String(_)); - test_type_conversion!(values, 3, Field::Int(_)); - test_type_conversion!(values, 4, Field::Float(_)); - test_type_conversion!(values, 5, Field::Float(_)); - test_type_conversion!(values, 6, Field::Float(_)); - test_type_conversion!(values, 7, Field::String(_)); - test_type_conversion!(values, 8, Field::String(_)); - - if let Field::Int(id) = values.first().unwrap() { - if *id == 2 || *id == 12 { - test_type_conversion!(values, 9, Field::Float(_)); - } else { - test_type_conversion!(values, 9, Field::Null); - } - } - } else { - panic!("Unexpected message"); - } - - i += 1; - } - - let row = iterator.next(); - if let Some(IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingDone { .. })) = row { - } else { - panic!("Unexpected message"); - } -} - -#[test] -fn test_csv_read_only_one_marker() { - let local_storage = get_local_storage_config("csv", "marker_only_one"); - - let connector = ObjectStoreConnector::new(local_storage); - - let (mut iterator, _) = create_runtime_and_spawn_connector_all_tables(connector); - - let row = iterator.next(); - if let Some(IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingStarted)) = row { - } else { - panic!("Unexpected message"); - } - - let mut i = 1; - while i < 11 { - // no. of row in the csv data - let row = iterator.next(); - if let Some(IngestionMessage::OperationEvent { - op: Operation::Insert { new }, - .. - }) = row - { - let values = new.values; - - test_type_conversion!(values, 0, Field::Int(_)); - test_type_conversion!(values, 1, Field::String(_)); - test_type_conversion!(values, 2, Field::String(_)); - test_type_conversion!(values, 3, Field::Int(_)); - test_type_conversion!(values, 4, Field::Float(_)); - test_type_conversion!(values, 5, Field::Float(_)); - test_type_conversion!(values, 6, Field::Float(_)); - test_type_conversion!(values, 7, Field::String(_)); - test_type_conversion!(values, 8, Field::String(_)); - - if let Field::Int(id) = values.first().unwrap() { - if *id == 2 || *id == 12 { - test_type_conversion!(values, 9, Field::Float(_)); - } else { - test_type_conversion!(values, 9, Field::Null); - } - } - } else { - panic!("Unexpected message"); - } - - i += 1; - } - - // No data to be snapshotted - - let row = iterator.next(); - if let Some(IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingDone { .. })) = row { - } else { - panic!("Unexpected message"); - } -} - -#[test] -fn test_csv_read_no_marker() { - let local_storage = get_local_storage_config("csv", "no_marker"); - - let connector = ObjectStoreConnector::new(local_storage); - - let (mut iterator, _) = create_runtime_and_spawn_connector_all_tables(connector); - - let row = iterator.next(); - if let Some(IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingStarted)) = row { - } else { - panic!("Unexpected message"); - } - - // No data to be snapshotted - - let row = iterator.next(); - if let Some(IngestionMessage::TransactionInfo(TransactionInfo::SnapshottingDone { .. })) = row { - } else { - panic!("Unexpected message"); - } -} diff --git a/dozer-ingestion/object-store/src/tests/mod.rs b/dozer-ingestion/object-store/src/tests/mod.rs deleted file mode 100644 index b7aa489d67..0000000000 --- a/dozer-ingestion/object-store/src/tests/mod.rs +++ /dev/null @@ -1,2 +0,0 @@ -mod local_storage_tests; -mod test_utils; diff --git a/dozer-ingestion/object-store/src/tests/test_utils.rs b/dozer-ingestion/object-store/src/tests/test_utils.rs deleted file mode 100644 index ec680b44c7..0000000000 --- a/dozer-ingestion/object-store/src/tests/test_utils.rs +++ /dev/null @@ -1,67 +0,0 @@ -use dozer_ingestion_connector::dozer_types::models::ingestion_types::{ - CsvConfig, LocalDetails, LocalStorage, ParquetConfig, Table, TableConfig, -}; -use std::path::PathBuf; - -pub fn get_local_storage_config(typ: &str, prefix: &str) -> LocalStorage { - let p = PathBuf::from("src/tests/files".to_string()); - match typ { - "parquet" => match prefix { - "" => LocalStorage { - details: LocalDetails { - path: p.to_str().unwrap().to_string(), - }, - tables: vec![Table { - config: TableConfig::Parquet(ParquetConfig { - extension: typ.to_string(), - path: format!("all_types_{typ}"), - marker_extension: None, - }), - name: format!("all_types_{typ}"), - }], - }, - &_ => LocalStorage { - details: LocalDetails { - path: p.to_str().unwrap().to_string(), - }, - tables: vec![Table { - config: TableConfig::Parquet(ParquetConfig { - extension: typ.to_string(), - path: format!("{prefix}_{typ}"), - marker_extension: Some(String::from(".marker")), - }), - name: format!("{prefix}_{typ}"), - }], - }, - }, - "csv" => match prefix { - "" => LocalStorage { - details: LocalDetails { - path: p.to_str().unwrap().to_string(), - }, - tables: vec![Table { - config: TableConfig::CSV(CsvConfig { - extension: typ.to_string(), - path: format!("all_types_{typ}"), - marker_extension: None, - }), - name: format!("all_types_{typ}"), - }], - }, - &_ => LocalStorage { - details: LocalDetails { - path: p.to_str().unwrap().to_string(), - }, - tables: vec![Table { - config: TableConfig::CSV(CsvConfig { - extension: typ.to_string(), - path: format!("{prefix}_{typ}"), - marker_extension: Some(String::from(".marker")), - }), - name: format!("{prefix}_{typ}"), - }], - }, - }, - other => panic!("Unsupported type: {}", other), - } -} diff --git a/dozer-ingestion/postgres/Cargo.toml b/dozer-ingestion/postgres/Cargo.toml deleted file mode 100644 index 22c8bdcfd2..0000000000 --- a/dozer-ingestion/postgres/Cargo.toml +++ /dev/null @@ -1,30 +0,0 @@ -[package] -name = "dozer-ingestion-postgres" -version = "0.4.0" -edition = "2021" -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-ingestion-connector = { path = "../connector" } -postgres-protocol = "0.6.4" -postgres-types = { version = "0.2.4", features = [ - "with-serde_json-1", - "with-uuid-1", -] } -tokio-postgres = { version = "0.7.7", features = [ - "with-chrono-0_4", - "with-geo-types-0_7", - "with-uuid-1", -] } -uuid = { version = "1.3.1", features = ["serde", "v4"] } -rustls = "0.22" -tokio-postgres-rustls = "0.11.1" -rustls-native-certs = "0.7.0" -regex = "1" -rand = "0.8.5" - -[dev-dependencies] -serial_test = "1.0.0" -tokio = { version = "1", features = ["rt", "macros"] } diff --git a/dozer-ingestion/postgres/src/connection.rs b/dozer-ingestion/postgres/src/connection.rs deleted file mode 100644 index 668f31a2e4..0000000000 --- a/dozer-ingestion/postgres/src/connection.rs +++ /dev/null @@ -1,4 +0,0 @@ -pub mod client; -pub mod helper; -mod tables_validator; -pub mod validator; diff --git a/dozer-ingestion/postgres/src/connection/client.rs b/dozer-ingestion/postgres/src/connection/client.rs deleted file mode 100644 index ec59b0952b..0000000000 --- a/dozer-ingestion/postgres/src/connection/client.rs +++ /dev/null @@ -1,264 +0,0 @@ -use std::pin::Pin; -use std::sync::Arc; -use std::task::{ready, Poll}; - -use dozer_ingestion_connector::{ - dozer_types::{self, bytes}, - futures::future::BoxFuture, - futures::stream::BoxStream, - futures::Stream, - retry_on_network_failure, - tokio::{self, sync::Mutex}, -}; -use tokio_postgres::types::ToSql; -use tokio_postgres::{Config, CopyBothDuplex, Row, SimpleQueryMessage, Statement, ToStatement}; - -use crate::connection::helper::is_network_failure; -use crate::PostgresConnectorError; - -use super::helper; - -#[derive(Debug)] -pub struct Client { - config: tokio_postgres::Config, - inner: tokio_postgres::Client, -} - -impl Client { - pub fn new(config: Config, client: tokio_postgres::Client) -> Self { - Self { - config, - inner: client, - } - } - - pub fn config(&self) -> &Config { - &self.config - } - - pub async fn prepare(&mut self, query: &str) -> Result { - retry_on_network_failure!( - "prepare", - self.inner.prepare(query).await, - is_network_failure, - self.reconnect().await? - ) - } - - pub async fn simple_query( - &mut self, - query: &str, - ) -> Result, tokio_postgres::Error> { - retry_on_network_failure!( - "simple_query", - self.inner.simple_query(query).await, - is_network_failure, - self.reconnect().await? - ) - } - - pub async fn query_one( - &mut self, - statement: &T, - params: &[&(dyn ToSql + Sync)], - ) -> Result - where - T: ?Sized + ToStatement, - { - retry_on_network_failure!( - "query_one", - self.inner.query_one(statement, params).await, - is_network_failure, - self.reconnect().await? - ) - } - - pub async fn query( - &mut self, - statement: &T, - params: &[&(dyn ToSql + Sync)], - ) -> Result, tokio_postgres::Error> - where - T: ?Sized + ToStatement, - { - retry_on_network_failure!( - "query", - self.inner.query(statement, params).await, - is_network_failure, - self.reconnect().await? - ) - } - - pub async fn query_raw( - &mut self, - query: String, - params: Vec, - ) -> Result>, tokio_postgres::Error> { - let client = Self::connect(self.config.clone()).await?; - let row_stream = RowStream::new(client, query, params).await?; - Ok(Box::pin(row_stream)) - } - - pub async fn copy_both_simple( - &mut self, - query: &str, - ) -> Result, tokio_postgres::Error> - where - T: bytes::Buf + 'static + Send, - { - retry_on_network_failure!( - "copy_both_simple", - self.inner.copy_both_simple(query).await, - is_network_failure, - self.reconnect().await? - ) - } - - pub async fn batch_execute(&mut self, query: &str) -> Result<(), tokio_postgres::Error> { - retry_on_network_failure!( - "batch_execute", - self.inner.batch_execute(query).await, - is_network_failure, - self.reconnect().await? - ) - } - - pub async fn reconnect(&mut self) -> Result<(), tokio_postgres::Error> { - let new_client = Self::connect(self.config.clone()).await?; - self.inner = new_client.inner; - Ok(()) - } - - pub async fn connect(config: Config) -> Result { - let client = match helper::connect(config).await { - Ok(client) => client, - Err(PostgresConnectorError::ConnectionFailure(err)) => return Err(err), - Err(err) => panic!("unexpected error {err}"), - }; - Ok(client) - } - - async fn query_raw_internal( - &mut self, - statement: Statement, - params: Vec, - ) -> Result { - retry_on_network_failure!( - "query_raw", - self.inner.query_raw(&statement, ¶ms).await, - is_network_failure, - self.reconnect().await? - ) - } -} - -pub struct RowStream { - client: Arc>, - query: String, - query_params: Vec, - cursor_position: u64, - inner: Pin>, - pending_resume: Option< - BoxFuture<'static, Result<(Client, tokio_postgres::RowStream), tokio_postgres::Error>>, - >, -} - -impl RowStream { - pub async fn new( - mut client: Client, - query: String, - params: Vec, - ) -> Result { - let statement = client.prepare(&query).await?; - let inner = client.query_raw_internal(statement, params.clone()).await?; - Ok(Self { - client: Arc::new(Mutex::new(client)), - query, - query_params: params, - cursor_position: 0, - inner: Box::pin(inner), - pending_resume: None, - }) - } - - fn resume( - &mut self, - ) -> BoxFuture<'static, Result<(Client, tokio_postgres::RowStream), tokio_postgres::Error>> - { - async fn resume_async( - client: Arc>, - query: String, - params: Vec, - offset: u64, - ) -> Result<(Client, tokio_postgres::RowStream), tokio_postgres::Error> { - let config = client.lock().await.config().clone(); - // reconnect - let mut client = Client::connect(config).await?; - // send query with offset - let statement = client.prepare(&add_query_offset(&query, offset)).await?; - let row_stream = client.query_raw_internal(statement, params).await?; - Ok((client, row_stream)) - } - - Box::pin(resume_async( - self.client.clone(), - self.query.clone(), - self.query_params.clone(), - self.cursor_position, - )) - } -} - -impl Stream for RowStream { - type Item = Result; - - fn poll_next( - self: std::pin::Pin<&mut Self>, - cx: &mut std::task::Context<'_>, - ) -> Poll> { - let this = self.get_mut(); - - loop { - if let Some(resume) = this.pending_resume.as_mut() { - match ready!(resume.as_mut().poll(cx)) { - Ok((client, inner)) => { - this.pending_resume = None; - this.client = Arc::new(Mutex::new(client)); - this.inner = Box::pin(inner); - } - Err(err) => return Poll::Ready(Some(Err(err))), - } - } - - match ready!(this.inner.as_mut().poll_next(cx)) { - Some(Ok(row)) => { - this.cursor_position += 1; - return Poll::Ready(Some(Ok(row))); - } - Some(Err(err)) => { - if is_network_failure(&err) { - this.pending_resume = Some(this.resume()); - continue; - } else { - return Poll::Ready(Some(Err(err))); - } - } - None => return Poll::Ready(None), - } - } - } -} - -fn add_query_offset(query: &str, offset: u64) -> String { - assert!(query - .trim_start() - .get(0..7) - .map(|s| s.to_uppercase() == "SELECT ") - .unwrap_or(false)); - - if offset == 0 { - query.into() - } else { - format!("{query} OFFSET {offset}") - } -} diff --git a/dozer-ingestion/postgres/src/connection/helper.rs b/dozer-ingestion/postgres/src/connection/helper.rs deleted file mode 100644 index 3531e32e42..0000000000 --- a/dozer-ingestion/postgres/src/connection/helper.rs +++ /dev/null @@ -1,183 +0,0 @@ -use crate::PostgresConnectorError; - -use super::client::Client; -use dozer_ingestion_connector::{ - dozer_types::{ - self, - log::{debug, error}, - models::connection::ConnectionConfig, - }, - retry_on_network_failure, tokio, -}; -use rustls::Error; -use rustls::{ - client::danger::{HandshakeSignatureValid, ServerCertVerified, ServerCertVerifier}, - SignatureScheme, -}; -use std::sync::Arc; -use tokio_postgres::config::SslMode; -use tokio_postgres::{Connection, NoTls, Socket}; - -pub fn map_connection_config( - auth_details: &ConnectionConfig, -) -> Result { - if let ConnectionConfig::Postgres(postgres) = auth_details { - let config_replenished = match postgres.replenish() { - Ok(conf) => conf, - Err(e) => return Err(PostgresConnectorError::WrongConnectionConfiguration(e)), - }; - let mut config = tokio_postgres::Config::new(); - config - .host(&config_replenished.host) - .port(config_replenished.port as u16) - .user(&config_replenished.user) - .dbname(&config_replenished.database) - .password(&config_replenished.password) - .ssl_mode(config_replenished.sslmode); - Ok(config) - } else { - panic!("Postgres config was expected") - } -} - -#[derive(Debug)] -pub struct AcceptAllVerifier {} - -impl ServerCertVerifier for AcceptAllVerifier { - fn verify_tls12_signature( - &self, - _message: &[u8], - _cert: &rustls::pki_types::CertificateDer<'_>, - _dss: &rustls::DigitallySignedStruct, - ) -> Result { - Ok(HandshakeSignatureValid::assertion()) - } - - fn verify_tls13_signature( - &self, - _message: &[u8], - _cert: &rustls::pki_types::CertificateDer<'_>, - _dss: &rustls::DigitallySignedStruct, - ) -> Result { - Ok(HandshakeSignatureValid::assertion()) - } - - fn supported_verify_schemes(&self) -> Vec { - vec![ - SignatureScheme::RSA_PKCS1_SHA1, - SignatureScheme::ECDSA_SHA1_Legacy, - SignatureScheme::RSA_PKCS1_SHA256, - SignatureScheme::ECDSA_NISTP256_SHA256, - SignatureScheme::RSA_PKCS1_SHA384, - SignatureScheme::ECDSA_NISTP384_SHA384, - SignatureScheme::RSA_PKCS1_SHA512, - SignatureScheme::ECDSA_NISTP521_SHA512, - SignatureScheme::RSA_PSS_SHA256, - SignatureScheme::RSA_PSS_SHA384, - SignatureScheme::RSA_PSS_SHA512, - SignatureScheme::ED25519, - SignatureScheme::ED448, - ] - } - - fn verify_server_cert( - &self, - _end_entity: &rustls::pki_types::CertificateDer<'_>, - _intermediates: &[rustls::pki_types::CertificateDer<'_>], - _server_name: &rustls::pki_types::ServerName<'_>, - _ocsp_response: &[u8], - _now: rustls::pki_types::UnixTime, - ) -> Result { - Ok(ServerCertVerified::assertion()) - } -} - -pub async fn connect(config: tokio_postgres::Config) -> Result { - let mut roots = rustls::RootCertStore::empty(); - for cert in - rustls_native_certs::load_native_certs().map_err(PostgresConnectorError::LoadNativeCerts)? - { - if let Err(e) = roots.add(cert) { - debug!("Failed to add certificate: {}", e); - } - } - let mut rustls_config = rustls::ClientConfig::builder() - .with_root_certificates(roots) - .with_no_client_auth(); - - match config.get_ssl_mode() { - SslMode::Disable => { - // tokio-postgres::Config::SslMode::Disable + NoTLS connection - let (client, connection) = connect_helper(config, NoTls) - .await - .map_err(PostgresConnectorError::ConnectionFailure)?; - tokio::spawn(async move { - if let Err(e) = connection.await { - error!("Postgres connection error: {}", e); - } - }); - Ok(client) - } - SslMode::Prefer => { - // tokio-postgres::Config::SslMode::Prefer + TLS connection with no verification - rustls_config - .dangerous() - .set_certificate_verifier(Arc::new(AcceptAllVerifier {})); - let (client, connection) = connect_helper( - config, - tokio_postgres_rustls::MakeRustlsConnect::new(rustls_config), - ) - .await - .map_err(PostgresConnectorError::ConnectionFailure)?; - tokio::spawn(async move { - if let Err(e) = connection.await { - error!("Postgres connection error: {}", e); - } - }); - Ok(client) - } - // SslMode::Allow => unimplemented!(), - SslMode::Require => { - // tokio-postgres::Config::SslMode::Require + TLS connection with verification - let (client, connection) = connect_helper( - config, - tokio_postgres_rustls::MakeRustlsConnect::new(rustls_config), - ) - .await - .map_err(PostgresConnectorError::ConnectionFailure)?; - tokio::spawn(async move { - if let Err(e) = connection.await { - error!("Postgres connection error: {}", e); - } - }); - Ok(client) - } - ssl_mode => Err(PostgresConnectorError::InvalidSslError(ssl_mode)), - } -} - -async fn connect_helper( - config: tokio_postgres::Config, - tls: T, -) -> Result<(Client, Connection), tokio_postgres::Error> -where - T: tokio_postgres::tls::MakeTlsConnect + Clone, -{ - retry_on_network_failure!( - "connect", - config.connect(tls.clone()).await, - is_network_failure - ) - .map(|(client, connection)| (Client::new(config, client), connection)) -} - -pub fn is_network_failure(err: &tokio_postgres::Error) -> bool { - let err_str = err.to_string(); - err_str.starts_with("error communicating with the server") - || err_str.starts_with("error performing TLS handshake") - || err_str.starts_with("connection closed") - || err_str.starts_with("error connecting to server") - || err_str.starts_with("timeout waiting for server") - || (err_str.starts_with("db error") - && err_str.contains("canceling statement due to statement timeout")) -} diff --git a/dozer-ingestion/postgres/src/connection/tables_validator.rs b/dozer-ingestion/postgres/src/connection/tables_validator.rs deleted file mode 100644 index 562a2c77ac..0000000000 --- a/dozer-ingestion/postgres/src/connection/tables_validator.rs +++ /dev/null @@ -1,195 +0,0 @@ -use std::collections::hash_map::Entry; -use std::collections::HashMap; - -use dozer_ingestion_connector::utils::ListOrFilterColumns; - -use crate::schema::helper::DEFAULT_SCHEMA_NAME; -use crate::{PostgresConnectorError, PostgresSchemaError}; - -use super::client::Client; - -pub struct TablesValidator<'a> { - tables: HashMap<(String, String), &'a ListOrFilterColumns>, - tables_identifiers: Vec, -} - -type PostgresTableIdentifier = (String, String); -type PostgresTablesColumns = HashMap>; -type PostgresTablesWithTypes = HashMap>; - -impl<'a> TablesValidator<'a> { - pub fn new(table_info: &'a [ListOrFilterColumns]) -> Self { - let mut tables = HashMap::new(); - let mut tables_identifiers = vec![]; - table_info.iter().for_each(|t| { - let schema = t - .schema - .as_ref() - .map_or(DEFAULT_SCHEMA_NAME.to_string(), |s| s.clone()); - tables.insert((schema.clone(), t.name.clone()), t); - tables_identifiers.push(format!("{schema}.{}", t.name.clone())) - }); - - Self { - tables, - tables_identifiers, - } - } - - async fn fetch_tables( - &self, - client: &mut Client, - ) -> Result>, PostgresConnectorError> { - let result = client - .query( - "SELECT table_schema, table_name, table_type \ - FROM information_schema.tables \ - WHERE CONCAT(table_schema, '.', table_name) = ANY($1)", - &[&self.tables_identifiers], - ) - .await - .map_err(PostgresConnectorError::InvalidQueryError)?; - - let mut tables = HashMap::new(); - for r in result.iter() { - let schema_name: String = r - .try_get(0) - .map_err(PostgresConnectorError::InvalidQueryError)?; - let table_name: String = r - .try_get(1) - .map_err(PostgresConnectorError::InvalidQueryError)?; - let table_type: Option = r - .try_get(2) - .map_err(PostgresConnectorError::InvalidQueryError)?; - - tables.insert((schema_name, table_name), table_type); - } - - Ok(tables) - } - - async fn fetch_columns( - &self, - client: &mut Client, - ) -> Result { - let tables_columns = client - .query( - "SELECT table_schema, table_name, column_name \ - FROM information_schema.columns \ - WHERE CONCAT(table_schema, '.', table_name) = ANY($1)", - &[&self.tables_identifiers], - ) - .await - .map_err(PostgresConnectorError::InvalidQueryError)?; - - let mut table_columns_map: HashMap> = HashMap::new(); - tables_columns.iter().for_each(|r| { - let schema_name: String = r.try_get(0).unwrap(); - let tbl_name: String = r.try_get(1).unwrap(); - let col_name: String = r.try_get(2).unwrap(); - - if let Entry::Vacant(e) = - table_columns_map.entry((schema_name.clone(), tbl_name.clone())) - { - let cols = vec![col_name]; - e.insert(cols); - } else { - let cols = table_columns_map.get_mut(&(schema_name, tbl_name)).unwrap(); - cols.push(col_name); - } - }); - - Ok(table_columns_map) - } - - async fn fetch_data( - &self, - client: &mut Client, - ) -> Result<(PostgresTablesWithTypes, PostgresTablesColumns), PostgresConnectorError> { - let tables = self.fetch_tables(client).await?; - let columns = self.fetch_columns(client).await?; - - Ok((tables, columns)) - } - - pub async fn validate(&self, client: &mut Client) -> Result<(), PostgresConnectorError> { - let (tables, tables_columns) = self.fetch_data(client).await?; - - let missing_columns = self.find_missing_columns(tables_columns)?; - if !missing_columns.is_empty() { - let error_columns: Vec = missing_columns - .iter() - .map(|(schema, table, column)| { - format!("{0} in {1}.{2} table", column, schema, table) - }) - .collect(); - - return Err(PostgresConnectorError::ColumnsNotFound( - error_columns.join(", "), - )); - } - - let missing_tables = self.find_missing_tables(tables)?; - if !missing_tables.is_empty() { - return Err(PostgresConnectorError::TablesNotFound(missing_tables)); - } - - Ok(()) - } - - fn find_missing_columns( - &self, - tables_columns: PostgresTablesColumns, - ) -> Result, PostgresConnectorError> { - let mut errors = vec![]; - for (key, columns) in tables_columns.iter() { - let table_info = self - .tables - .get(key) - .ok_or(PostgresConnectorError::TablesNotFound(vec![key.clone()]))?; - - if let Some(column_names) = table_info.columns.clone() { - for c in column_names { - if !columns.contains(&c) { - errors.push((key.0.clone(), key.1.clone(), c)); - } - } - } - } - - Ok(errors) - } - - fn find_missing_tables( - &self, - existing_tables: PostgresTablesWithTypes, - ) -> Result, PostgresConnectorError> { - let mut missing_tables = vec![]; - - for key in self.tables.keys() { - existing_tables.get(key).map_or_else( - || { - missing_tables.push(key.clone()); - Ok(()) - }, - |table_type| { - table_type - .as_ref() - .map_or(Err(PostgresSchemaError::TableTypeNotFound), |typ| { - if typ.clone() != *"BASE TABLE" { - Err(PostgresSchemaError::UnsupportedTableType( - typ.clone(), - format!("{}.{}", key.0, key.1), - )) - } else { - Ok(()) - } - }) - .map_err(PostgresConnectorError::PostgresSchemaError) - }, - )?; - } - - Ok(missing_tables) - } -} diff --git a/dozer-ingestion/postgres/src/connection/validator.rs b/dozer-ingestion/postgres/src/connection/validator.rs deleted file mode 100644 index af1416e828..0000000000 --- a/dozer-ingestion/postgres/src/connection/validator.rs +++ /dev/null @@ -1,668 +0,0 @@ -use crate::{connector::ReplicationSlotInfo, PostgresConnectorError}; - -use super::{client::Client, tables_validator::TablesValidator}; -use dozer_ingestion_connector::{ - dozer_types::indicatif::{ProgressBar, ProgressStyle}, - utils::ListOrFilterColumns, -}; -use postgres_types::PgLsn; -use regex::Regex; - -pub enum Validations { - Details, - User, - Tables, - WALLevel, - Slot, -} - -pub async fn validate_connection( - name: &str, - config: tokio_postgres::Config, - tables: Option<&Vec>, - replication_info: Option, -) -> Result<(), PostgresConnectorError> { - let validations_order: Vec = vec![ - Validations::Details, - Validations::User, - Validations::Tables, - Validations::WALLevel, - Validations::Slot, - ]; - let pb = ProgressBar::new(validations_order.len() as u64); - pb.set_style( - ProgressStyle::with_template(&format!( - "[{}] {}", - name, "{spinner:.green} {wide_msg} {bar}" - )) - .unwrap(), - ); - pb.set_message("Validating connection to source"); - - let mut client = super::helper::connect(config).await?; - - for validation_type in validations_order { - match validation_type { - Validations::Details => validate_details(&mut client).await?, - Validations::User => validate_user(&mut client).await?, - Validations::Tables => { - if let Some(tables_info) = &tables { - validate_tables(&mut client, tables_info).await?; - } - } - Validations::WALLevel => validate_wal_level(&mut client).await?, - Validations::Slot => { - if let Some(replication_details) = &replication_info { - validate_slot(&mut client, replication_details, tables).await?; - } else { - validate_limit_of_replications(&mut client).await?; - } - } - } - - pb.inc(1); - } - - pb.finish_and_clear(); - - Ok(()) -} - -async fn validate_details(client: &mut Client) -> Result<(), PostgresConnectorError> { - client - .simple_query("SELECT version()") - .await - .map_err(PostgresConnectorError::ConnectionFailure)?; - - Ok(()) -} - -async fn validate_user(client: &mut Client) -> Result<(), PostgresConnectorError> { - client - .query_one( - " - SELECT r.rolcanlogin AS can_login, r.rolreplication AS is_replication_role, - ARRAY(SELECT b.rolname - FROM pg_catalog.pg_auth_members m - JOIN pg_catalog.pg_roles b ON (m.roleid = b.oid) - WHERE m.member = r.oid - ) && - '{rds_superuser, rdsadmin, rdsrepladmin, rds_replication}'::name[] AS is_aws_replication_role - FROM pg_roles r - WHERE r.rolname = current_user - ", - &[], - ) - .await - .map_or(Err(PostgresConnectorError::ReplicationIsNotAvailableForUserError), |row| { - let can_login: bool = row.get("can_login"); - let is_replication_role: bool = row.get("is_replication_role"); - let is_aws_replication_role: bool = row.get("is_aws_replication_role"); - - if can_login && (is_replication_role || is_aws_replication_role) { - Ok(()) - } else { - Err(PostgresConnectorError::ReplicationIsNotAvailableForUserError) - } - }) -} - -async fn validate_wal_level(client: &mut Client) -> Result<(), PostgresConnectorError> { - let result = client - .query_one("SHOW wal_level", &[]) - .await - .map_err(|_e| PostgresConnectorError::WALLevelIsNotCorrect())?; - - let wal_level: Result = result.try_get(0); - wal_level.map_or_else( - |e| Err(PostgresConnectorError::InvalidQueryError(e)), - |level| { - if level == "logical" { - Ok(()) - } else { - Err(PostgresConnectorError::WALLevelIsNotCorrect()) - } - }, - ) -} - -fn validate_tables_names( - table_info: &Vec, -) -> Result<(), PostgresConnectorError> { - let table_regex = Regex::new(r"^([[:lower:]_][[:alnum:]_]*)$").unwrap(); - for t in table_info { - if !table_regex.is_match(&t.name) { - return Err(PostgresConnectorError::TableNameNotValid(t.name.clone())); - } - } - - Ok(()) -} - -fn validate_columns_names( - table_info: &Vec, -) -> Result<(), PostgresConnectorError> { - let column_name_regex = Regex::new(r"^([[:lower:]_][[:alnum:]_]*)$").unwrap(); - for t in table_info { - if let Some(columns) = &t.columns { - for column in columns { - if !column_name_regex.is_match(column) { - return Err(PostgresConnectorError::ColumnNameNotValid(column.clone())); - } - } - } - } - - Ok(()) -} - -async fn validate_tables( - client: &mut Client, - table_info: &Vec, -) -> Result<(), PostgresConnectorError> { - validate_tables_names(table_info)?; - validate_columns_names(table_info)?; - - let tables_validator = TablesValidator::new(table_info); - tables_validator.validate(client).await?; - - Ok(()) -} - -pub async fn validate_slot( - client: &mut Client, - replication_info: &ReplicationSlotInfo, - tables: Option<&Vec>, -) -> Result<(), PostgresConnectorError> { - let result = client - .query_one( - "SELECT active, confirmed_flush_lsn FROM pg_replication_slots WHERE slot_name = $1", - &[&replication_info.name], - ) - .await - .map_err(PostgresConnectorError::InvalidQueryError)?; - - let is_already_running: bool = result - .try_get(0) - .map_err(PostgresConnectorError::InvalidQueryError)?; - if is_already_running { - return Err(PostgresConnectorError::SlotIsInUseError( - replication_info.name.clone(), - )); - } - - let flush_lsn: PgLsn = result - .try_get(1) - .map_err(PostgresConnectorError::InvalidQueryError)?; - - if flush_lsn.gt(&replication_info.start_lsn) { - return Err(PostgresConnectorError::StartLsnIsBeforeLastFlushedLsnError( - flush_lsn.to_string(), - replication_info.start_lsn.to_string(), - )); - } - - if let Some(tables_list) = tables { - let result = client - .query( - "SELECT pc.relname FROM pg_publication pb - LEFT OUTER JOIN pg_publication_rel pbl on pb.oid = pbl.prpubid - LEFT OUTER JOIN pg_class pc on pc.oid = pbl.prrelid - WHERE pubname = $1", - &[&replication_info.name], - ) - .await - .map_err(|_e| { - PostgresConnectorError::SlotNotExistError(replication_info.name.clone()) - })?; - - let mut publication_tables: Vec = vec![]; - for row in result { - publication_tables.push(row.get(0)); - } - - for t in tables_list { - if !publication_tables.contains(&t.name) { - return Err(PostgresConnectorError::MissingTableInReplicationSlot( - t.name.clone(), - )); - } - } - } - - Ok(()) -} - -async fn validate_limit_of_replications(client: &mut Client) -> Result<(), PostgresConnectorError> { - let slots_limit_result = client - .query_one("SHOW max_replication_slots", &[]) - .await - .map_err(PostgresConnectorError::ConnectionFailure)?; - - let slots_limit_str: String = slots_limit_result - .try_get(0) - .map_err(PostgresConnectorError::InvalidQueryError)?; - let slots_limit: i64 = slots_limit_str.parse().unwrap(); - - let used_slots_result = client - .query_one("SELECT COUNT(*) FROM pg_replication_slots;", &[]) - .await - .map_err(PostgresConnectorError::ConnectionFailure)?; - - let used_slots: i64 = used_slots_result - .try_get(0) - .map_err(PostgresConnectorError::InvalidQueryError)?; - - if used_slots == slots_limit { - Err(PostgresConnectorError::NoAvailableSlotsError) - } else { - Ok(()) - } -} - -#[cfg(test)] -mod tests { - use crate::{ - connection::helper::{connect, map_connection_config}, - test_utils::load_test_connection_config, - tests::client::TestPostgresClient, - PostgresSchemaError, - }; - - use super::*; - - use dozer_ingestion_connector::tokio; - use postgres_types::PgLsn; - use rand::Rng; - use serial_test::serial; - use std::panic; - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_validation_connection_fail_to_connect() { - let config = load_test_connection_config().await; - let mut config = map_connection_config(&config).unwrap(); - config.dbname("not_existing"); - - let result = validate_connection("pg_test_conn", config, None, None).await; - - assert!(result.is_err()); - - match result { - Ok(_) => panic!("Validation should fail"), - Err(e) => { - assert!(matches!(e, PostgresConnectorError::ConnectionFailure(_))); - - if let PostgresConnectorError::ConnectionFailure(msg) = e { - assert_eq!( - msg.to_string(), - "db error: FATAL: database \"not_existing\" does not exist" - ); - } else { - panic!("Unexpected error occurred"); - } - } - } - } - - // #[test] - // #[ignore] - // #[serial] - // fn test_connector_validation_connection_user_not_have_permission_to_use_replication() { - // run_connector_test("postgres", |app_config| { - // let mut config = get_config(app_config); - // let mut client = postgres::Config::from(config.clone()) - // .connect(NoTls) - // .unwrap(); - // - // client - // .simple_query("DROP USER if exists dozer_test_without_permission") - // .expect("User delete failed"); - // - // client - // .simple_query("CREATE USER dozer_test_without_permission") - // .expect("User creation failed"); - // - // client - // .simple_query("ALTER ROLE dozer_test_without_permission WITH NOREPLICATION") - // .expect("Role update failed"); - // - // config.user("dozer_test_without_permission"); - // - // let result = validate_connection(config, None, None); - // - // assert!(result.is_err()); - // - // match result { - // Ok(_) => panic!("Validation should fail"), - // Err(e) => { - // assert!(matches!( - // e, - // PostgresConnectorError::ReplicationIsNotAvailableForUserError - // )); - // } - // } - // }); - // } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_validation_connection_requested_tables_not_exist() { - let config = load_test_connection_config().await; - let config = map_connection_config(&config).unwrap(); - let mut client = connect(config.clone()).await.unwrap(); - - client - .simple_query("DROP TABLE IF EXISTS not_existing") - .await - .expect("User creation failed"); - - let tables = vec![ListOrFilterColumns { - name: "not_existing".to_string(), - schema: Some("public".to_string()), - columns: None, - }]; - let result = validate_connection("pg_test_conn", config, Some(&tables), None).await; - - assert!(result.is_err()); - - match result { - Ok(_) => panic!("Validation should fail"), - Err(e) => { - assert!(matches!(e, PostgresConnectorError::TablesNotFound(_))); - - if let PostgresConnectorError::TablesNotFound(msg) = e { - assert_eq!( - msg, - vec![("public".to_string(), "not_existing".to_string())] - ); - } else { - panic!("Unexpected error occurred"); - } - } - } - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_validation_connection_requested_columns_not_exist() { - let config = load_test_connection_config().await; - let config = map_connection_config(&config).unwrap(); - let mut client = connect(config.clone()).await.unwrap(); - - client - .simple_query("CREATE TABLE IF NOT EXISTS existing(column_1 serial PRIMARY KEY, column_2 serial);") - .await - .expect("User creation failed"); - - let columns = vec![ - String::from("column_not_existing_1"), - String::from("column_not_existing_2"), - ]; - - let tables = vec![ListOrFilterColumns { - name: "existing".to_string(), - schema: Some("public".to_string()), - columns: Some(columns), - }]; - - let result = validate_connection("pg_test_conn", config, Some(&tables), None).await; - - assert!(result.is_err()); - - match result { - Ok(_) => panic!("Validation should fail"), - Err(e) => { - assert!(matches!(e, PostgresConnectorError::ColumnsNotFound(_))); - - if let PostgresConnectorError::ColumnsNotFound(msg) = e { - assert_eq!(msg, "column_not_existing_1 in public.existing table, column_not_existing_2 in public.existing table"); - } else { - panic!("Unexpected error occurred"); - } - } - } - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_validation_connection_replication_slot_not_exist() { - let config = load_test_connection_config().await; - let config = map_connection_config(&config).unwrap(); - - let new_slot = "not_existing_slot"; - let replication_info = ReplicationSlotInfo { - name: new_slot.to_string(), - start_lsn: PgLsn::from(0), - }; - - let result = - validate_connection("pg_test_conn", config, None, Some(replication_info)).await; - - assert!(result.is_err()); - - match result { - Ok(_) => panic!("Validation should fail"), - Err(e) => { - assert!(matches!(e, PostgresConnectorError::InvalidQueryError(_))); - } - } - } - - #[test] - #[ignore] - #[serial] - fn test_start_lsn_is_before_last_flush_lsn() { - // let config = get_config(); - // let mut client = postgres::Config::from(config.clone()) - // .connect(NoTls) - // .unwrap(); - // - // client - // .query( - // r#"SELECT pg_create_logical_replication_slot('existing_slot', 'pgoutput');"#, - // &[], - // ) - // .expect("User creation failed"); - // - // let replication_info = ReplicationSlotInfo { - // name: "existing_slot".to_string(), - // start_lsn: PgLsn::from(0), - // }; - // let result = validate_connection(config, None, Some(replication_info)); - // - // client - // .query(r#"SELECT pg_drop_replication_slot('existing_slot');"#, &[]) - // .expect("Slot drop failed"); - // - // assert!(result.is_err()); - // - // match result { - // Ok(_) => panic!("Validation should fail"), - // Err(PostgresConnectorError::StartLsnIsBeforeLastFlushedLsnError(_, _)) => {} - // Err(_) => panic!("Unexpected error occurred"), - // } - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_validation_connection_valid_number_of_replication_slots() { - let config = load_test_connection_config().await; - let config = map_connection_config(&config).unwrap(); - let mut client = connect(config.clone()).await.unwrap(); - - let slots_limit_result = client - .query_one("SHOW max_replication_slots", &[]) - .await - .unwrap(); - - let slots_limit_str: String = slots_limit_result.try_get(0).unwrap(); - let slots_limit: i64 = slots_limit_str.parse().unwrap(); - - let used_slots_result = client - .query_one("SELECT COUNT(*) FROM pg_replication_slots;", &[]) - .await - .unwrap(); - - let used_slots: i64 = used_slots_result.try_get(0).unwrap(); - - let range = used_slots..slots_limit - 1; - for n in range { - let slot_name = format!("slot_{n}"); - client - .query( - r#"SELECT pg_create_logical_replication_slot($1, 'pgoutput');"#, - &[&slot_name], - ) - .await - .unwrap(); - } - - // One replication slot is available - let result = validate_connection("pg_test_conn", config, None, None).await; - assert!(result.is_ok()); - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_validation_connection_not_any_replication_slot_availble() { - let config = load_test_connection_config().await; - let config = map_connection_config(&config).unwrap(); - let mut client = connect(config.clone()).await.unwrap(); - - let slots_limit_result = client - .query_one("SHOW max_replication_slots", &[]) - .await - .unwrap(); - - let slots_limit_str: String = slots_limit_result.try_get(0).unwrap(); - let slots_limit: i64 = slots_limit_str.parse().unwrap(); - - let used_slots_result = client - .query_one("SELECT COUNT(*) FROM pg_replication_slots;", &[]) - .await - .unwrap(); - - let used_slots: i64 = used_slots_result.try_get(0).unwrap(); - - let range = used_slots..slots_limit; - for n in range { - let slot_name = format!("slot_{n}"); - client - .query( - r#"SELECT pg_create_logical_replication_slot($1, 'pgoutput');"#, - &[&slot_name], - ) - .await - .unwrap(); - } - - let result = validate_connection("pg_test_conn", config, None, None).await; - - assert!(result.is_err()); - - match result { - Ok(_) => panic!("Validation should fail"), - Err(e) => { - assert!(matches!(e, PostgresConnectorError::NoAvailableSlotsError)); - } - } - } - - #[test] - fn test_connector_validate_tables_names_with_valid_tables_names() { - let tables_with_result = vec![ - ("test", true), - ("Test", false), - (";Drop table test", false), - ("test_with_underscore", true), - ]; - - for (table_name, expected_result) in tables_with_result { - let res = validate_tables_names(&vec![ListOrFilterColumns { - name: table_name.to_string(), - schema: Some("public".to_string()), - columns: None, - }]); - - assert_eq!(expected_result, res.is_ok()); - } - } - - #[test] - fn test_connector_validate_columns_names_with_valid_column_names() { - let columns_names_with_result = vec![ - ("test", true), - ("Test", false), - (";Drop table test", false), - ("test_with_underscore", true), - ]; - - for (column_name, expected_result) in columns_names_with_result { - let res = validate_columns_names(&vec![ListOrFilterColumns { - schema: Some("public".to_string()), - name: "column_test_table".to_string(), - columns: Some(vec![column_name.to_string()]), - }]); - - assert_eq!(expected_result, res.is_ok()); - } - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_return_error_on_view_in_table_validation() { - let config = load_test_connection_config().await; - let mut client = TestPostgresClient::new(&config).await; - - let mut rng = rand::thread_rng(); - - let schema = format!("schema_helper_test_{}", rng.gen::()); - let table_name = format!("products_test_{}", rng.gen::()); - let view_name = format!("products_view_test_{}", rng.gen::()); - - client.create_schema(&schema).await; - client.create_simple_table(&schema, &table_name).await; - client.create_view(&schema, &table_name, &view_name).await; - - let config = map_connection_config(&config).unwrap(); - let mut pg_client = connect(config).await.unwrap(); - - let result = validate_tables( - &mut pg_client, - &vec![ListOrFilterColumns { - name: table_name, - schema: Some(schema.clone()), - columns: None, - }], - ) - .await; - - assert!(result.is_ok()); - - let result = validate_tables( - &mut pg_client, - &vec![ListOrFilterColumns { - name: view_name, - schema: Some(schema), - columns: None, - }], - ) - .await; - - assert!(result.is_err()); - assert!(matches!( - result, - Err(PostgresConnectorError::PostgresSchemaError( - PostgresSchemaError::UnsupportedTableType(_, _) - )) - )); - } -} diff --git a/dozer-ingestion/postgres/src/connector.rs b/dozer-ingestion/postgres/src/connector.rs deleted file mode 100644 index 3ba69bc02e..0000000000 --- a/dozer-ingestion/postgres/src/connector.rs +++ /dev/null @@ -1,265 +0,0 @@ -use dozer_ingestion_connector::dozer_types::node::OpIdentifier; -use dozer_ingestion_connector::{ - async_trait, - dozer_types::{errors::internal::BoxedError, types::FieldType}, - utils::ListOrFilterColumns, - Connector, Ingestor, SourceSchemaResult, TableIdentifier, TableInfo, -}; -use postgres_types::PgLsn; -use rand::distributions::Alphanumeric; -use rand::Rng; -use tokio_postgres::config::ReplicationMode; -use tokio_postgres::Config; - -use crate::{ - connection::validator::validate_connection, - iterator::PostgresIterator, - schema::helper::{SchemaHelper, DEFAULT_SCHEMA_NAME}, - PostgresConnectorError, -}; - -use super::connection::client::Client; -use super::connection::helper; - -#[derive(Clone, Debug)] -pub struct PostgresConfig { - pub name: String, - pub config: Config, - pub schema: Option, - pub batch_size: usize, -} - -#[derive(Debug)] -pub struct PostgresConnector { - pub name: String, - pub slot_name: String, - replication_conn_config: Config, - conn_config: Config, - schema_helper: SchemaHelper, - pub schema: Option, - batch_size: usize, -} - -#[derive(Debug)] -pub struct ReplicationSlotInfo { - pub name: String, - pub start_lsn: PgLsn, -} - -pub const REPLICATION_SLOT_PREFIX: &str = "dozer_slot"; - -impl PostgresConnector { - pub fn new( - config: PostgresConfig, - state: Option>, - ) -> Result { - let mut replication_conn_config = config.config.clone(); - replication_conn_config.replication_mode(ReplicationMode::Logical); - - let helper = SchemaHelper::new(config.config.clone(), config.schema.clone()); - - // conn_str - replication_conn_config - // conn_str_plain- conn_config - - let slot_name = state - .map(String::from_utf8) - .transpose()? - .unwrap_or_else(|| get_slot_name(&config.name)); - - Ok(PostgresConnector { - name: config.name, - slot_name, - conn_config: config.config, - replication_conn_config, - schema_helper: helper, - schema: config.schema, - batch_size: config.batch_size, - }) - } -} - -#[async_trait] -impl Connector for PostgresConnector { - fn types_mapping() -> Vec<(String, Option)> - where - Self: Sized, - { - todo!() - } - - async fn validate_connection(&mut self) -> Result<(), BoxedError> { - validate_connection(&self.name, self.conn_config.clone(), None, None) - .await - .map_err(Into::into) - } - - async fn list_tables(&mut self) -> Result, BoxedError> { - Ok(self - .schema_helper - .get_tables(None) - .await? - .into_iter() - .map(|table| TableIdentifier::new(Some(table.schema), table.name)) - .collect()) - } - - async fn validate_tables(&mut self, tables: &[TableIdentifier]) -> Result<(), BoxedError> { - let tables = tables - .iter() - .map(|table| ListOrFilterColumns { - schema: table.schema.clone(), - name: table.name.clone(), - columns: None, - }) - .collect::>(); - validate_connection(&self.name, self.conn_config.clone(), Some(&tables), None) - .await - .map_err(Into::into) - } - - async fn list_columns( - &mut self, - tables: Vec, - ) -> Result, BoxedError> { - let table_infos = tables - .iter() - .map(|table| ListOrFilterColumns { - schema: table.schema.clone(), - name: table.name.clone(), - columns: None, - }) - .collect::>(); - Ok(self - .schema_helper - .get_tables(Some(&table_infos)) - .await? - .into_iter() - .map(|table| TableInfo { - schema: Some(table.schema), - name: table.name, - column_names: table.columns, - }) - .collect()) - } - - async fn get_schemas( - &mut self, - table_infos: &[TableInfo], - ) -> Result, BoxedError> { - let table_infos = table_infos - .iter() - .map(|table| ListOrFilterColumns { - schema: table.schema.clone(), - name: table.name.clone(), - columns: Some(table.column_names.clone()), - }) - .collect::>(); - Ok(self - .schema_helper - .get_schemas(&table_infos) - .await? - .into_iter() - .map(|schema_result| schema_result.map_err(Into::into)) - .collect()) - } - - async fn serialize_state(&self) -> Result, BoxedError> { - Ok(self.slot_name.as_bytes().to_vec()) - } - - async fn start( - &mut self, - ingestor: &Ingestor, - tables: Vec, - last_checkpoint: Option, - ) -> Result<(), BoxedError> { - let lsn = last_checkpoint.map(|checkpoint| checkpoint.txid.into()); - - if lsn.is_none() { - let client = helper::connect(self.replication_conn_config.clone()).await?; - let table_identifiers = tables - .iter() - .map(|table| TableIdentifier::new(table.schema.clone(), table.name.clone())) - .collect::>(); - create_publication(client, &self.name, Some(&table_identifiers)).await?; - } - - let tables = tables - .into_iter() - .map(|table| ListOrFilterColumns { - schema: table.schema, - name: table.name, - columns: Some(table.column_names), - }) - .collect::>(); - let iterator = PostgresIterator::new( - self.name.clone(), - get_publication_name(&self.name), - self.slot_name.clone(), - self.schema_helper.get_tables(Some(&tables)).await?, - self.replication_conn_config.clone(), - ingestor, - self.conn_config.clone(), - self.schema.clone(), - self.batch_size, - ); - iterator.start(lsn).await.map_err(Into::into) - } -} - -fn get_publication_name(conn_name: &str) -> String { - format!("dozer_publication_{}", conn_name) -} - -pub fn get_slot_name(conn_name: &str) -> String { - let rand_name_suffix: String = rand::thread_rng() - .sample_iter(&Alphanumeric) - .take(7) - .map(char::from) - .collect(); - - format!( - "{REPLICATION_SLOT_PREFIX}_{}_{}", - conn_name, - rand_name_suffix.to_lowercase() - ) -} - -pub async fn create_publication( - mut client: Client, - conn_name: &str, - table_identifiers: Option<&[TableIdentifier]>, -) -> Result<(), PostgresConnectorError> { - let publication_name = get_publication_name(conn_name); - let table_str: String = match table_identifiers { - None => "ALL TABLES".to_string(), - Some(table_identifiers) => { - let table_names = table_identifiers - .iter() - .map(|table_identifier| { - format!( - r#""{}"."{}""#, - table_identifier - .schema - .as_deref() - .unwrap_or(DEFAULT_SCHEMA_NAME), - table_identifier.name - ) - }) - .collect::>(); - format!("TABLE {}", table_names.join(" , ")) - } - }; - - client - .simple_query(format!("DROP PUBLICATION IF EXISTS {publication_name}").as_str()) - .await - .map_err(PostgresConnectorError::DropPublicationError)?; - - client - .simple_query(format!("CREATE PUBLICATION {publication_name} FOR {table_str}").as_str()) - .await - .map_err(PostgresConnectorError::CreatePublicationError)?; - - Ok(()) -} diff --git a/dozer-ingestion/postgres/src/helper.rs b/dozer-ingestion/postgres/src/helper.rs deleted file mode 100644 index 9d2fb3f4c5..0000000000 --- a/dozer-ingestion/postgres/src/helper.rs +++ /dev/null @@ -1,539 +0,0 @@ -use dozer_ingestion_connector::dozer_types::{ - bytes::Bytes, - chrono::{DateTime, FixedOffset, NaiveDate, NaiveDateTime, Offset, Utc}, - errors::types::TypeError, - geo::Point as GeoPoint, - json_types::{parse_json_slice, serde_json_to_json_value, JsonArray, JsonValue}, - ordered_float::OrderedFloat, - rust_decimal, serde_json, - types::*, -}; -use postgres_types::{FromSql, Type, WasNull}; -use rust_decimal::prelude::FromPrimitive; -use rust_decimal::Decimal; -use std::error::Error; -use std::num::ParseIntError; - -use dozer_ingestion_connector::dozer_types::chrono::{LocalResult, NaiveTime}; -use std::vec; -use tokio_postgres::{Column, Row}; -use uuid::Uuid; - -use crate::DateConversionError::{AmbiguousTimeResult, InvalidDate, InvalidTime}; -use crate::{ - xlog_mapper::TableColumn, DateConversionError, PostgresConnectorError, PostgresSchemaError, -}; - -/// This function converts any offset string (+03, +03:00 and etc) to FixedOffset -/// -fn parse_timezone_offset(offset_string: String) -> Result, ParseIntError> { - // Fill right side with zeros when offset is not full length - let offset_string = format!("{:0<9}", offset_string); - - let sign = &offset_string[0..1]; - let hour = offset_string[1..3].parse::()?; - let min = offset_string[4..6].parse::()?; - let sec = offset_string[7..9].parse::()?; - - let secs = (hour * 3600) + (min * 60) + sec; - - if sign == "-" { - Ok(FixedOffset::west_opt(secs)) - } else { - Ok(FixedOffset::east_opt(secs)) - } -} - -fn convert_date(date: String) -> Result { - // Fill right side with zeros when date time is not full - let date_string = format!("{:0<26}", date); - - let year: i32 = date_string[0..4].parse()?; - let month: u32 = date_string[5..7].parse()?; - let day: u32 = date_string[8..10].parse()?; - let hour: u32 = date_string[11..13].parse()?; - let minute: u32 = date_string[14..16].parse()?; - let second: u32 = date_string[17..19].parse()?; - let microseconds = if date_string.len() == 19 { - 0 - } else { - date_string[20..26].parse()? - }; - - let naive_date = NaiveDate::from_ymd_opt(year, month, day).ok_or(InvalidDate)?; - let naive_time = - NaiveTime::from_hms_micro_opt(hour, minute, second, microseconds).ok_or(InvalidTime)?; - - Ok(NaiveDateTime::new(naive_date, naive_time)) -} - -fn convert_date_with_timezone(date: String) -> Result, DateConversionError> { - // Find position of last + or -, which is the start of timezone offset - let pos_plus = date.rfind('+'); - let pos_min = date.rfind('-'); - - let pos = match (pos_plus, pos_min) { - (Some(plus), Some(min)) => { - if plus > min { - plus - } else { - min - } - } - (None, Some(pos)) | (Some(pos), None) => pos, - (None, None) => 0, - }; - - let (date, offset_string) = date.split_at(pos); - - let offset = parse_timezone_offset(offset_string.to_string())?.map_or(Utc.fix(), |x| x); - - match convert_date(date.to_string())?.and_local_timezone(offset) { - LocalResult::None => Err(InvalidTime), - LocalResult::Single(date) => Ok(date), - LocalResult::Ambiguous(_, _) => Err(AmbiguousTimeResult), - } -} - -pub fn postgres_type_to_field( - value: Option<&Bytes>, - column: &TableColumn, -) -> Result { - let column_type = column.r#type.clone(); - value.map_or(Ok(Field::Null), |v| match column_type { - Type::INT2 | Type::INT4 | Type::INT8 => Ok(Field::Int( - String::from_utf8(v.to_vec()).unwrap().parse().unwrap(), - )), - Type::FLOAT4 | Type::FLOAT8 => Ok(Field::Float(OrderedFloat( - String::from_utf8(v.to_vec()) - .unwrap() - .parse::() - .unwrap(), - ))), - Type::TEXT | Type::VARCHAR | Type::CHAR | Type::BPCHAR | Type::ANYENUM => { - Ok(Field::String(String::from_utf8(v.to_vec()).unwrap())) - } - Type::UUID => Ok(Field::String(String::from_utf8(v.to_vec()).unwrap())), - Type::BYTEA => Ok(Field::Binary(v.to_vec())), - Type::NUMERIC => Ok(Field::Decimal( - Decimal::from_f64( - String::from_utf8(v.to_vec()) - .unwrap() - .parse::() - .unwrap(), - ) - .unwrap(), - )), - Type::TIMESTAMP => { - let date_string = String::from_utf8(v.to_vec())?; - - Ok(Field::Timestamp(DateTime::from_naive_utc_and_offset( - convert_date(date_string)?, - Utc.fix(), - ))) - } - Type::TIMESTAMPTZ => { - let date_string = String::from_utf8(v.to_vec())?; - - Ok(convert_date_with_timezone(date_string).map(Field::Timestamp)?) - } - Type::DATE => { - let date: NaiveDate = NaiveDate::parse_from_str( - String::from_utf8(v.to_vec()).unwrap().as_str(), - DATE_FORMAT, - ) - .unwrap(); - Ok(Field::from(date)) - } - Type::JSONB | Type::JSON => { - let val: serde_json::Value = serde_json::from_slice(v).map_err(|_| { - PostgresSchemaError::JSONBParseError(format!( - "Error converting to a single row for: {}", - column_type.name() - )) - })?; - let json: JsonValue = serde_json_to_json_value(val) - .map_err(|e| PostgresSchemaError::TypeError(TypeError::DeserializationError(e)))?; - Ok(Field::Json(json)) - } - Type::JSONB_ARRAY | Type::JSON_ARRAY => { - let json_val = parse_json_slice(v).map_err(|_| { - PostgresSchemaError::JSONBParseError(format!( - "Error converting to a single row for: {}", - column_type.name() - )) - })?; - Ok(Field::Json(json_val)) - } - Type::BOOL => Ok(Field::Boolean(v.slice(0..1) == "t")), - Type::POINT => Ok(Field::Point( - String::from_utf8(v.to_vec()) - .map_err(PostgresSchemaError::StringParseError)? - .parse::() - .map_err(|_| PostgresSchemaError::PointParseError)?, - )), - _ => Err(PostgresSchemaError::ColumnTypeNotSupported( - column_type.name().to_string(), - )), - }) -} - -pub fn postgres_type_to_dozer_type(column_type: Type) -> Result { - match column_type { - Type::BOOL => Ok(FieldType::Boolean), - Type::INT2 | Type::INT4 | Type::INT8 => Ok(FieldType::Int), - Type::CHAR | Type::TEXT | Type::VARCHAR | Type::BPCHAR | Type::UUID | Type::ANYENUM => { - Ok(FieldType::String) - } - Type::FLOAT4 | Type::FLOAT8 => Ok(FieldType::Float), - Type::BYTEA => Ok(FieldType::Binary), - Type::TIMESTAMP | Type::TIMESTAMPTZ => Ok(FieldType::Timestamp), - Type::NUMERIC => Ok(FieldType::Decimal), - Type::JSONB - | Type::JSON - | Type::JSONB_ARRAY - | Type::JSON_ARRAY - | Type::TEXT_ARRAY - | Type::CHAR_ARRAY - | Type::VARCHAR_ARRAY - | Type::BPCHAR_ARRAY => Ok(FieldType::Json), - Type::DATE => Ok(FieldType::Date), - Type::POINT => Ok(FieldType::Point), - _ => Err(PostgresSchemaError::ColumnTypeNotSupported( - column_type.name().to_string(), - )), - } -} - -fn handle_error(e: tokio_postgres::error::Error) -> Result { - if let Some(e) = e.source() { - if let Some(_e) = e.downcast_ref::() { - Ok(Field::Null) - } else { - Err(PostgresSchemaError::ValueConversionError(e.to_string())) - } - } else { - Err(PostgresSchemaError::ValueConversionError(e.to_string())) - } -} - -macro_rules! conversion_fn { - ($name:ident, $typ:expr) => { - fn $name(row: &Row, idx: usize) -> Result { - row.try_get(idx).map($typ).or_else(handle_error) - } - }; -} - -conversion_fn!(convert_bool, Field::Boolean); -conversion_fn!(convert_string, Field::String); -conversion_fn!(convert_timestamp, |v: NaiveDateTime| Field::Timestamp( - v.and_utc().fixed_offset() -)); -conversion_fn!(convert_timestamptz, Field::Timestamp); -conversion_fn!(convert_date_snapshot, Field::Date); -conversion_fn!(convert_binary, Field::Binary); -conversion_fn!(convert_point, |v: GeoPoint| Field::Point(v.x_y().into())); -conversion_fn!(convert_decimal, Field::Decimal); - -#[inline(always)] -fn convert_int<'a, T: Into + FromSql<'a>>( - row: &'a Row, - idx: usize, -) -> Result { - row.try_get(idx) - .map(|v: T| Field::Int(v.into())) - .or_else(handle_error) -} - -fn convert_int2(row: &Row, idx: usize) -> Result { - convert_int::(row, idx) -} -fn convert_int4(row: &Row, idx: usize) -> Result { - convert_int::(row, idx) -} -fn convert_int8(row: &Row, idx: usize) -> Result { - convert_int::(row, idx) -} - -conversion_fn!(convert_float, |v: f32| Field::Float(OrderedFloat(v.into()))); -conversion_fn!(convert_double, |v| Field::Float(OrderedFloat(v))); - -fn convert_json(row: &Row, idx: usize) -> Result { - let value: Result = row.try_get(idx); - value.map_or_else(handle_error, |val| { - Ok(Field::Json(serde_json_to_json_value(val).map_err(|e| { - PostgresSchemaError::TypeError(TypeError::DeserializationError(e)) - })?)) - }) -} -fn convert_jsonarray(row: &Row, idx: usize) -> Result { - let value: Result, _> = row.try_get(idx); - value.map_or_else(handle_error, |val| { - let field = val - .into_iter() - .map(serde_json_to_json_value) - .collect::>() - .map_err(TypeError::DeserializationError)?; - Ok(Field::Json(field.into())) - }) -} -fn convert_textarray(row: &Row, idx: usize) -> Result { - let value: Result, _> = row.try_get(idx); - value.map_or_else(handle_error, |val| Ok(Field::Json(val.into()))) -} - -fn convert_uuid(row: &Row, idx: usize) -> Result { - let value: Result = row.try_get(idx); - value.map_or_else(handle_error, |val| Ok(Field::from(val.to_string()))) -} - -type ConversionFn = fn(&Row, usize) -> Result; - -pub fn get_conversion_fn(col_type: &Type) -> Result { - match col_type { - &Type::BOOL => Ok(convert_bool), - &Type::INT2 => Ok(convert_int2), - &Type::INT4 => Ok(convert_int4), - &Type::INT8 => Ok(convert_int8), - &Type::CHAR | &Type::TEXT | &Type::VARCHAR | &Type::BPCHAR | &Type::ANYENUM => { - Ok(convert_string) - } - &Type::FLOAT4 => Ok(convert_float), - &Type::FLOAT8 => Ok(convert_double), - &Type::TIMESTAMP => Ok(convert_timestamp), - &Type::TIMESTAMPTZ => Ok(convert_timestamptz), - &Type::NUMERIC => Ok(convert_decimal), - &Type::DATE => Ok(convert_date_snapshot), - &Type::BYTEA => Ok(convert_binary), - &Type::JSONB | &Type::JSON => Ok(convert_json), - &Type::JSONB_ARRAY | &Type::JSON_ARRAY => Ok(convert_jsonarray), - &Type::CHAR_ARRAY | &Type::TEXT_ARRAY | &Type::VARCHAR_ARRAY | &Type::BPCHAR_ARRAY => { - Ok(convert_textarray) - } - &Type::POINT => Ok(convert_point), - // &Type::UUID => convert_row_value_to_field!(row, idx, Uuid), - &Type::UUID => Ok(convert_uuid), - _ => { - if col_type.schema() == "pg_catalog" { - Err(PostgresSchemaError::ColumnTypeNotSupported( - col_type.name().to_string(), - )) - } else { - Err(PostgresSchemaError::CustomTypeNotSupported( - col_type.name().to_string(), - )) - } - } - } -} - -pub fn get_values( - row: &Row, - conversion: &[ConversionFn], -) -> Result, PostgresSchemaError> { - let mut values: Vec = Vec::with_capacity(conversion.len()); - for (idx, fun) in conversion.iter().enumerate() { - values.push(fun(row, idx)?) - } - Ok(values) -} - -pub fn map_row_to_record( - row: &Row, - conversion: &[ConversionFn], -) -> Result { - get_values(row, conversion).map(Record::new) -} - -pub fn map_schema(columns: &[Column]) -> Result { - let field_defs: Result, _> = - columns.iter().map(convert_column_to_field).collect(); - - Ok(Schema { - fields: field_defs.unwrap(), - primary_index: vec![0], - }) -} - -pub fn convert_column_to_field(column: &Column) -> Result { - postgres_type_to_dozer_type(column.type_().clone()).map(|typ| FieldDefinition { - name: column.name().to_string(), - typ, - nullable: true, - source: SourceDefinition::Dynamic, - description: None, - }) -} - -#[cfg(test)] -mod tests { - use super::*; - use dozer_ingestion_connector::dozer_types::{chrono::NaiveDate, json_types::json}; - - #[macro_export] - macro_rules! test_conversion { - ($a:expr,$b:expr,$c:expr) => { - let value = postgres_type_to_field( - Some(&Bytes::from($a)), - &TableColumn { - name: "column".to_string(), - flags: 0, - r#type: $b, - column_index: 0, - }, - ); - assert_eq!(value.unwrap(), $c); - }; - } - - #[macro_export] - macro_rules! test_type_mapping { - ($a:expr,$b:expr) => { - let value = postgres_type_to_dozer_type($a); - assert_eq!(value.unwrap(), $b); - }; - } - - #[test] - fn it_converts_postgres_type_to_field() { - test_conversion!("12", Type::INT8, Field::Int(12)); - test_conversion!("4.7809", Type::FLOAT8, Field::Float(OrderedFloat(4.7809))); - let value = String::from("Test text"); - test_conversion!("Test text", Type::TEXT, Field::String(value.clone())); - test_conversion!("Test text", Type::ANYENUM, Field::String(value)); - - let value = String::from("a0eebc99-9c0b-4ef8-bb6d-6bb9bd380a11"); - test_conversion!( - "a0eebc99-9c0b-4ef8-bb6d-6bb9bd380a11", - Type::UUID, - Field::String(value) - ); - - // UTF-8 bytes representation of json (https://www.charset.org/utf-8) - let value: Vec = vec![98, 121, 116, 101, 97]; - test_conversion!("bytea", Type::BYTEA, Field::Binary(value)); - - let value = rust_decimal::Decimal::from_f64(8.28).unwrap(); - test_conversion!("8.28", Type::NUMERIC, Field::Decimal(value)); - - let value = DateTime::from_naive_utc_and_offset( - NaiveDate::from_ymd_opt(2022, 9, 16) - .unwrap() - .and_hms_opt(5, 56, 29) - .unwrap(), - Utc.fix(), - ); - test_conversion!( - "2022-09-16 05:56:29", - Type::TIMESTAMP, - Field::Timestamp(value) - ); - - let value = DateTime::from_naive_utc_and_offset( - NaiveDate::from_ymd_opt(2022, 9, 16) - .unwrap() - .and_hms_milli_opt(7, 59, 29, 321) - .unwrap(), - Utc.fix(), - ); - test_conversion!( - "2022-09-16 07:59:29.321", - Type::TIMESTAMP, - Field::Timestamp(value) - ); - - let value = DateTime::from_naive_utc_and_offset( - NaiveDate::from_ymd_opt(2022, 9, 16) - .unwrap() - .and_hms_micro_opt(3, 56, 30, 959787) - .unwrap(), - Utc.fix(), - ); - test_conversion!( - "2022-09-16 10:56:30.959787+07", - Type::TIMESTAMPTZ, - Field::Timestamp(value) - ); - - let value = DateTime::from_naive_utc_and_offset( - NaiveDate::from_ymd_opt(2022, 9, 16) - .unwrap() - .and_hms_micro_opt(7, 56, 30, 959787) - .unwrap(), - Utc.fix(), - ); - test_conversion!( - "2022-09-16 10:56:30.959787+03", - Type::TIMESTAMPTZ, - Field::Timestamp(value) - ); - - let value = DateTime::from_naive_utc_and_offset( - NaiveDate::from_ymd_opt(2022, 9, 16) - .unwrap() - .and_hms_micro_opt(14, 26, 30, 959787) - .unwrap(), - Utc.fix(), - ); - test_conversion!( - "2022-09-16 10:56:30.959787-03:30", - Type::TIMESTAMPTZ, - Field::Timestamp(value) - ); - - let value = json!({"abc": "foo"}); - test_conversion!("{\"abc\":\"foo\"}", Type::JSONB, Field::Json(value.clone())); - test_conversion!("{\"abc\":\"foo\"}", Type::JSON, Field::Json(value)); - - let value = json!([{"abc": "foo"}]); - test_conversion!( - "[{\"abc\":\"foo\"}]", - Type::JSON_ARRAY, - Field::Json(value.clone()) - ); - test_conversion!("[{\"abc\":\"foo\"}]", Type::JSONB_ARRAY, Field::Json(value)); - - test_conversion!("t", Type::BOOL, Field::Boolean(true)); - test_conversion!("f", Type::BOOL, Field::Boolean(false)); - - test_conversion!( - "(1.234,2.456)", - Type::POINT, - Field::Point(DozerPoint::from((1.234, 2.456))) - ); - } - - #[test] - fn it_maps_postgres_type_to_dozer_type() { - test_type_mapping!(Type::INT8, FieldType::Int); - test_type_mapping!(Type::FLOAT8, FieldType::Float); - test_type_mapping!(Type::VARCHAR, FieldType::String); - test_type_mapping!(Type::ANYENUM, FieldType::String); - test_type_mapping!(Type::UUID, FieldType::String); - test_type_mapping!(Type::BYTEA, FieldType::Binary); - test_type_mapping!(Type::NUMERIC, FieldType::Decimal); - test_type_mapping!(Type::TIMESTAMP, FieldType::Timestamp); - test_type_mapping!(Type::TIMESTAMPTZ, FieldType::Timestamp); - test_type_mapping!(Type::JSONB, FieldType::Json); - test_type_mapping!(Type::JSON, FieldType::Json); - test_type_mapping!(Type::JSONB_ARRAY, FieldType::Json); - test_type_mapping!(Type::JSON_ARRAY, FieldType::Json); - test_type_mapping!(Type::BOOL, FieldType::Boolean); - test_type_mapping!(Type::POINT, FieldType::Point); - } - - #[test] - fn test_none_value() { - let value = postgres_type_to_field( - None, - &TableColumn { - name: "column".to_string(), - flags: 0, - r#type: Type::VARCHAR, - column_index: 0, - }, - ); - assert_eq!(value.unwrap(), Field::Null); - } -} diff --git a/dozer-ingestion/postgres/src/iterator.rs b/dozer-ingestion/postgres/src/iterator.rs deleted file mode 100644 index a618691dd1..0000000000 --- a/dozer-ingestion/postgres/src/iterator.rs +++ /dev/null @@ -1,213 +0,0 @@ -use std::str::FromStr; -use std::sync::Arc; - -use dozer_ingestion_connector::dozer_types::log::debug; -use dozer_ingestion_connector::utils::ListOrFilterColumns; -use dozer_ingestion_connector::Ingestor; -use postgres_types::PgLsn; - -use crate::connection::helper; -use crate::connector::REPLICATION_SLOT_PREFIX; -use crate::replication_slot_helper::ReplicationSlotHelper; -use crate::replicator::CDCHandler; -use crate::snapshotter::PostgresSnapshotter; -use crate::PostgresConnectorError; - -use super::schema::helper::PostgresTableInfo; - -pub struct Details { - name: String, - publication_name: String, - slot_name: String, - tables: Vec, - replication_conn_config: tokio_postgres::Config, - conn_config: tokio_postgres::Config, - schema: Option, - batch_size: usize, -} - -#[derive(Debug, Clone, Copy)] -pub enum ReplicationState { - Pending, - SnapshotInProgress, - Replicating, -} - -pub struct PostgresIterator<'a> { - details: Arc
, - ingestor: &'a Ingestor, -} - -impl<'a> PostgresIterator<'a> { - #![allow(clippy::too_many_arguments)] - pub fn new( - name: String, - publication_name: String, - slot_name: String, - tables: Vec, - replication_conn_config: tokio_postgres::Config, - ingestor: &'a Ingestor, - conn_config: tokio_postgres::Config, - schema: Option, - batch_size: usize, - ) -> Self { - let details = Arc::new(Details { - name, - publication_name, - slot_name, - tables, - replication_conn_config, - conn_config, - schema, - batch_size, - }); - PostgresIterator { details, ingestor } - } -} - -impl<'a> PostgresIterator<'a> { - pub async fn start(self, lsn: Option) -> Result<(), PostgresConnectorError> { - let state = ReplicationState::Pending; - let details = self.details.clone(); - - let mut stream_inner = PostgresIteratorHandler { - details, - ingestor: self.ingestor, - state, - lsn, - }; - stream_inner.start().await - } -} - -pub struct PostgresIteratorHandler<'a> { - pub details: Arc
, - pub lsn: Option, - pub state: ReplicationState, - pub ingestor: &'a Ingestor, -} - -impl<'a> PostgresIteratorHandler<'a> { - /* - Replication involves 3 states - 1) Pending - - Initialize a replication slot. - - Initialize snapshots - - 2) SnapshotInProgress - - Sync initial snapshots of specified tables - - Commit with lsn - - 3) Replicating - - Replicate CDC events using lsn - */ - pub async fn start(&mut self) -> Result<(), PostgresConnectorError> { - let details = Arc::clone(&self.details); - let replication_conn_config = details.replication_conn_config.to_owned(); - let mut client = helper::connect(replication_conn_config).await?; - - // TODO: Handle cases: - // - When snapshot replication is not completed - // - When there is gap between available lsn (in case when slot dropped and new created) and last lsn - // - When publication tables changes - - // We clear inactive replication slots before starting replication - ReplicationSlotHelper::clear_inactive_slots( - &mut client, - REPLICATION_SLOT_PREFIX, - Some(&details.slot_name), - ) - .await?; - - if self.lsn.is_none() { - debug!("\nCreating Slot...."); - let slot_exist = - ReplicationSlotHelper::replication_slot_exists(&mut client, &details.slot_name) - .await?; - - if slot_exist { - // We dont have lsn, so we need to drop replication slot and start from scratch - ReplicationSlotHelper::drop_replication_slot(&mut client, &details.slot_name) - .await - .map_err(PostgresConnectorError::InvalidQueryError)?; - } - - client - .simple_query("BEGIN READ ONLY ISOLATION LEVEL REPEATABLE READ;") - .await - .map_err(|_e| { - debug!("failed to begin txn for replication"); - PostgresConnectorError::BeginReplication - })?; - - let replication_slot_lsn = - ReplicationSlotHelper::create_replication_slot(&mut client, &details.slot_name) - .await?; - if let Some(lsn) = replication_slot_lsn { - self.lsn = PgLsn::from_str(&lsn).map_or_else( - |_| Err(PostgresConnectorError::LsnParseError(lsn.to_string())), - |lsn| Ok(Some(lsn)), - )?; - } else { - return Err(PostgresConnectorError::LsnNotReturnedFromReplicationSlot); - } - - self.state = ReplicationState::SnapshotInProgress; - - /* ##################### SnapshotInProgress ###################### */ - debug!("\nInitializing snapshots..."); - - let snapshotter = PostgresSnapshotter { - conn_config: details.conn_config.to_owned(), - ingestor: self.ingestor, - schema: details.schema.clone(), - batch_size: details.batch_size, - }; - let tables = details - .tables - .iter() - .map(|table_info| ListOrFilterColumns { - name: table_info.name.clone(), - columns: Some(table_info.columns.clone()), - schema: Some(table_info.schema.clone()), - }) - .collect::>(); - snapshotter.sync_tables(&tables).await?; - - debug!("\nInitialized with tables: {:?}", details.tables); - - client.simple_query("COMMIT;").await.map_err(|_e| { - debug!("failed to commit txn for replication"); - PostgresConnectorError::CommitReplication - })?; - } - - self.state = ReplicationState::Replicating; - - /* #################### Replicating ###################### */ - self.replicate().await - } - - async fn replicate(&self) -> Result<(), PostgresConnectorError> { - let lsn = self - .lsn - .as_ref() - .ok_or(PostgresConnectorError::LSNNotStoredError)?; - - let publication_name = self.details.publication_name.clone(); - let slot_name = self.details.slot_name.clone(); - let tables = self.details.tables.clone(); - let mut replicator = CDCHandler { - replication_conn_config: self.details.replication_conn_config.clone(), - ingestor: self.ingestor, - start_lsn: *lsn, - begin_lsn: 0, - offset_lsn: 0, - publication_name, - slot_name, - last_commit_lsn: 0, - name: self.details.name.clone(), - }; - replicator.start(tables).await - } -} diff --git a/dozer-ingestion/postgres/src/lib.rs b/dozer-ingestion/postgres/src/lib.rs deleted file mode 100644 index 85e7324117..0000000000 --- a/dozer-ingestion/postgres/src/lib.rs +++ /dev/null @@ -1,221 +0,0 @@ -use std::num::ParseIntError; -use std::string::FromUtf8Error; - -use dozer_ingestion_connector::dozer_types::{ - chrono, - errors::types::{DeserializationError, TypeError}, - thiserror::{self, Error}, -}; -use tokio_postgres::config::SslMode; - -pub mod connection; -pub mod connector; -pub mod helper; -pub mod iterator; -mod replication_slot_helper; -pub mod replicator; -mod schema; -pub mod snapshotter; -#[cfg(test)] -pub mod test_utils; -#[cfg(test)] -pub mod tests; -pub mod xlog_mapper; - -pub use tokio_postgres; - -#[derive(Error, Debug)] -pub enum PostgresConnectorError { - #[error("Failed to map configuration: {0}")] - WrongConnectionConfiguration(DeserializationError), - - #[error("Invalid SslMode: {0:?}")] - InvalidSslError(SslMode), - - #[error("Failed to convert slot name from state. Error: {0}")] - StringReadError(#[from] FromUtf8Error), - - #[error("Query failed in connector: {0}")] - InvalidQueryError(#[source] tokio_postgres::Error), - - #[error("Failed to connect to postgres with the specified configuration. {0}")] - ConnectionFailure(#[source] tokio_postgres::Error), - - #[error("Replication is not available for user")] - ReplicationIsNotAvailableForUserError, - - #[error("WAL level should be 'logical'")] - WALLevelIsNotCorrect(), - - #[error("Cannot find tables {0:?}")] - TablesNotFound(Vec<(String, String)>), - - #[error("Cannot find column {0} in {1}")] - ColumnNotFound(String, String), - - #[error("Cannot find columns {0}")] - ColumnsNotFound(String), - - #[error("Failed to create a replication slot \"{0}\". Error: {1}")] - CreateSlotError(String, #[source] tokio_postgres::Error), - - #[error("Failed to create publication: {0}")] - CreatePublicationError(#[source] tokio_postgres::Error), - - #[error("Failed to drop publication: {0}")] - DropPublicationError(#[source] tokio_postgres::Error), - - #[error("Failed to begin txn for replication")] - BeginReplication, - - #[error("Failed to begin txn for replication")] - CommitReplication, - - #[error("Fetch of replication slot info failed. Error: {0}")] - FetchReplicationSlotError(#[source] tokio_postgres::Error), - - #[error("No slots available or all available slots are used")] - NoAvailableSlotsError, - - #[error("Slot {0} not found")] - SlotNotExistError(String), - - #[error("Slot {0} is already used by another process")] - SlotIsInUseError(String), - - #[error("Table {0} changes is not replicated to slot")] - MissingTableInReplicationSlot(String), - - #[error("Start lsn is before first available lsn - {0} < {1}")] - StartLsnIsBeforeLastFlushedLsnError(String, String), - - #[error("fetch of replication slot info failed. Error: {0}")] - SyncWithSnapshotError(String), - - #[error("Replication stream error. Error: {0}")] - ReplicationStreamError(tokio_postgres::Error), - - #[error("Received unexpected message in replication stream")] - UnexpectedReplicationMessageError, - - #[error("Replication stream error")] - ReplicationStreamEndError, - - #[error(transparent)] - PostgresSchemaError(#[from] PostgresSchemaError), - - #[error("LSN not stored for replication slot")] - LSNNotStoredError, - - #[error("LSN parse error. Given lsn: {0}")] - LsnParseError(String), - - #[error("LSN not returned from replication slot creation query")] - LsnNotReturnedFromReplicationSlot, - - #[error("Table name \"{0}\" not valid")] - TableNameNotValid(String), - - #[error("Column name \"{0}\" not valid")] - ColumnNameNotValid(String), - - #[error("Relation not found in replication: {0}")] - RelationNotFound(#[source] std::io::Error), - - #[error("Failed to send message on snapshot read channel")] - SnapshotReadError, - - #[error("Failed to load native certs: {0}")] - LoadNativeCerts(#[source] std::io::Error), - - #[error("Non utf8 column name in table {table_index} column {column_index}")] - NonUtf8ColumnName { - table_index: usize, - column_index: usize, - }, - - #[error("Column type changed in table {table_index} column {column_name} from {old_type} to {new_type}")] - ColumnTypeChanged { - table_index: usize, - column_name: String, - old_type: postgres_types::Type, - new_type: postgres_types::Type, - }, - - #[error("Unexpected query message")] - UnexpectedQueryMessageError, -} - -#[derive(Error, Debug)] -pub enum PostgresSchemaError { - #[error("Schema's '{0}' doesn't have primary key")] - PrimaryKeyIsMissingInSchema(String), - - #[error("Table: '{0}' replication identity settings are not correct. It is either not set or NOTHING. Missing a primary key ?")] - SchemaReplicationIdentityError(String), - - #[error("Column type {0} not supported")] - ColumnTypeNotSupported(String), - - #[error("Custom type {0:?} is not supported yet. Join our Discord at https://discord.com/invite/3eWXBgJaEQ - we're here to help with your use case!")] - CustomTypeNotSupported(String), - - #[error("ColumnTypeNotFound")] - ColumnTypeNotFound, - - #[error("Invalid column type of column {0}")] - InvalidColumnType(String), - - #[error("Value conversion error: {0}")] - ValueConversionError(String), - - #[error("String parse failed")] - StringParseError(#[source] FromUtf8Error), - - #[error("JSONB parse failed: {0}")] - JSONBParseError(String), - - #[error("Point parse failed")] - PointParseError, - - #[error("Unsupported replication type - '{0}'")] - UnsupportedReplicationType(String), - - #[error( - "Table type '{0}' of '{1}' table is not supported. Only 'BASE TABLE' type is supported" - )] - UnsupportedTableType(String, String), - - #[error("Table type cannot be determined")] - TableTypeNotFound, - - #[error("Column not found")] - ColumnNotFound, - - #[error("Type error: {0}")] - TypeError(#[from] TypeError), - - #[error("Failed to read string from utf8. Error: {0}")] - StringReadError(#[from] FromUtf8Error), - - #[error("Failed to read date. Error: {0}")] - DateReadError(#[from] chrono::ParseError), - - #[error(transparent)] - DateConversionError(#[from] DateConversionError), -} - -#[derive(Error, Debug)] -pub enum DateConversionError { - #[error("Failed to read error part. Error: {0}")] - FailedParseDate(#[from] ParseIntError), - - #[error("Failed to convert date")] - InvalidDate, - - #[error("Failed to convert time")] - InvalidTime, - - #[error("Ambiguous date result")] - AmbiguousTimeResult, -} diff --git a/dozer-ingestion/postgres/src/readme.md b/dozer-ingestion/postgres/src/readme.md deleted file mode 100644 index 3cc68bad02..0000000000 --- a/dozer-ingestion/postgres/src/readme.md +++ /dev/null @@ -1,50 +0,0 @@ -# Postgres requirements - -### Version -At least **v10**. To verify it, you can run -```sql -SHOW server_version; -``` - -### WAL Level -**Logical**. It can be verified with -```sql -SHOW wal_level; -``` - -You can change wal_level with this query. After changing you must restart your server. -```sql -ALTER SYSTEM SET wal_level = logical; -``` - -In AWS RDS you need to use custom parameters group. https://aws.amazon.com/premiumsupport/knowledge-center/rds-postgresql-use-logical-replication/ -> To turn on logical replication in RDS for PostgreSQL, modify a custom parameter group to set rds.logical_replication to 1 and attach rds.logical_replication to the DB instance. Update the parameter group to set rds.logical_replication to 1 if a custom parameter group is attached to a DB instance. The rds.logical_replication parameter is a static parameter that requires a DB instance reboot to take effect. When the DB instance reboots, the wal_level parameter is set to logical. -> -> -- [AWS tutorial][1] - -### Replication slots -Database should have at least one available replication slot. - -```sql --- Fetch max replication slots -SHOW max_replication_slots; - ---- Get current used replication slots count -SELECT COUNT(*) FROM pg_replication_slots; - --- If there is no empty slot, you can drop any unused slot with -SELECT * FROM pg_drop_replication_slot('slot_name'); -``` - -### User -To use replication postgres database user should have replication permission - `userepl`.
-Permission can be checked with this query -```sql -SELECT usename, userepl FROM pg_user WHERE usename = "current_user"() -``` -If it is not enabled, you can grant permission with this query -```sql -ALTER USER WITH REPLICATION; -``` - -[1]: https://aws.amazon.com/premiumsupport/knowledge-center/rds-postgresql-use-logical-replication/ \ No newline at end of file diff --git a/dozer-ingestion/postgres/src/readme_replication.md b/dozer-ingestion/postgres/src/readme_replication.md deleted file mode 100644 index 075576c180..0000000000 --- a/dozer-ingestion/postgres/src/readme_replication.md +++ /dev/null @@ -1,133 +0,0 @@ -# Replica identity and mapped operation - -```sql -CREATE SEQUENCE users_id_seq; - -CREATE TABLE users -( - id INTEGER NOT NULL DEFAULT nextval('users_id_seq') - CONSTRAINT users_pk - PRIMARY KEY, - email VARCHAR(255) NOT NULL, - phone VARCHAR(255) NOT NULL -); - -CREATE UNIQUE INDEX email_index - ON users (email); - -ALTER SEQUENCE users_id_seq - OWNED BY users.id; - -``` -Main difference between replica identity types is what old record data is sent with `update` and `delete` operations - -## DEFAULT -By default on update and delete events only sending primary key from old record - -```sql --- Change of replication identity -ALTER TABLE users REPLICA IDENTITY DEFAULT; -``` - -```sql --- Example of update -BEGIN; -INSERT INTO users (email, phone) VALUES ('test1@email.com', '98765421'); -COMMIT; - -BEGIN; -UPDATE users SET phone = '99339439439' WHERE email = 'test1@email.com'; -COMMIT; -``` - - -#### Example of update replication messages in update transaction - -| Replication message | Operation | -|----------------------------------------------------------------------------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| -| `BEGIN (transaction id)` | | -| ```UPDATE (old: { id: 1}, new: {id: 1, phone: '99339439439', 'email': 'test1@email.com'})``` |
OperationEvent(
Operation::Update {
new: Record {schema_id: 1,values: vec![Field::Int(1), Field::String('test1@email.com'), Field::String('99339439439')],},
old: Record {schema_id: 1,values: vec![Field::Int(1), Field::Null, Field::Null],
}
}
| -| `COMMIT (commit_lsn)` | | - -## USING INDEX -When using index as replica identity only values used in unique key are sent - -```sql --- Change of replication identity -ALTER TABLE users REPLICA IDENTITY USING INDEX email_index; -``` - -```sql --- Example of update -BEGIN; -INSERT INTO users (email, phone) VALUES ('test2@email.com', '98765422'); -COMMIT; - -BEGIN; -UPDATE users SET phone = '99339439440' WHERE email = 'test2@email.com'; -COMMIT; -``` - -#### Example of update replication messages in update transaction - -| Replication message | Operation | -|-----------------------------------------------------------------------------------------------------------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| -| `BEGIN (transaction id)` | | -| ```UPDATE (old: { email: 'test2@email.com'}, new: {id: 2, phone: '99339439440', 'email': 'test2@email.com'})``` |
OperationEvent(
Operation::Update {
new: Record {schema_id: 1,values: vec![Field::Int(2), Field::String('test2@email.com'), Field::String('99339439440')],},
old: Record {schema_id: 1,values: vec![Field::Null, Field::String("test2@email.com"), Field::Null],
}
}
| -| `COMMIT (commit_lsn)` | | - - -## FULL -When using full mode as replica identity, all row values are sent during update -```sql --- Change of replication identity -ALTER TABLE users REPLICA IDENTITY FULL; -``` - -```sql --- Example of update -BEGIN; -INSERT INTO users (email, phone) VALUES ('test3@email.com', '98765423'); -COMMIT; - -BEGIN; -UPDATE users SET phone = '99339439441' WHERE email = 'test3@email.com'; -COMMIT; -``` - -#### Example of update replication messages in update transaction - -| Replication message | Operation | -|-------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| -| `BEGIN (transaction id)` | | -| ```UPDATE (old: { email: 'test3@email.com', phone: '98765423', id: 3}, new: {id: 3, phone: '99339439441', 'email': 'test3@email.com'})``` |
OperationEvent(
Operation::Update {
new: Record {schema_id: 1,values: vec![Field::Int(3), Field::String('test3@email.com'), Field::String('99339439441')],},
old: Record {schema_id: 1,values: vec![Field::Int(3), Field::String("test3@email.com"), Field::String("98765423")],
}
}
| -| `COMMIT (commit_lsn)` | | - - - -## NOTHING -Nothing is sent from old row when using this type - -```sql --- Change of replication identity -ALTER TABLE users REPLICA IDENTITY NOTHING; -``` - -```sql --- Example of update -BEGIN; -INSERT INTO users (email, phone) VALUES ('test4@email.com', '98765424'); -COMMIT; - -BEGIN; -UPDATE users SET phone = '99339439442' WHERE email = 'test4@email.com'; -COMMIT; -``` - -#### Example of update replication messages in update transaction - -| Replication message | Operation | -|-------------------------------------------------------------------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| -| `BEGIN (transaction id)` | | -| ```UPDATE (new: {id: 4, phone: '99339439442', 'email': 'test4@email.com'})``` |
OperationEvent(
Operation::Update {
new: Record {schema_id: 1,values: vec![Field::Int(4), Field::String('test4@email.com'), Field::String('99339439442')],},
old: Record {schema_id: 1,values: vec![Field::Null, Field::Null, Field::Null],
}
}
| -| `COMMIT (commit_lsn)` | | diff --git a/dozer-ingestion/postgres/src/replication_slot_helper.rs b/dozer-ingestion/postgres/src/replication_slot_helper.rs deleted file mode 100644 index 8532b405de..0000000000 --- a/dozer-ingestion/postgres/src/replication_slot_helper.rs +++ /dev/null @@ -1,255 +0,0 @@ -use crate::PostgresConnectorError; - -use super::connection::client::Client; -use dozer_ingestion_connector::dozer_types::log::debug; -use tokio_postgres::{Error, SimpleQueryMessage}; - -pub struct ReplicationSlotHelper {} - -impl ReplicationSlotHelper { - pub async fn drop_replication_slot( - client: &mut Client, - slot_name: &str, - ) -> Result, Error> { - let res = client - .simple_query(format!("select pg_drop_replication_slot('{slot_name}');").as_ref()) - .await; - match res { - Ok(_) => debug!("dropped replication slot {}", slot_name), - Err(_) => debug!("failed to drop replication slot..."), - }; - - res - } - - pub async fn create_replication_slot( - client: &mut Client, - slot_name: &str, - ) -> Result, PostgresConnectorError> { - let create_replication_slot_query = - format!(r#"CREATE_REPLICATION_SLOT {slot_name:?} LOGICAL "pgoutput" USE_SNAPSHOT"#); - - let slot_query_row = client - .simple_query(&create_replication_slot_query) - .await - .map_err(|e| { - debug!("failed to create replication slot {}", slot_name); - PostgresConnectorError::CreateSlotError(slot_name.to_string(), e) - })?; - - if let SimpleQueryMessage::Row(row) = &slot_query_row[0] { - Ok(row.get("consistent_point").map(|lsn| lsn.to_string())) - } else { - Err(PostgresConnectorError::UnexpectedQueryMessageError) - } - } - - pub async fn replication_slot_exists( - client: &mut Client, - slot_name: &str, - ) -> Result { - let replication_slot_info_query = - format!(r#"SELECT * FROM pg_replication_slots where slot_name = '{slot_name}';"#); - - let slot_query_row = client - .simple_query(&replication_slot_info_query) - .await - .map_err(PostgresConnectorError::FetchReplicationSlotError)?; - - Ok(matches!( - slot_query_row.first(), - Some(SimpleQueryMessage::Row(_)) - )) - } - - pub async fn clear_inactive_slots( - client: &mut Client, - slot_name_prefix: &str, - current_slot_name: Option<&str>, - ) -> Result<(), PostgresConnectorError> { - let condition = match current_slot_name { - Some(name) => format!("AND slot_name != '{name}'"), - None => "".to_string(), - }; - - let inactive_slots_query = format!( - r#"SELECT * FROM pg_replication_slots where active = false AND slot_name LIKE '{slot_name_prefix}%' {condition};"# - ); - - let slots = client - .simple_query(&inactive_slots_query) - .await - .map_err(PostgresConnectorError::FetchReplicationSlotError)?; - - let column_index = if let Some(SimpleQueryMessage::Row(row)) = slots.first() { - row.columns().iter().position(|c| c.name() == "slot_name") - } else { - None - }; - - for slot_message in slots { - if let SimpleQueryMessage::Row(row) = slot_message { - if let Some(index) = column_index { - let slot_name = row.get(index); - - if let Some(name) = slot_name { - Self::drop_replication_slot(client, name) - .await - .map_err(PostgresConnectorError::InvalidQueryError)?; - } - } - } - } - - Ok(()) - } -} - -#[cfg(test)] -mod tests { - use dozer_ingestion_connector::tokio; - use serial_test::serial; - use tokio_postgres::config::ReplicationMode; - - use crate::{ - connection::helper::{connect, map_connection_config}, - test_utils::load_test_connection_config, - PostgresConnectorError, - }; - - use super::ReplicationSlotHelper; - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_replication_slot_create_successfully() { - let config = load_test_connection_config().await; - let mut config = map_connection_config(&config).unwrap(); - config.replication_mode(ReplicationMode::Logical); - - let mut client = connect(config).await.unwrap(); - - client - .simple_query("BEGIN READ ONLY ISOLATION LEVEL REPEATABLE READ;") - .await - .unwrap(); - - let actual = ReplicationSlotHelper::create_replication_slot(&mut client, "test").await; - - assert!(actual.is_ok()); - - match actual { - Err(_) => panic!("Validation should fail"), - Ok(result) => { - if let Some(address) = result { - assert_ne!(address, "") - } else { - panic!("Validation should fail") - } - } - } - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_replication_slot_create_failed_if_existed() { - let slot_name = "test"; - let config = load_test_connection_config().await; - let mut config = map_connection_config(&config).unwrap(); - config.replication_mode(ReplicationMode::Logical); - - let mut client = connect(config).await.unwrap(); - - client - .simple_query("BEGIN READ ONLY ISOLATION LEVEL REPEATABLE READ;") - .await - .unwrap(); - - let create_replication_slot_query = - format!(r#"CREATE_REPLICATION_SLOT {slot_name:?} LOGICAL "pgoutput" USE_SNAPSHOT"#); - - client - .simple_query(&create_replication_slot_query) - .await - .expect("failed"); - - let actual = ReplicationSlotHelper::create_replication_slot(&mut client, slot_name).await; - - assert!(actual.is_err()); - - match actual { - Ok(_) => panic!("Validation should fail"), - Err(e) => { - if let PostgresConnectorError::CreateSlotError(_, err) = e { - assert_eq!( - err.as_db_error().unwrap().message(), - format!("replication slot \"{slot_name}\" already exists") - ); - } else { - panic!("Unexpected error occurred"); - } - } - } - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_replication_slot_drop_successfully() { - let slot_name = "test"; - let config = load_test_connection_config().await; - let mut config = map_connection_config(&config).unwrap(); - config.replication_mode(ReplicationMode::Logical); - - let mut client = connect(config).await.unwrap(); - - client - .simple_query("BEGIN READ ONLY ISOLATION LEVEL REPEATABLE READ;") - .await - .unwrap(); - - let create_replication_slot_query = - format!(r#"CREATE_REPLICATION_SLOT {slot_name:?} LOGICAL "pgoutput" USE_SNAPSHOT"#); - - client - .simple_query(&create_replication_slot_query) - .await - .expect("failed"); - - let actual = ReplicationSlotHelper::drop_replication_slot(&mut client, slot_name).await; - - assert!(actual.is_ok()); - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_replication_slot_drop_failed_if_slot_not_exist() { - let slot_name = "test"; - let config = load_test_connection_config().await; - let mut config = map_connection_config(&config).unwrap(); - config.replication_mode(ReplicationMode::Logical); - - let mut client = connect(config).await.unwrap(); - - client - .simple_query("BEGIN READ ONLY ISOLATION LEVEL REPEATABLE READ;") - .await - .unwrap(); - - let actual = ReplicationSlotHelper::drop_replication_slot(&mut client, slot_name).await; - - assert!(actual.is_err()); - - match actual { - Ok(_) => panic!("Validation should fail"), - Err(e) => { - assert_eq!( - e.as_db_error().unwrap().message(), - format!("replication slot \"{slot_name}\" does not exist") - ); - } - } - } -} diff --git a/dozer-ingestion/postgres/src/replicator.rs b/dozer-ingestion/postgres/src/replicator.rs deleted file mode 100644 index 6a472f5039..0000000000 --- a/dozer-ingestion/postgres/src/replicator.rs +++ /dev/null @@ -1,272 +0,0 @@ -use dozer_ingestion_connector::dozer_types::bytes; -use dozer_ingestion_connector::dozer_types::chrono::{TimeZone, Utc}; -use dozer_ingestion_connector::dozer_types::log::{error, info}; -use dozer_ingestion_connector::dozer_types::models::ingestion_types::{ - IngestionMessage, TransactionInfo, -}; -use dozer_ingestion_connector::dozer_types::node::OpIdentifier; -use dozer_ingestion_connector::futures::StreamExt; -use dozer_ingestion_connector::Ingestor; -use postgres_protocol::message::backend::ReplicationMessage::*; -use postgres_protocol::message::backend::{LogicalReplicationMessage, ReplicationMessage}; -use postgres_protocol::Lsn; -use postgres_types::PgLsn; -use tokio_postgres::Error; - -use std::pin::Pin; -use std::time::SystemTime; - -use crate::connection::client::Client; -use crate::connection::helper::{self, is_network_failure}; -use crate::xlog_mapper::XlogMapper; -use crate::PostgresConnectorError; - -use super::schema::helper::PostgresTableInfo; -use super::xlog_mapper::MappedReplicationMessage; - -pub struct CDCHandler<'a> { - pub name: String, - pub ingestor: &'a Ingestor, - - pub replication_conn_config: tokio_postgres::Config, - pub publication_name: String, - pub slot_name: String, - - pub start_lsn: PgLsn, - pub begin_lsn: Lsn, - pub offset_lsn: Lsn, - pub last_commit_lsn: Lsn, -} - -impl<'a> CDCHandler<'a> { - pub async fn start( - &mut self, - tables: Vec, - ) -> Result<(), PostgresConnectorError> { - let replication_conn_config = self.replication_conn_config.clone(); - let client = helper::connect(replication_conn_config).await?; - - info!( - "[{}] Starting Replication: {:?}, {:?}", - self.name.clone(), - self.start_lsn, - self.publication_name.clone() - ); - - let lsn = self.start_lsn; - let options = format!( - r#"("proto_version" '1', "publication_names" '{publication_name}')"#, - publication_name = self.publication_name - ); - - self.offset_lsn = Lsn::from(lsn); - self.last_commit_lsn = Lsn::from(lsn); - - let mut stream = - LogicalReplicationStream::new(client, self.slot_name.clone(), lsn, options) - .await - .map_err(PostgresConnectorError::ReplicationStreamError)?; - - let tables_columns = tables - .into_iter() - .enumerate() - .map(|(table_index, table_info)| { - (table_info.relation_id, (table_index, table_info.columns)) - }) - .collect(); - let mut mapper = XlogMapper::new(tables_columns); - - loop { - let message = stream.next().await; - if let Some(Ok(PrimaryKeepAlive(ref k))) = message { - if k.reply() == 1 { - // Postgres' keep alive feedback function expects time from 2000-01-01 00:00:00 - let since_the_epoch = SystemTime::now() - .duration_since(SystemTime::from( - Utc.with_ymd_and_hms(2020, 1, 1, 0, 0, 0).unwrap(), - )) - .unwrap() - .as_millis(); - stream - .standby_status_update( - PgLsn::from(self.last_commit_lsn), - PgLsn::from(self.last_commit_lsn), - PgLsn::from(self.last_commit_lsn), - since_the_epoch as i64, - 1, - ) - .await - .unwrap(); - } - } else { - self.handle_replication_message(message, &mut mapper) - .await?; - } - } - } - - pub async fn handle_replication_message( - &mut self, - message: Option, Error>>, - mapper: &mut XlogMapper, - ) -> Result<(), PostgresConnectorError> { - match message { - Some(Ok(XLogData(body))) => { - let lsn = body.wal_start(); - let message = mapper.handle_message(body)?; - - match message { - Some(MappedReplicationMessage::Commit(lsn)) => { - self.last_commit_lsn = lsn; - if self - .ingestor - .handle_message(IngestionMessage::TransactionInfo( - TransactionInfo::Commit { - id: Some(OpIdentifier::new(self.begin_lsn, 0)), - source_time: None, - }, - )) - .await - .is_err() - { - return Ok(()); - } - } - Some(MappedReplicationMessage::Begin) => { - self.begin_lsn = lsn; - } - Some(MappedReplicationMessage::Operation { table_index, op }) => { - if self.begin_lsn != self.offset_lsn - && self - .ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index, - op, - id: Some(OpIdentifier::new(self.begin_lsn, 0)), - }) - .await - .is_err() - { - // If the ingestion channel is closed, we should stop the replication - return Ok(()); - } - } - None => {} - } - - Ok(()) - } - Some(Ok(msg)) => { - error!("Unexpected message: {:?}", msg); - Err(PostgresConnectorError::UnexpectedReplicationMessageError) - } - Some(Err(e)) => Err(PostgresConnectorError::ReplicationStreamError(e)), - None => Err(PostgresConnectorError::ReplicationStreamEndError), - } - } -} - -pub struct LogicalReplicationStream { - client: Client, - slot_name: String, - resume_lsn: PgLsn, - options: String, - inner: Pin>, -} - -impl LogicalReplicationStream { - pub async fn new( - mut client: Client, - slot_name: String, - lsn: PgLsn, - options: String, - ) -> Result { - let inner = - Box::pin(Self::open_replication_stream(&mut client, &slot_name, lsn, &options).await?); - Ok(Self { - client, - slot_name, - resume_lsn: lsn, - options, - inner, - }) - } - - pub async fn next( - &mut self, - ) -> Option, tokio_postgres::Error>> { - loop { - let result = self.inner.next().await; - match result.as_ref() { - Some(Err(err)) if is_network_failure(err) => { - if let Err(err) = self.resume().await { - return Some(Err(err)); - } - continue; - } - Some(Ok(XLogData(body))) => self.resume_lsn = body.wal_end().into(), - _ => {} - } - return result; - } - } - - pub async fn standby_status_update( - &mut self, - write_lsn: PgLsn, - flush_lsn: PgLsn, - apply_lsn: PgLsn, - ts: i64, - reply: u8, - ) -> Result<(), Error> { - loop { - match self - .inner - .as_mut() - .standby_status_update(write_lsn, flush_lsn, apply_lsn, ts, reply) - .await - { - Err(err) if is_network_failure(&err) => { - self.resume().await?; - continue; - } - _ => {} - } - break Ok(()); - } - } - - async fn resume(&mut self) -> Result<(), tokio_postgres::Error> { - self.client.reconnect().await?; - - let stream = Self::open_replication_stream( - &mut self.client, - &self.slot_name, - self.resume_lsn, - &self.options, - ) - .await?; - - self.inner = Box::pin(stream); - - Ok(()) - } - - async fn open_replication_stream( - client: &mut Client, - slot_name: &str, - lsn: PgLsn, - options: &str, - ) -> Result { - let query = format!( - r#"START_REPLICATION SLOT {:?} LOGICAL {} {}"#, - slot_name, lsn, options - ); - - let copy_stream = client.copy_both_simple::(&query).await?; - - Ok(tokio_postgres::replication::LogicalReplicationStream::new( - copy_stream, - )) - } -} diff --git a/dozer-ingestion/postgres/src/schema/helper.rs b/dozer-ingestion/postgres/src/schema/helper.rs deleted file mode 100644 index aa869c7e2d..0000000000 --- a/dozer-ingestion/postgres/src/schema/helper.rs +++ /dev/null @@ -1,441 +0,0 @@ -use std::collections::HashMap; - -use dozer_ingestion_connector::{ - dozer_types::types::{FieldDefinition, FieldType, Schema, SourceDefinition}, - utils::ListOrFilterColumns, - CdcType, SourceSchema, -}; -use postgres_types::Type; -use tokio_postgres::Row; - -use crate::{ - connection::helper, helper::postgres_type_to_dozer_type, PostgresConnectorError, - PostgresSchemaError, -}; - -use super::sorter::sort_schemas; - -#[derive(Debug)] -pub struct SchemaHelper { - conn_config: tokio_postgres::Config, - // Postgres schema - schema: Option, -} - -struct PostgresTableRow { - schema: String, - table_name: String, - field: FieldDefinition, - is_column_used_in_index: bool, - replication_type: String, -} - -#[derive(Clone, Debug)] -pub struct PostgresTable { - fields: Vec, - // Indexes of fields, which are used for replication identity - // Default - uses PK for identity - // Index - uses selected index fields for identity - // Full - all fields are used for identity - // Nothing - no fields can be used for identity. - // Postgres will not return old values in update and delete replication messages - index_keys: Vec, - replication_type: String, -} - -pub(crate) type SchemaTableIdentifier = (String, String); -impl PostgresTable { - pub fn new(replication_type: String) -> Self { - Self { - fields: vec![], - index_keys: vec![], - replication_type, - } - } - - pub fn add_field(&mut self, field: FieldDefinition, is_column_used_in_index: bool) { - self.fields.push(field); - self.index_keys.push(is_column_used_in_index); - } - - pub fn fields(&self) -> &Vec { - &self.fields - } - - pub fn is_index_field(&self, index: usize) -> Option<&bool> { - self.index_keys.get(index) - } - - pub fn get_field(&self, index: usize) -> Option<&FieldDefinition> { - self.fields.get(index) - } - - pub fn replication_type(&self) -> &String { - &self.replication_type - } -} - -#[derive(Debug, Clone)] -pub struct PostgresTableInfo { - pub schema: String, - pub name: String, - pub relation_id: u32, - pub columns: Vec, -} - -type RowsWithColumnsMap = (Vec, HashMap>); - -impl SchemaHelper { - pub fn new(conn_config: tokio_postgres::Config, schema: Option) -> SchemaHelper { - Self { - conn_config, - schema, - } - } - - pub async fn get_tables( - &self, - tables: Option<&[ListOrFilterColumns]>, - ) -> Result, PostgresConnectorError> { - let (results, tables_columns_map) = self.get_columns(tables).await?; - - let mut table_columns_map: HashMap)> = - HashMap::new(); - for row in results { - let schema: String = row.get(8); - let table_name: String = row.get(0); - let column_name: String = row.get(1); - let relation_id: u32 = row.get(4); - - let schema_table_tuple = (schema, table_name); - let add_column_table = tables_columns_map - .get(&schema_table_tuple) - .map_or(true, |columns| { - columns.is_empty() || columns.contains(&column_name) - }); - - if add_column_table { - match table_columns_map.get_mut(&schema_table_tuple) { - Some((existing_relation_id, columns)) => { - columns.push(column_name); - assert_eq!(*existing_relation_id, relation_id); - } - None => { - table_columns_map - .insert(schema_table_tuple, (relation_id, vec![column_name])); - } - } - } - } - - Ok(if let Some(tables) = tables { - let mut result = vec![]; - for table in tables { - result.push(find_table( - &table_columns_map, - table.schema.as_deref(), - &table.name, - )?); - } - result - } else { - table_columns_map - .into_iter() - .map( - |((schema, name), (relation_id, columns))| PostgresTableInfo { - name, - relation_id, - columns, - schema, - }, - ) - .collect() - }) - } - - async fn get_columns( - &self, - tables: Option<&[ListOrFilterColumns]>, - ) -> Result { - let mut tables_columns_map: HashMap> = HashMap::new(); - let mut client = helper::connect(self.conn_config.clone()).await?; - let query = if let Some(tables) = tables { - tables.iter().for_each(|t| { - if let Some(columns) = t.columns.clone() { - tables_columns_map.insert( - ( - t.schema - .as_ref() - .map_or(DEFAULT_SCHEMA_NAME.to_string(), |s| s.to_string()), - t.name.clone(), - ), - columns, - ); - } - }); - - let schemas: Vec = tables - .iter() - .map(|t| { - t.schema - .as_ref() - .map_or_else(|| DEFAULT_SCHEMA_NAME.to_string(), |s| s.clone()) - }) - .collect(); - let table_names: Vec = tables.iter().map(|t| t.name.clone()).collect(); - let sql = str::replace( - SQL, - ":tables_name_condition", - "t.table_schema = ANY($1) AND t.table_name = ANY($2)", - ); - client.query(&sql, &[&schemas, &table_names]).await - } else if let Some(schema) = &self.schema { - let sql = str::replace( - SQL, - ":tables_name_condition", - "t.table_schema = $1 AND t.table_type = 'BASE TABLE'", - ); - client.query(&sql, &[&schema]).await - } else { - let sql = str::replace(SQL, ":tables_name_condition", "t.table_type = 'BASE TABLE'"); - client.query(&sql, &[]).await - }; - - query - .map_err(PostgresConnectorError::InvalidQueryError) - .map(|rows| (rows, tables_columns_map)) - } - - pub async fn get_schemas( - &self, - tables: &[ListOrFilterColumns], - ) -> Result>, PostgresConnectorError> { - let (results, tables_columns_map) = self.get_columns(Some(tables)).await?; - - let mut columns_map: HashMap< - SchemaTableIdentifier, - Result, - > = HashMap::new(); - - results - .iter() - .filter_map(|row| { - let schema: String = row.get(8); - let table_name: String = row.get(0); - let column_name: String = row.get(1); - - tables_columns_map - .get(&(schema.clone(), table_name.clone())) - .map_or(Some((schema.clone(), table_name.clone(), row)), |columns| { - if columns.is_empty() || columns.contains(&column_name) { - Some((schema, table_name, row)) - } else { - None - } - }) - }) - .map(|(schema, table_name, r)| (schema, table_name, self.convert_row(r))) - .try_for_each( - |(schema, table_name, table_row)| -> Result<(), PostgresSchemaError> { - match table_row { - Ok(row) => { - columns_map - .entry((row.schema, row.table_name)) - .and_modify(|table| { - if let Ok(ref mut t) = table { - t.add_field(row.field.clone(), row.is_column_used_in_index); - } - }) - .or_insert_with(|| { - let mut table = PostgresTable::new(row.replication_type); - table.add_field(row.field, row.is_column_used_in_index); - Ok(table) - }); - } - Err(e) => { - columns_map.insert((schema, table_name), Err(e)); - } - } - - Ok(()) - }, - )?; - - Ok(Self::map_columns_to_schemas(sort_schemas( - tables, - columns_map, - )?)) - } - - fn map_columns_to_schemas( - postgres_tables: Vec<( - SchemaTableIdentifier, - Result, - )>, - ) -> Vec> { - postgres_tables - .into_iter() - .map(|((_, table_name), table)| { - table - .map_err(PostgresConnectorError::PostgresSchemaError) - .and_then(|table| { - Self::map_schema(&table_name, table) - .map_err(PostgresConnectorError::PostgresSchemaError) - }) - }) - .collect() - } - - fn map_schema( - table_name: &str, - table: PostgresTable, - ) -> Result { - let primary_index: Vec = table - .index_keys - .iter() - .enumerate() - .filter(|(_, b)| **b) - .map(|(idx, _)| idx) - .collect(); - - let schema = Schema { - fields: table.fields.clone(), - primary_index, - }; - - let cdc_type = match table.replication_type.as_str() { - "d" => { - if schema.primary_index.is_empty() { - Ok(CdcType::Nothing) - } else { - Ok(CdcType::OnlyPK) - } - } - "i" => Ok(CdcType::OnlyPK), - "n" => Ok(CdcType::Nothing), - "f" => Ok(CdcType::FullChanges), - typ => Err(PostgresSchemaError::UnsupportedReplicationType( - typ.to_string(), - )), - }?; - - let source_schema = SourceSchema::new(schema, cdc_type); - Self::validate_schema_replication_identity(table_name, &source_schema)?; - - Ok(source_schema) - } - - fn validate_schema_replication_identity( - table_name: &str, - schema: &SourceSchema, - ) -> Result<(), PostgresSchemaError> { - if schema.cdc_type == CdcType::OnlyPK && schema.schema.primary_index.is_empty() { - Err(PostgresSchemaError::PrimaryKeyIsMissingInSchema( - table_name.to_string(), - )) - } else { - Ok(()) - } - } - fn convert_row(&self, row: &Row) -> Result { - let schema: String = row.get(8); - let table_name: String = row.get(0); - let table_type: Option = row.get(7); - if let Some(typ) = table_type { - if typ != *"BASE TABLE" { - return Err(PostgresSchemaError::UnsupportedTableType(typ, table_name)); - } - } else { - return Err(PostgresSchemaError::TableTypeNotFound); - } - - let column_name: String = row.get(1); - let is_nullable: bool = row.get(2); - let is_column_used_in_index: bool = row.get(3); - let replication_type_int: i8 = row.get(5); - let type_oid: u32 = row.get(6); - - // TODO: workaround - in case of custom enum - let typ = if type_oid == 28862 { - FieldType::String - } else { - let oid_typ = Type::from_oid(type_oid); - oid_typ.map_or_else( - || Err(PostgresSchemaError::InvalidColumnType(column_name.clone())), - postgres_type_to_dozer_type, - )? - }; - - let replication_type = - String::from_utf8(vec![replication_type_int as u8]).map_err(|_e| { - PostgresSchemaError::ValueConversionError("Replication type".to_string()) - })?; - - Ok(PostgresTableRow { - schema, - table_name, - field: FieldDefinition::new(column_name, typ, is_nullable, SourceDefinition::Dynamic), - is_column_used_in_index, - replication_type, - }) - } -} - -pub const DEFAULT_SCHEMA_NAME: &str = "public"; - -fn find_table( - table_columns_map: &HashMap)>, - schema_name: Option<&str>, - table_name: &str, -) -> Result { - let schema_name = schema_name.unwrap_or(DEFAULT_SCHEMA_NAME); - let schema_table_identifier = (schema_name.to_string(), table_name.to_string()); - if let Some((relation_id, columns)) = table_columns_map.get(&schema_table_identifier) { - Ok(PostgresTableInfo { - schema: schema_table_identifier.0, - name: schema_table_identifier.1, - relation_id: *relation_id, - columns: columns.clone(), - }) - } else { - Err(PostgresConnectorError::TablesNotFound(vec![ - schema_table_identifier, - ])) - } -} - -const SQL: &str = " -SELECT table_info.table_name, - table_info.column_name, - CASE WHEN table_info.is_nullable = 'NO' THEN false ELSE true END AS is_nullable, - CASE - WHEN pc.relreplident = 'd' OR pc.relreplident = 'i' - THEN pa.attrelid IS NOT NULL - WHEN pc.relreplident = 'n' THEN false - WHEN pc.relreplident = 'f' THEN true - ELSE false - END AS is_column_used_in_index, - pc.oid, - pc.relreplident, - pt.oid AS type_oid, - t.table_type, - t.table_schema -FROM information_schema.columns table_info - LEFT JOIN information_schema.tables t ON t.table_name = table_info.table_name AND t.table_schema = table_info.table_schema - LEFT JOIN pg_namespace ns ON t.table_schema = ns.nspname - LEFT JOIN pg_class pc ON t.table_name = pc.relname AND ns.oid = pc.relnamespace - LEFT JOIN pg_type pt ON table_info.udt_name = pt.typname - LEFT JOIN pg_index pi ON pc.oid = pi.indrelid AND - ((pi.indisreplident = true AND pc.relreplident = 'i') OR (pi.indisprimary AND pc.relreplident = 'd')) - LEFT JOIN pg_attribute pa ON - pa.attrelid = pi.indrelid - AND pa.attnum = ANY (pi.indkey) - AND pa.attnum > 0 - AND pa.attname = table_info.column_name -WHERE :tables_name_condition AND ns.nspname not in ('information_schema', 'pg_catalog') - and ns.nspname not like 'pg_toast%' - and ns.nspname not like 'pg_temp_%' -ORDER BY table_info.table_schema, - table_info.table_catalog, - table_info.table_name, - table_info.ordinal_position;"; diff --git a/dozer-ingestion/postgres/src/schema/mod.rs b/dozer-ingestion/postgres/src/schema/mod.rs deleted file mode 100644 index 9899524721..0000000000 --- a/dozer-ingestion/postgres/src/schema/mod.rs +++ /dev/null @@ -1,5 +0,0 @@ -pub mod helper; -mod sorter; - -#[cfg(test)] -mod tests; diff --git a/dozer-ingestion/postgres/src/schema/sorter.rs b/dozer-ingestion/postgres/src/schema/sorter.rs deleted file mode 100644 index 9bd1a36f3e..0000000000 --- a/dozer-ingestion/postgres/src/schema/sorter.rs +++ /dev/null @@ -1,365 +0,0 @@ -use std::collections::HashMap; - -use dozer_ingestion_connector::{dozer_types::types::FieldDefinition, utils::ListOrFilterColumns}; - -use crate::PostgresSchemaError; - -use super::helper::{PostgresTable, SchemaTableIdentifier, DEFAULT_SCHEMA_NAME}; - -pub type PostgresTableResult = Result; - -pub fn sort_schemas( - expected_tables_order: &[ListOrFilterColumns], - mut mapped_tables: HashMap, -) -> Result, PostgresSchemaError> { - let mut sorted_tables: Vec<(SchemaTableIdentifier, PostgresTableResult)> = Vec::new(); - - for table in expected_tables_order.iter() { - let table_identifier = ( - table - .schema - .clone() - .unwrap_or(DEFAULT_SCHEMA_NAME.to_string()), - table.name.clone(), - ); - - let postgres_table_result = mapped_tables - .remove(&table_identifier) - .ok_or(PostgresSchemaError::ColumnNotFound)?; - - let sorted_table = match postgres_table_result { - Ok(postgres_table) => table.columns.as_ref().map_or_else( - || Ok::(postgres_table.clone()), - |expected_order| { - if expected_order.is_empty() { - Ok(postgres_table.clone()) - } else { - match sort_fields(&postgres_table, expected_order) { - Ok(sorted_fields) => { - let mut new_table = - PostgresTable::new(postgres_table.replication_type().clone()); - sorted_fields.into_iter().for_each(|(f, is_index_field)| { - new_table.add_field(f.clone(), is_index_field) - }); - Ok(new_table) - } - Err(e) => Err(e), - } - } - }, - ), - Err(e) => Err(e), - }; - - sorted_tables.push((table_identifier, sorted_table)) - } - - Ok(sorted_tables) -} - -fn sort_fields( - postgres_table: &PostgresTable, - expected_order: &[String], -) -> Result, PostgresSchemaError> { - let mut sorted_fields = Vec::new(); - - for c in expected_order { - let current_index = postgres_table - .fields() - .iter() - .position(|f| c == &f.name) - .ok_or(PostgresSchemaError::ColumnNotFound)?; - - let field = postgres_table - .get_field(current_index) - .ok_or(PostgresSchemaError::ColumnNotFound)?; - let is_index_field = postgres_table - .is_index_field(current_index) - .ok_or(PostgresSchemaError::ColumnNotFound)?; - - sorted_fields.push((field.clone(), *is_index_field)); - } - - Ok(sorted_fields) -} - -#[cfg(test)] -mod tests { - use dozer_ingestion_connector::{ - dozer_types::types::{FieldDefinition, FieldType, SourceDefinition}, - utils::ListOrFilterColumns, - }; - use std::collections::HashMap; - - use crate::schema::{ - helper::PostgresTable, - sorter::{sort_fields, sort_schemas}, - }; - - fn generate_postgres_table() -> PostgresTable { - let mut postgres_table = PostgresTable::new("d".to_string()); - postgres_table.add_field( - FieldDefinition { - name: "second field".to_string(), - typ: FieldType::UInt, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - false, - ); - postgres_table.add_field( - FieldDefinition { - name: "first field".to_string(), - typ: FieldType::UInt, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - true, - ); - postgres_table.add_field( - FieldDefinition { - name: "third field".to_string(), - typ: FieldType::UInt, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - false, - ); - postgres_table - } - #[test] - fn test_fields_sort() { - let postgres_table = generate_postgres_table(); - - let expected_order = [ - "first field".to_string(), - "second field".to_string(), - "third field".to_string(), - ]; - - let result = sort_fields(&postgres_table, &expected_order).unwrap(); - assert_eq!(result.first().unwrap().0.name, "first field"); - assert_eq!(result.get(1).unwrap().0.name, "second field"); - assert_eq!(result.get(2).unwrap().0.name, "third field"); - - assert!(result.first().unwrap().1); - assert!(!result.get(1).unwrap().1); - assert!(!result.get(2).unwrap().1); - } - - #[test] - fn test_tables_sort_without_columns() { - let postgres_table = generate_postgres_table(); - let mut mapped_tables = HashMap::new(); - mapped_tables.insert( - ("public".to_string(), "sort_test".to_string()), - Ok(postgres_table.clone()), - ); - - let expected_table_order = &[ListOrFilterColumns { - name: "sort_test".to_string(), - schema: Some("public".to_string()), - columns: None, - }]; - - let result = sort_schemas(expected_table_order, mapped_tables).unwrap(); - let fields = result.first().unwrap().1.as_ref().unwrap().fields(); - assert_eq!( - fields.first().unwrap().name, - postgres_table.get_field(0).unwrap().name - ); - assert_eq!( - fields.get(1).unwrap().name, - postgres_table.get_field(1).unwrap().name - ); - assert_eq!( - fields.get(2).unwrap().name, - postgres_table.get_field(2).unwrap().name - ); - - assert_eq!( - result - .first() - .unwrap() - .1 - .as_ref() - .unwrap() - .is_index_field(0), - postgres_table.is_index_field(0) - ); - assert_eq!( - result - .first() - .unwrap() - .1 - .as_ref() - .unwrap() - .is_index_field(1), - postgres_table.is_index_field(1) - ); - assert_eq!( - result - .first() - .unwrap() - .1 - .as_ref() - .unwrap() - .is_index_field(2), - postgres_table.is_index_field(2) - ); - } - - #[test] - fn test_tables_sort_with_single_column() { - let postgres_table = generate_postgres_table(); - let mut mapped_tables = HashMap::new(); - mapped_tables.insert( - ("public".to_string(), "sort_test".to_string()), - Ok(postgres_table), - ); - - let columns_order = vec!["third field".to_string()]; - let expected_table_order = &[ListOrFilterColumns { - name: "sort_test".to_string(), - schema: Some("public".to_string()), - columns: Some(columns_order.clone()), - }]; - - let result = sort_schemas(expected_table_order, mapped_tables).unwrap(); - assert_eq!( - &result - .first() - .unwrap() - .1 - .as_ref() - .unwrap() - .fields() - .first() - .unwrap() - .name, - columns_order.first().unwrap() - ); - assert_eq!( - result.first().unwrap().1.as_ref().unwrap().fields().len(), - 1 - ); - } - - #[test] - fn test_tables_sort_with_multi_columns() { - let postgres_table = generate_postgres_table(); - let mut mapped_tables = HashMap::new(); - mapped_tables.insert( - ("public".to_string(), "sort_test".to_string()), - Ok(postgres_table), - ); - - let columns_order = vec![ - "first field".to_string(), - "second field".to_string(), - "third field".to_string(), - ]; - let expected_table_order = &[ListOrFilterColumns { - name: "sort_test".to_string(), - schema: Some("public".to_string()), - columns: Some(columns_order.clone()), - }]; - - let result = sort_schemas(expected_table_order, mapped_tables).unwrap(); - let fields = result.first().unwrap().1.as_ref().unwrap().fields(); - assert_eq!( - &fields.first().unwrap().name, - columns_order.first().unwrap() - ); - assert_eq!(&fields.get(1).unwrap().name, columns_order.get(1).unwrap()); - assert_eq!(&fields.get(2).unwrap().name, columns_order.get(2).unwrap()); - assert_eq!( - result.first().unwrap().1.as_ref().unwrap().fields().len(), - 3 - ); - } - - #[test] - fn test_tables_sort_with_multi_tables() { - let postgres_table_1 = generate_postgres_table(); - let postgres_table_2 = generate_postgres_table(); - let mut mapped_tables = HashMap::new(); - mapped_tables.insert( - ("public".to_string(), "sort_test_second".to_string()), - Ok(postgres_table_1), - ); - mapped_tables.insert( - ("public".to_string(), "sort_test_first".to_string()), - Ok(postgres_table_2), - ); - - let columns_order_1 = vec![ - "first field".to_string(), - "second field".to_string(), - "third field".to_string(), - ]; - let columns_order_2 = vec![ - "third field".to_string(), - "second field".to_string(), - "first field".to_string(), - ]; - let expected_table_order = &[ - ListOrFilterColumns { - name: "sort_test_first".to_string(), - schema: Some("public".to_string()), - columns: Some(columns_order_1.clone()), - }, - ListOrFilterColumns { - name: "sort_test_second".to_string(), - schema: Some("public".to_string()), - columns: Some(columns_order_2.clone()), - }, - ]; - - let result = sort_schemas(expected_table_order, mapped_tables).unwrap(); - let first_table_after_sort = result.first().unwrap(); - let second_table_after_sort = result.get(1).unwrap(); - - let first_table = first_table_after_sort.1.as_ref().unwrap().clone(); - let first_table_fields = first_table.fields(); - - let second_table = second_table_after_sort.1.as_ref().unwrap().clone(); - let second_table_fields = second_table.fields(); - - assert_eq!( - first_table_after_sort.0 .1, - expected_table_order.first().unwrap().name - ); - assert_eq!( - second_table_after_sort.0 .1, - expected_table_order.get(1).unwrap().name - ); - assert_eq!( - &first_table_fields.first().unwrap().name, - columns_order_1.first().unwrap() - ); - assert_eq!( - &first_table_fields.get(1).unwrap().name, - columns_order_1.get(1).unwrap() - ); - assert_eq!( - &first_table_fields.get(2).unwrap().name, - columns_order_1.get(2).unwrap() - ); - assert_eq!( - &second_table_fields.first().unwrap().name, - columns_order_2.first().unwrap() - ); - assert_eq!( - &second_table_fields.get(1).unwrap().name, - columns_order_2.get(1).unwrap() - ); - assert_eq!( - &second_table_fields.get(2).unwrap().name, - columns_order_2.get(2).unwrap() - ); - } -} diff --git a/dozer-ingestion/postgres/src/schema/tests.rs b/dozer-ingestion/postgres/src/schema/tests.rs deleted file mode 100644 index 1a31816b2d..0000000000 --- a/dozer-ingestion/postgres/src/schema/tests.rs +++ /dev/null @@ -1,176 +0,0 @@ -use dozer_ingestion_connector::tokio; -use dozer_ingestion_connector::utils::ListOrFilterColumns; -use rand::Rng; -use serial_test::serial; -use std::collections::HashSet; - -use crate::schema::helper::SchemaHelper; -use crate::test_utils::load_test_connection_config; -use crate::tests::client::TestPostgresClient; -use crate::{PostgresConnectorError, PostgresSchemaError}; - -macro_rules! assert_vec_eq { - ($a:expr, $b:expr) => {{ - let a: HashSet<_> = $a.iter().cloned().collect(); - let b: HashSet<_> = $b.iter().cloned().collect(); - - assert_eq!(a, b) - }}; -} - -#[tokio::test] -#[ignore] -#[serial] -async fn test_connector_get_tables() { - let config = load_test_connection_config().await; - let mut client = TestPostgresClient::new(&config).await; - - let mut rng = rand::thread_rng(); - - let schema = format!("schema_helper_test_{}", rng.gen::()); - let table_name = format!("products_test_{}", rng.gen::()); - - client.create_schema(&schema).await; - client.create_simple_table(&schema, &table_name).await; - - let schema_helper = SchemaHelper::new(client.postgres_config.clone(), None); - let result = schema_helper.get_tables(None).await.unwrap(); - - let table = result.first().unwrap(); - assert_eq!(table_name, table.name); - assert_vec_eq!( - &[ - "name".to_string(), - "description".to_string(), - "weight_single".to_string(), - "weight_double".to_string(), - "id".to_string(), - "index2".to_string(), - "index4".to_string(), - "index8".to_string(), - ], - &table.columns - ); - - client.drop_schema(&schema).await; -} - -#[tokio::test] -#[ignore] -#[serial] -async fn test_connector_get_schema_with_selected_columns() { - let config = load_test_connection_config().await; - let mut client = TestPostgresClient::new(&config).await; - - let mut rng = rand::thread_rng(); - - let schema = format!("schema_helper_test_{}", rng.gen::()); - let table_name = format!("products_test_{}", rng.gen::()); - - client.create_schema(&schema).await; - client.create_simple_table(&schema, &table_name).await; - - let schema_helper = SchemaHelper::new(client.postgres_config.clone(), None); - let table_info = ListOrFilterColumns { - schema: Some(schema.clone()), - name: table_name.clone(), - columns: Some(vec!["name".to_string(), "id".to_string()]), - }; - let result = schema_helper.get_tables(Some(&[table_info])).await.unwrap(); - - let table = result.first().unwrap(); - assert_eq!(table_name, table.name); - assert_vec_eq!(&["name".to_string(), "id".to_string()], &table.columns); - - client.drop_schema(&schema).await; -} - -#[tokio::test] -#[ignore] -#[serial] -async fn test_connector_get_schema_without_selected_columns() { - let config = load_test_connection_config().await; - let mut client = TestPostgresClient::new(&config).await; - - let mut rng = rand::thread_rng(); - - let schema = format!("schema_helper_test_{}", rng.gen::()); - let table_name = format!("products_test_{}", rng.gen::()); - - client.create_schema(&schema).await; - client.create_simple_table(&schema, &table_name).await; - - let schema_helper = SchemaHelper::new(client.postgres_config.clone(), None); - let table_info = ListOrFilterColumns { - name: table_name.clone(), - schema: Some(schema.clone()), - columns: Some(vec![]), - }; - let result = schema_helper.get_tables(Some(&[table_info])).await.unwrap(); - - let table = result.first().unwrap(); - assert_eq!(table_name, table.name.clone()); - assert_vec_eq!( - &[ - "id".to_string(), - "name".to_string(), - "description".to_string(), - "weight_single".to_string(), - "weight_double".to_string(), - "index2".to_string(), - "index4".to_string(), - "index8".to_string(), - ], - &table.columns - ); - - client.drop_schema(&schema).await; -} - -#[tokio::test] -#[ignore] -#[serial] -async fn test_connector_view_cannot_be_used() { - let config = load_test_connection_config().await; - let mut client = TestPostgresClient::new(&config).await; - - let mut rng = rand::thread_rng(); - - let schema = format!("schema_helper_test_{}", rng.gen::()); - let table_name = format!("products_test_{}", rng.gen::()); - let view_name = format!("products_view_test_{}", rng.gen::()); - - client.create_schema(&schema).await; - client.create_simple_table(&schema, &table_name).await; - client.create_view(&schema, &table_name, &view_name).await; - - let schema_helper = SchemaHelper::new(client.postgres_config.clone(), None); - let table_info = ListOrFilterColumns { - name: view_name, - schema: Some(schema.clone()), - columns: Some(vec![]), - }; - - let result = schema_helper.get_schemas(&[table_info]).await; - assert!( - result.as_ref().unwrap().first().unwrap().is_err(), - "Result is not an error. Result: {:?}", - result - ); - assert!(matches!( - result.unwrap().first().unwrap(), - Err(PostgresConnectorError::PostgresSchemaError( - PostgresSchemaError::UnsupportedTableType(_, _) - )) - )); - - let table_info = ListOrFilterColumns { - name: table_name, - schema: Some(schema.clone()), - columns: Some(vec![]), - }; - let result = schema_helper.get_schemas(&[table_info]).await; - assert!(result.unwrap().first().unwrap().is_ok()); - - client.drop_schema(&schema).await; -} diff --git a/dozer-ingestion/postgres/src/snapshotter.rs b/dozer-ingestion/postgres/src/snapshotter.rs deleted file mode 100644 index 2141e06d0e..0000000000 --- a/dozer-ingestion/postgres/src/snapshotter.rs +++ /dev/null @@ -1,325 +0,0 @@ -use dozer_ingestion_connector::{ - dozer_types::{ - models::ingestion_types::{IngestionMessage, TransactionInfo}, - types::{Operation, Schema}, - }, - futures::StreamExt, - tokio::{ - self, - sync::mpsc::{channel, Sender}, - task::JoinSet, - }, - utils::ListOrFilterColumns, - Ingestor, SourceSchema, -}; - -use crate::{ - connection::helper as connection_helper, helper::get_conversion_fn, - schema::helper::SchemaHelper, PostgresConnectorError, -}; - -use super::helper; - -pub struct PostgresSnapshotter<'a> { - pub conn_config: tokio_postgres::Config, - pub ingestor: &'a Ingestor, - pub schema: Option, - pub batch_size: usize, -} - -impl<'a> PostgresSnapshotter<'a> { - pub async fn get_tables( - &self, - tables: &[ListOrFilterColumns], - ) -> Result>, PostgresConnectorError> { - let helper = SchemaHelper::new(self.conn_config.clone(), self.schema.clone()); - helper.get_schemas(tables).await - } - - pub async fn sync_table( - schema: Schema, - schema_name: String, - table_name: String, - table_index: usize, - conn_config: tokio_postgres::Config, - batch_size: usize, - sender: Sender>, - ) -> Result<(), PostgresConnectorError> { - let mut client_plain = connection_helper::connect(conn_config).await?; - - let column_str: Vec = schema - .fields - .iter() - .map(|f| format!("\"{0}\"", f.name)) - .collect(); - - let column_str = column_str.join(","); - let query = format!(r#"select {column_str} from "{schema_name}"."{table_name}""#); - let stmt = client_plain - .prepare(&query) - .await - .map_err(PostgresConnectorError::InvalidQueryError)?; - let columns = stmt.columns(); - let conversions: Vec<_> = columns - .iter() - .map(|col| get_conversion_fn(col.type_())) - .collect::, _>>()?; - - let empty_vec: Vec = Vec::new(); - let row_stream = client_plain - .query_raw(query, empty_vec) - .await - .map_err(PostgresConnectorError::InvalidQueryError)?; - tokio::pin!(row_stream); - - let mut batch = Vec::with_capacity(batch_size); - - while let Some(msg) = row_stream.next().await { - match msg { - Ok(msg) => { - let record = helper::map_row_to_record(&msg, &conversions) - .map_err(PostgresConnectorError::PostgresSchemaError)?; - batch.push(record); - - if batch.len() == batch_size { - let Ok(_) = sender - .send(Ok((table_index, Operation::BatchInsert { new: batch }))) - .await - else { - // If we can't send, the parent task has quit. There is - // no use in going on, but if there was an error, it was - // handled by the parent. - return Ok(()); - }; - batch = Vec::with_capacity(batch_size); - } - } - Err(e) => return Err(PostgresConnectorError::SyncWithSnapshotError(e.to_string())), - } - } - - if !batch.is_empty() { - let _ = sender - .send(Ok((table_index, Operation::BatchInsert { new: batch }))) - .await; - } - - Ok(()) - } - - pub async fn sync_tables( - &self, - tables: &[ListOrFilterColumns], - ) -> Result<(), PostgresConnectorError> { - let schemas = self.get_tables(tables).await?; - - let (tx, mut rx) = channel(16); - - let mut joinset = JoinSet::new(); - for (table_index, (schema, table)) in schemas.into_iter().zip(tables).enumerate() { - let schema = schema?; - let schema = schema.schema; - let schema_name = table.schema.clone().unwrap_or("public".to_string()); - let table_name = table.name.clone(); - let conn_config = self.conn_config.clone(); - let batch_size = self.batch_size; - let sender = tx.clone(); - joinset.spawn(async move { - if let Err(e) = Self::sync_table( - schema, - schema_name, - table_name, - table_index, - conn_config, - batch_size, - sender.clone(), - ) - .await - { - sender.send(Err(e)).await.unwrap(); - } - }); - } - // Make sure the last sender is dropped so receiving on the channel doesn't - // deadlock - drop(tx); - - if self - .ingestor - .handle_message(IngestionMessage::TransactionInfo( - TransactionInfo::SnapshottingStarted, - )) - .await - .is_err() - { - // If receiving side is closed, we can stop - return Ok(()); - } - - while let Some(message) = rx.recv().await { - let (table_index, evt) = message?; - if self - .ingestor - .handle_message(IngestionMessage::OperationEvent { - table_index, - op: evt, - id: None, - }) - .await - .is_err() - { - // If receiving side is closed, we can stop - return Ok(()); - } - } - - if self - .ingestor - .handle_message(IngestionMessage::TransactionInfo( - TransactionInfo::SnapshottingDone { id: None }, - )) - .await - .is_err() - { - // If receiving side is closed, we can stop - return Ok(()); - } - - // All tasks in the joinset should have finished (because they have dropped their senders) - // Otherwise, they will be aborted when the joinset is dropped - Ok(()) - } -} - -#[cfg(test)] -mod tests { - use std::time::Duration; - - use dozer_ingestion_connector::{tokio, utils::ListOrFilterColumns, IngestionConfig, Ingestor}; - use rand::Rng; - use serial_test::serial; - - use crate::{ - connection::helper::map_connection_config, test_utils::load_test_connection_config, - tests::client::TestPostgresClient, - }; - - use super::PostgresSnapshotter; - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_snapshotter_sync_tables_successfully_1_requested_table() { - let config = load_test_connection_config().await; - - let mut test_client = TestPostgresClient::new(&config).await; - - let mut rng = rand::thread_rng(); - let table_name = format!("test_table_{}", rng.gen::()); - - test_client.create_simple_table("public", &table_name).await; - test_client.insert_rows(&table_name, 2, None).await; - - let conn_config = map_connection_config(&config).unwrap(); - - let input_tables = vec![ListOrFilterColumns { - name: table_name, - schema: Some("public".to_string()), - columns: None, - }]; - - let ingestion_config = IngestionConfig::default(); - let (ingestor, mut iterator) = Ingestor::initialize_channel(ingestion_config); - - let snapshotter = PostgresSnapshotter { - conn_config, - ingestor: &ingestor, - schema: None, - batch_size: 1000, - }; - - snapshotter.sync_tables(&input_tables).await.unwrap(); - - let mut i = 0; - while i < 2 { - if iterator - .next_timeout(Duration::from_secs(1)) - .await - .is_none() - { - panic!("Unexpected operation"); - } - i += 1; - } - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_snapshotter_sync_tables_successfully_not_match_table() { - let config = load_test_connection_config().await; - - let mut test_client = TestPostgresClient::new(&config).await; - - let mut rng = rand::thread_rng(); - let table_name = format!("test_table_{}", rng.gen::()); - - test_client.create_simple_table("public", &table_name).await; - test_client.insert_rows(&table_name, 2, None).await; - - let conn_config = map_connection_config(&config).unwrap(); - - let input_table_name = String::from("not_existing_table"); - let input_tables = vec![ListOrFilterColumns { - name: input_table_name, - schema: Some("public".to_string()), - columns: None, - }]; - - let ingestion_config = IngestionConfig::default(); - let (ingestor, mut _iterator) = Ingestor::initialize_channel(ingestion_config); - - let snapshotter = PostgresSnapshotter { - conn_config, - ingestor: &ingestor, - schema: None, - batch_size: 1000, - }; - - let actual = snapshotter.sync_tables(&input_tables).await; - - assert!(actual.is_err()); - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_snapshotter_sync_tables_successfully_table_not_exist() { - let config = load_test_connection_config().await; - - let mut rng = rand::thread_rng(); - let table_name = format!("test_table_{}", rng.gen::()); - - let conn_config = map_connection_config(&config).unwrap(); - - let input_tables = vec![ListOrFilterColumns { - name: table_name, - schema: Some("public".to_string()), - columns: None, - }]; - - let ingestion_config = IngestionConfig::default(); - let (ingestor, mut _iterator) = Ingestor::initialize_channel(ingestion_config); - - let snapshotter = PostgresSnapshotter { - conn_config, - ingestor: &ingestor, - schema: None, - batch_size: 1000, - }; - - let actual = snapshotter.sync_tables(&input_tables).await; - - assert!(actual.is_err()); - } -} diff --git a/dozer-ingestion/postgres/src/test_utils.rs b/dozer-ingestion/postgres/src/test_utils.rs deleted file mode 100644 index f8dba354ec..0000000000 --- a/dozer-ingestion/postgres/src/test_utils.rs +++ /dev/null @@ -1,72 +0,0 @@ -use dozer_ingestion_connector::dozer_types::models::connection::ConnectionConfig; -use postgres_types::PgLsn; -use std::error::Error; -use std::str::FromStr; -use tokio_postgres::{error::DbError, Error as PostgresError, SimpleQueryMessage}; - -use crate::connection::helper::map_connection_config; -use crate::replication_slot_helper::ReplicationSlotHelper; -use crate::tests::client::TestPostgresClient; - -use super::connection::client::Client; - -pub async fn create_slot(client_mut: &mut Client, slot_name: &str) -> PgLsn { - client_mut - .simple_query("BEGIN READ ONLY ISOLATION LEVEL REPEATABLE READ;") - .await - .unwrap(); - - let created_lsn = ReplicationSlotHelper::create_replication_slot(client_mut, slot_name) - .await - .unwrap() - .unwrap(); - client_mut.simple_query("COMMIT;").await.unwrap(); - - PgLsn::from_str(&created_lsn).unwrap() -} - -pub async fn retry_drop_active_slot( - e: PostgresError, - client_mut: &mut Client, - slot_name: &str, -) -> Result, PostgresError> { - match e.source() { - None => Err(e), - Some(err) => match err.downcast_ref::() { - Some(db_error) if db_error.code().code().eq("55006") => { - let err = db_error.to_string(); - let parts = err.rsplit_once(' ').unwrap(); - - client_mut - .simple_query(format!("select pg_terminate_backend('{}');", parts.1).as_ref()) - .await - .unwrap(); - - ReplicationSlotHelper::drop_replication_slot(client_mut, slot_name).await - } - _ => Err(e), - }, - } -} - -pub async fn load_test_connection_config() -> ConnectionConfig { - let config = dozer_ingestion_connector::test_util::load_test_connection_config(); - let postgres_config = map_connection_config(&config).unwrap(); - // We're going to drop `dozer_test` so connect to another database. - let mut connect_config = postgres_config.clone(); - connect_config.dbname("postgres"); - let mut client = TestPostgresClient::new_with_postgres_config(connect_config).await; - client - .execute_query(&format!( - "DROP DATABASE IF EXISTS {}", - postgres_config.get_dbname().unwrap() - )) - .await; - client - .execute_query(&format!( - "CREATE DATABASE {}", - postgres_config.get_dbname().unwrap() - )) - .await; - config -} diff --git a/dozer-ingestion/postgres/src/tests/client.rs b/dozer-ingestion/postgres/src/tests/client.rs deleted file mode 100644 index 0bfa88c3cc..0000000000 --- a/dozer-ingestion/postgres/src/tests/client.rs +++ /dev/null @@ -1,107 +0,0 @@ -use std::fmt::Write; - -use dozer_ingestion_connector::dozer_types::{ - models::connection::ConnectionConfig, rust_decimal::Decimal, -}; - -use crate::connection::{ - client::Client, - helper::{connect, map_connection_config}, -}; - -pub struct TestPostgresClient { - client: Client, - pub postgres_config: tokio_postgres::Config, -} - -impl TestPostgresClient { - pub async fn new(auth: &ConnectionConfig) -> Self { - let postgres_config = map_connection_config(auth).unwrap(); - - let client = connect(postgres_config.clone()).await.unwrap(); - - Self { - client, - postgres_config, - } - } - - pub async fn new_with_postgres_config(postgres_config: tokio_postgres::Config) -> Self { - let client = connect(postgres_config.clone()).await.unwrap(); - - Self { - client, - postgres_config, - } - } - - pub async fn execute_query(&mut self, query: &str) { - self.client.query(query, &[]).await.unwrap(); - } - - pub async fn create_simple_table(&mut self, schema: &str, table_name: &str) { - self.execute_query(&format!( - "CREATE TABLE {schema}.{table_name} - ( - id SERIAL PRIMARY KEY, - name VARCHAR(255) NOT NULL, - description VARCHAR(512), - weight_single REAL, - weight_double DOUBLE PRECISION, - index2 INT2, - index4 INT4, - index8 INT8 - );" - )) - .await; - } - - pub async fn create_view(&mut self, schema: &str, table_name: &str, view_name: &str) { - self.execute_query(&format!( - "CREATE VIEW {schema}.{view_name} AS - SELECT id, name - FROM {schema}.{table_name}" - )) - .await; - } - - pub async fn drop_schema(&mut self, schema: &str) { - self.execute_query(&format!("DROP SCHEMA IF EXISTS {schema} CASCADE")) - .await; - } - - pub async fn drop_table(&mut self, schema: &str, table_name: &str) { - self.execute_query(&format!("DROP TABLE IF EXISTS {schema}.{table_name}")) - .await; - } - - pub async fn create_schema(&mut self, schema: &str) { - self.drop_schema(schema).await; - self.execute_query(&format!("CREATE SCHEMA {schema}")).await; - } - - pub async fn insert_rows(&mut self, table_name: &str, count: u64, offset: Option) { - let offset = offset.map_or(0, |o| o); - let mut buf = String::new(); - for i in 0..count { - if i > 0 { - buf.write_str(",").unwrap(); - } - buf.write_fmt(format_args!( - "(\'Product {}\',\'Product {} description\',{}, {}, {}, {}, {})", - i + offset, - i + offset, - Decimal::new((i * 41) as i64, 2), - Decimal::new((i * 41) as i64, 2), - i + offset, - i + offset, - i + offset, - )) - .unwrap(); - } - - let query = format!("insert into {table_name}(name, description, weight_single, weight_double, index2, index4, index8) values {buf}",); - - self.execute_query(&query).await; - } -} diff --git a/dozer-ingestion/postgres/src/tests/continue_replication_tests.rs b/dozer-ingestion/postgres/src/tests/continue_replication_tests.rs deleted file mode 100644 index 6771f5fb0f..0000000000 --- a/dozer-ingestion/postgres/src/tests/continue_replication_tests.rs +++ /dev/null @@ -1,165 +0,0 @@ -#[cfg(test)] -mod tests { - use dozer_ingestion_connector::{tokio, TableIdentifier}; - // use crate::connectors::Connector; - // use crate::ingestion::IngestionConfig; - // use dozer_types::models::ingestion_types::IngestionMessage; - // use dozer_types::node::OpIdentifier; - use rand::Rng; - use serial_test::serial; - use tokio_postgres::config::ReplicationMode; - - use crate::{ - connection::helper::{self, map_connection_config}, - connector::{create_publication, PostgresConfig, PostgresConnector}, - replication_slot_helper::ReplicationSlotHelper, - test_utils::{create_slot, load_test_connection_config, retry_drop_active_slot}, - tests::client::TestPostgresClient, - }; - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_continue_replication() { - let config = load_test_connection_config().await; - let conn_config = map_connection_config(&config).unwrap(); - let postgres_config = PostgresConfig { - name: "test".to_string(), - config: conn_config.clone(), - schema: None, - batch_size: 1000, - }; - - let connector = PostgresConnector::new(postgres_config, None).unwrap(); - - // let result = connector.can_start_from((1, 0)).unwrap(); - // assert!(!result, "Cannot continue, because slot doesnt exist"); - - let mut replication_conn_config = conn_config; - replication_conn_config.replication_mode(ReplicationMode::Logical); - - // Creating publication - let client = helper::connect(replication_conn_config.clone()) - .await - .unwrap(); - create_publication(client, &connector.name, None) - .await - .unwrap(); - - // Creating slot - let mut client = helper::connect(replication_conn_config.clone()) - .await - .unwrap(); - let _parsed_lsn = create_slot(&mut client, &connector.slot_name).await; - - // let result = connector - // .can_start_from((u64::from(parsed_lsn), 0)) - // .unwrap(); - - ReplicationSlotHelper::drop_replication_slot(&mut client, &connector.slot_name) - .await - .unwrap(); - // assert!( - // result, - // "Replication slot is created and it should be possible to continue" - // ); - } - - #[tokio::test] - #[ignore] - #[serial] - async fn test_connector_continue_replication_from_lsn() { - let config = load_test_connection_config().await; - - let mut test_client = TestPostgresClient::new(&config).await; - let mut rng = rand::thread_rng(); - let table_name = format!("test_table_{}", rng.gen::()); - let connector_name = format!("pg_connector_{}", rng.gen::()); - test_client.create_simple_table("public", &table_name).await; - - let conn_config = map_connection_config(&config).unwrap(); - let postgres_config = PostgresConfig { - name: connector_name, - config: conn_config.clone(), - schema: None, - batch_size: 1000, - }; - - let connector = PostgresConnector::new(postgres_config, None).unwrap(); - - let mut replication_conn_config = conn_config; - replication_conn_config.replication_mode(ReplicationMode::Logical); - - // Creating publication - let client = helper::connect(replication_conn_config.clone()) - .await - .unwrap(); - let table_identifier = TableIdentifier { - schema: Some("public".to_string()), - name: table_name.clone(), - }; - create_publication(client, &connector.name, Some(&[table_identifier])) - .await - .unwrap(); - - // Creating slot - let mut client = helper::connect(replication_conn_config.clone()) - .await - .unwrap(); - - let _parsed_lsn = create_slot(&mut client, &connector.slot_name).await; - - // let config = IngestionConfig::default(); - // let (ingestor, mut iterator) = Ingestor::initialize_channel(config); - - test_client.insert_rows(&table_name, 4, None).await; - - // assume that we already received two rows - // let last_parsed_position = 2_u64; - // thread::spawn(move || { - // let connector = PostgresConnector::new(postgres_config); - // let _ = connector.start( - // Some((u64::from(parsed_lsn), last_parsed_position)), - // &ingestor, - // tables, - // ); - // }); - - // let mut i = last_parsed_position; - // while i < 4 { - // i += 1; - // if let Some(IngestionMessage { - // identifier: OpIdentifier { seq_in_tx, .. }, - // .. - // }) = iterator.next() - // { - // assert_eq!(i, seq_in_tx); - // } else { - // panic!("Unexpected operation"); - // } - // } - - // test_client.insert_rows(&table_name, 3, None); - // let mut i = 0; - // while i < 3 { - // i += 1; - // if let Some(IngestionMessage { - // identifier: OpIdentifier { seq_in_tx, .. }, - // .. - // }) = iterator.next() - // { - // assert_eq!(i, seq_in_tx); - // } else { - // panic!("Unexpected operation"); - // } - // } - - if let Err(e) = - ReplicationSlotHelper::drop_replication_slot(&mut client, &connector.slot_name).await - { - retry_drop_active_slot(e, &mut client, &connector.slot_name) - .await - .unwrap(); - } - } -} diff --git a/dozer-ingestion/postgres/src/tests/dozer-config.yaml b/dozer-ingestion/postgres/src/tests/dozer-config.yaml deleted file mode 100644 index ac138fd0a8..0000000000 --- a/dozer-ingestion/postgres/src/tests/dozer-config.yaml +++ /dev/null @@ -1,49 +0,0 @@ -app_name: 1-hypercharge-postgres-sample -version: 1 -connections: - - config: !Postgres - user: postgres - password: postgres - host: localhost - port: 5434 - database: dozer_test - name: stocks -sources: - - name: stocks - table_name: stocks - columns: - - id - - ticker - - date - - open - - high - - low - - close - - adj_close - - volume - connection: stocks - - name: stocks_meta - table_name: stocks_meta - columns: - - nasdaq_traded - - symbol - - security_name - - listing_exchange - - market_category - - etf - - round_lot_size - - test_issue - - financial_status - - cqs_symbol - - nasdaq_symbol - - next_shares - connection: stocks -sinks: - # Direct from source - - name: stocks - config: !Dummy - table_name: stocks - # Direct from source - - name: stocks_meta - config: !Dummy - table_name: stocks_meta diff --git a/dozer-ingestion/postgres/src/tests/e2e.rs b/dozer-ingestion/postgres/src/tests/e2e.rs deleted file mode 100644 index 4e378e717f..0000000000 --- a/dozer-ingestion/postgres/src/tests/e2e.rs +++ /dev/null @@ -1,48 +0,0 @@ -// use crate::connectors::postgres::tests::client::TestPostgresClient; -// use crate::test_util::load_config; -// use dozer_types::models::app_config::Config; -// -// use crate::connectors::postgres::test_utils::get_iterator; -// use dozer_types::serde_yaml; -// use dozer_types::types::{Field, Operation}; -// use rand::Rng; -// -// #[ignore] -// #[test] -// // fn connector_e2e_connect_postgres_stream() { -// fn connector_disabled_test_e2e_connect_postgres_stream() { -// let config = serde_yaml::from_str::(load_config("test.postgres.yaml")).unwrap(); -// let connection = config.connections.get(0).unwrap().clone(); -// let mut client = TestPostgresClient::new(&connection.config.to_owned().unwrap()); -// -// let mut rng = rand::thread_rng(); -// let table_name = format!("products_test_{}", rng.gen::()); -// -// client.create_simple_table("public", &table_name); -// -// let mut iterator = get_iterator(connection, table_name.clone()); -// -// client.insert_rows(&table_name, 10, None); -// -// let mut i = 1; -// while i < 10 { -// let op = iterator.next(); -// if let Some((_, Operation::Insert { new })) = op { -// assert_eq!(new.values.get(0).unwrap(), &Field::Int(i)); -// i += 1; -// } -// } -// client.insert_rows(&table_name, 10, None); -// -// while i < 20 { -// let op = iterator.next(); -// -// if let Some((_, Operation::Insert { new })) = op { -// assert_eq!(new.values.get(0).unwrap(), &Field::Int(i)); -// i += 1; -// } -// } -// -// client.drop_table("public", &table_name); -// assert_eq!(i, 20); -// } diff --git a/dozer-ingestion/postgres/src/tests/mod.rs b/dozer-ingestion/postgres/src/tests/mod.rs deleted file mode 100644 index 6ab3cb17a8..0000000000 --- a/dozer-ingestion/postgres/src/tests/mod.rs +++ /dev/null @@ -1,3 +0,0 @@ -pub mod client; -mod continue_replication_tests; -mod e2e; diff --git a/dozer-ingestion/postgres/src/xlog_mapper.rs b/dozer-ingestion/postgres/src/xlog_mapper.rs deleted file mode 100644 index 5996705270..0000000000 --- a/dozer-ingestion/postgres/src/xlog_mapper.rs +++ /dev/null @@ -1,271 +0,0 @@ -use dozer_ingestion_connector::dozer_types::types::{Field, Operation, Record}; -use postgres_protocol::message::backend::LogicalReplicationMessage::{ - Begin, Commit, Delete, Insert, Relation, Update, -}; -use postgres_protocol::message::backend::{ - LogicalReplicationMessage, RelationBody, ReplicaIdentity, TupleData, UpdateBody, XLogDataBody, -}; -use postgres_protocol::Lsn; -use postgres_types::Type; -use std::collections::hash_map::Entry; -use std::collections::HashMap; - -use crate::{ - helper::{self, postgres_type_to_dozer_type}, - PostgresConnectorError, PostgresSchemaError, -}; - -#[derive(Debug)] -pub struct Table { - columns: Vec, - replica_identity: ReplicaIdentity, -} - -#[derive(Debug)] -pub struct TableColumn { - pub name: String, - pub flags: i8, - pub r#type: Type, - pub column_index: usize, -} - -#[derive(Debug, Clone)] -pub enum MappedReplicationMessage { - Begin, - Commit(Lsn), - Operation { table_index: usize, op: Operation }, -} - -#[derive(Debug, Default)] -pub struct XlogMapper { - /// Relation id to table info from replication `Relation` message. - relations_map: HashMap, - /// Relation id to (table index, column names). - tables_columns: HashMap)>, -} - -impl XlogMapper { - pub fn new(tables_columns: HashMap)>) -> Self { - XlogMapper { - relations_map: HashMap::::new(), - tables_columns, - } - } - - pub fn handle_message( - &mut self, - message: XLogDataBody, - ) -> Result, PostgresConnectorError> { - match &message.data() { - Relation(relation) => { - self.ingest_schema(relation)?; - } - Commit(commit) => { - return Ok(Some(MappedReplicationMessage::Commit(commit.end_lsn()))); - } - Begin(_begin) => { - return Ok(Some(MappedReplicationMessage::Begin)); - } - Insert(insert) => { - let Some(table_columns) = self.tables_columns.get(&insert.rel_id()) else { - return Ok(None); - }; - let table_index = table_columns.0; - - let table = self.relations_map.get(&insert.rel_id()).unwrap(); - let new_values = insert.tuple().tuple_data(); - - let values = Self::convert_values_to_fields(table, new_values, false)?; - - let event = Operation::Insert { - new: Record::new(values), - }; - - return Ok(Some(MappedReplicationMessage::Operation { - table_index, - op: event, - })); - } - Update(update) => { - let Some(table_columns) = self.tables_columns.get(&update.rel_id()) else { - return Ok(None); - }; - let table_index = table_columns.0; - - let table = self.relations_map.get(&update.rel_id()).unwrap(); - let new_values = update.new_tuple().tuple_data(); - - let values = Self::convert_values_to_fields(table, new_values, false)?; - let old_values = Self::convert_old_value_to_fields(table, update)?; - - let event = Operation::Update { - old: Record::new(old_values), - new: Record::new(values), - }; - - return Ok(Some(MappedReplicationMessage::Operation { - table_index, - op: event, - })); - } - Delete(delete) => { - let Some(table_columns) = self.tables_columns.get(&delete.rel_id()) else { - return Ok(None); - }; - let table_index = table_columns.0; - - // TODO: Use only columns with .flags() = 0 - let table = self.relations_map.get(&delete.rel_id()).unwrap(); - let key_values = delete.key_tuple().unwrap().tuple_data(); - - let values = Self::convert_values_to_fields(table, key_values, true)?; - - let event = Operation::Delete { - old: Record::new(values), - }; - - return Ok(Some(MappedReplicationMessage::Operation { - table_index, - op: event, - })); - } - _ => {} - } - - Ok(None) - } - - fn ingest_schema(&mut self, relation: &RelationBody) -> Result<(), PostgresConnectorError> { - let rel_id = relation.rel_id(); - let Some((table_index, wanted_columns)) = self.tables_columns.get(&rel_id) else { - return Ok(()); - }; - - let mut columns = vec![]; - for (column_index, column) in relation.columns().iter().enumerate() { - let column_name = - column - .name() - .map_err(|_| PostgresConnectorError::NonUtf8ColumnName { - table_index: *table_index, - column_index, - })?; - - if !wanted_columns.is_empty() - && !wanted_columns - .iter() - .any(|column| column.as_str() == column_name) - { - continue; - } - - // TODO: workaround - in case of custom enum - let type_oid = column.type_id() as u32; - let typ = if type_oid == 28862 { - Type::VARCHAR - } else { - Type::from_oid(type_oid).ok_or_else(|| { - PostgresSchemaError::InvalidColumnType(column_name.to_string()) - })? - }; - - columns.push(TableColumn { - name: column_name.to_string(), - flags: column.flags(), - r#type: typ, - column_index, - }) - } - - columns.sort_by_cached_key(|column| { - wanted_columns - .iter() - .position(|wanted| wanted == &column.name) - // Unwrap is safe because we filtered on present keys above - .unwrap() - }); - - let replica_identity = match relation.replica_identity() { - ReplicaIdentity::Default => ReplicaIdentity::Default, - ReplicaIdentity::Nothing => ReplicaIdentity::Nothing, - ReplicaIdentity::Full => ReplicaIdentity::Full, - ReplicaIdentity::Index => ReplicaIdentity::Index, - }; - - let table = Table { - columns, - replica_identity, - }; - - for c in &table.columns { - postgres_type_to_dozer_type(c.r#type.clone())?; - } - - match self.relations_map.entry(rel_id) { - Entry::Occupied(mut entry) => { - // Check if type has changed. - for (existing_column, column) in entry.get().columns.iter().zip(&table.columns) { - if existing_column.r#type != column.r#type { - return Err(PostgresConnectorError::ColumnTypeChanged { - table_index: *table_index, - column_name: existing_column.name.clone(), - old_type: existing_column.r#type.clone(), - new_type: column.r#type.clone(), - }); - } - } - - entry.insert(table); - } - Entry::Vacant(entry) => { - entry.insert(table); - } - } - - Ok(()) - } - - fn convert_values_to_fields( - table: &Table, - new_values: &[TupleData], - only_key: bool, - ) -> Result, PostgresConnectorError> { - let mut values: Vec = vec![]; - - for column in &table.columns { - if column.flags == 1 || !only_key { - let value = new_values.get(column.column_index).unwrap(); - match value { - TupleData::Null => values.push( - helper::postgres_type_to_field(None, column) - .map_err(PostgresConnectorError::PostgresSchemaError)?, - ), - TupleData::UnchangedToast => {} - TupleData::Text(text) => values.push( - helper::postgres_type_to_field(Some(text), column) - .map_err(PostgresConnectorError::PostgresSchemaError)?, - ), - } - } else { - values.push(Field::Null); - } - } - - Ok(values) - } - - fn convert_old_value_to_fields( - table: &Table, - update: &UpdateBody, - ) -> Result, PostgresConnectorError> { - match table.replica_identity { - ReplicaIdentity::Default | ReplicaIdentity::Full | ReplicaIdentity::Index => { - update.key_tuple().map_or_else( - || Self::convert_values_to_fields(table, update.new_tuple().tuple_data(), true), - |key_tuple| Self::convert_values_to_fields(table, key_tuple.tuple_data(), true), - ) - } - ReplicaIdentity::Nothing => Ok(vec![]), - } - } -} diff --git a/dozer-ingestion/snowflake/Cargo.toml b/dozer-ingestion/snowflake/Cargo.toml deleted file mode 100644 index 54c9e508a7..0000000000 --- a/dozer-ingestion/snowflake/Cargo.toml +++ /dev/null @@ -1,16 +0,0 @@ -[package] -name = "dozer-ingestion-snowflake" -version = "0.4.0" -edition = "2021" -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-ingestion-connector = { path = "../connector" } -odbc = "0.17.0" -include_dir = "0.7.3" -genawaiter = "0.99.1" -memchr = "2.6.4" -rand = "0.8.5" -base64 = "0.21.0" diff --git a/dozer-ingestion/snowflake/src/README.md b/dozer-ingestion/snowflake/src/README.md deleted file mode 100644 index 150009dc65..0000000000 --- a/dozer-ingestion/snowflake/src/README.md +++ /dev/null @@ -1,55 +0,0 @@ -# Snowflake - -## Prerequisites -This module allows developer to connect snowflake database via ODBC protocol. -This connector requires the installation of odbc and snowflake odbc driver. - - -## Installation guide -https://docs.snowflake.com/en/user-guide/odbc.html - -## Benchmarking / Testing - -For benchmarking and testing we use customer table from snowflakes samples TPCH_SF10 database. -Datasets can be found in https://docs.snowflake.com/en/user-guide/sample-data.html - -```sql -DROP TABLE CUSTOMER; - -CREATE TABLE CUSTOMER LIKE SNOWFLAKE_SAMPLE_DATA.TPCH_SF10.CUSTOMER; - -INSERT INTO CUSTOMER SELECT * FROM "SNOWFLAKE_SAMPLE_DATA"."TPCH_SF10"."CUSTOMER"; -``` - -## Flow of data -![Flow](flow.png) - -``` -st=>start: Start -cond=>condition: Is stream table created? - cond_snapshot=>condition: Is snapshot stream table created? - create_snapshot_table=>operation: Create snapshot table (SHOW_INITIAL_ROWS = TRUE;) -fetch_data=>operation: Fetch data from snapshot table -create_stream=>operation: Create stream table (offset from snapshot stream table) -consume_stream=>operation: Fetch data from stream table -temp_table_condition=>condition: Is temp table created? - create_temp_table=>operation: Create temp table (copy everything from stream table) -fetch_temp_data=>operation: Fetch data from temp table -delete_temp_table=>operation: Delete temp table - -st->cond -cond(yes)->temp_table_condition -cond(no)->cond_snapshot -cond_snapshot(yes)->fetch_data->create_stream->temp_table_condition -cond_snapshot(no)->create_snapshot_table->fetch_data->create_stream->temp_table_condition -temp_table_condition(no)->create_temp_table->fetch_temp_data -temp_table_condition(yes)->fetch_temp_data->delete_temp_table->temp_table_condition -``` - -### Additional commands for M1 processor -``` -export LDFLAGS="-L/opt/homebrew/Cellar/unixodbc/2.3.11/lib" -export CPPFLAGS="-I/opt/homebrew/Cellar/unixodbc/2.3.11/include" - -export LIBRARY_PATH=$LIBRARY_PATH:$(brew --prefix)/lib:$(brew --prefix)/opt/odbc/lib -``` diff --git a/dozer-ingestion/snowflake/src/connection/client.rs b/dozer-ingestion/snowflake/src/connection/client.rs deleted file mode 100644 index a5d1c09962..0000000000 --- a/dozer-ingestion/snowflake/src/connection/client.rs +++ /dev/null @@ -1,687 +0,0 @@ -use dozer_ingestion_connector::{ - dozer_types::{ - chrono::{NaiveDate, NaiveDateTime, NaiveTime}, - indexmap::IndexMap, - log::debug, - models::ingestion_types::SnowflakeConfig, - models::sink_config::snowflake::ConnectionParameters as SnowflakeSinkConfig, - rust_decimal::Decimal, - types::*, - }, - CdcType, SourceSchema, -}; -use odbc::ffi::{SqlDataType, SQL_DATE_STRUCT, SQL_TIMESTAMP_STRUCT}; -use odbc::odbc_safe::{AutocommitOn, Odbc3}; -use odbc::{ColumnDescriptor, Cursor, DiagnosticRecord, Environment, Executed, HasResult}; -use rand::distributions::Alphanumeric; -use rand::Rng; -use std::fmt::Write; -use std::ops::Deref; -use std::{collections::HashMap, vec}; - -use crate::{ - schema_helper::SchemaHelper, SnowflakeError, SnowflakeSchemaError, SnowflakeStreamError, -}; - -use super::pool::{Conn, Pool}; -use super::{helpers::is_network_failure, params::OdbcValue}; - -fn convert_decimal(bytes: &[u8], scale: u16) -> Result { - let is_negative = bytes[bytes.len() - 4] == 255; - let mut multiplier: i64 = 1; - let mut result: i64 = 0; - let bytes: &[u8] = &bytes[4..11]; - bytes.iter().for_each(|w| { - let number = *w as i64; - result += number * multiplier; - multiplier *= 256; - }); - - if is_negative { - result = -result; - } - - Ok(Field::from( - Decimal::try_new(result, scale as u32) - .map_err(SnowflakeSchemaError::DecimalConvertError)?, - )) -} - -pub fn convert_data( - cursor: &mut Cursor, - i: u16, - column_descriptor: &ColumnDescriptor, -) -> Result { - match column_descriptor.data_type { - SqlDataType::SQL_CHAR | SqlDataType::SQL_VARCHAR => { - match cursor - .get_data::(i) - .map_err(|e| SnowflakeSchemaError::ValueConversionError(Box::new(e)))? - { - None => Ok(Field::Null), - Some(value) => Ok(Field::from(value)), - } - } - SqlDataType::SQL_DECIMAL - | SqlDataType::SQL_NUMERIC - | SqlDataType::SQL_INTEGER - | SqlDataType::SQL_SMALLINT => match column_descriptor.decimal_digits { - None => { - match cursor - .get_data::(i) - .map_err(|e| SnowflakeSchemaError::ValueConversionError(Box::new(e)))? - { - None => Ok(Field::Null), - Some(value) => Ok(Field::from(value)), - } - } - Some(digits) => { - match cursor - .get_data::<&[u8]>(i) - .map_err(|e| SnowflakeSchemaError::ValueConversionError(Box::new(e)))? - { - None => Ok(Field::Null), - Some(value) => convert_decimal(value, digits), - } - } - }, - SqlDataType::SQL_FLOAT | SqlDataType::SQL_REAL | SqlDataType::SQL_DOUBLE => { - match cursor - .get_data::(i) - .map_err(|e| SnowflakeSchemaError::ValueConversionError(Box::new(e)))? - { - None => Ok(Field::Null), - Some(value) => Ok(Field::from(value)), - } - } - SqlDataType::SQL_TIMESTAMP => { - match cursor - .get_data::(i) - .map_err(|e| SnowflakeSchemaError::ValueConversionError(Box::new(e)))? - { - None => Ok(Field::Null), - Some(value) => { - let date = NaiveDate::from_ymd_opt( - value.year as i32, - value.month as u32, - value.day as u32, - ) - .map_or_else(|| Err(SnowflakeSchemaError::InvalidDateError), Ok)?; - let time = NaiveTime::from_hms_nano_opt( - value.hour as u32, - value.minute as u32, - value.second as u32, - value.fraction, - ) - .map_or_else(|| Err(SnowflakeSchemaError::InvalidTimeError), Ok)?; - Ok(Field::from(NaiveDateTime::new(date, time))) - } - } - } - SqlDataType::SQL_DATE => { - match cursor - .get_data::(i) - .map_err(|e| SnowflakeSchemaError::ValueConversionError(Box::new(e)))? - { - None => Ok(Field::Null), - Some(value) => { - let date = NaiveDate::from_ymd_opt( - value.year as i32, - value.month as u32, - value.day as u32, - ) - .map_or_else(|| Err(SnowflakeSchemaError::InvalidDateError), Ok)?; - Ok(Field::from(date)) - } - } - } - SqlDataType::SQL_EXT_BIT => { - match cursor - .get_data::(i) - .map_err(|e| SnowflakeSchemaError::ValueConversionError(Box::new(e)))? - { - None => Ok(Field::Null), - Some(v) => Ok(Field::from(v)), - } - } - _ => Err(SnowflakeSchemaError::ColumnTypeNotSupported(format!( - "{:?}", - &column_descriptor.data_type - ))), - } -} - -pub struct ClientConfig { - pub server: String, - pub port: Option, - pub user: String, - pub password: String, - - pub role: Option, - pub driver: Option, - - pub warehouse: String, - - pub database: Option, - pub schema: Option, -} - -pub struct Client<'env> { - pool: Pool<'env>, - name: String, -} - -impl<'env> Client<'env> { - pub fn new(config: ClientConfig, env: &'env Environment) -> Self { - let mut conn_hashmap: HashMap = HashMap::new(); - let driver = match &config.driver { - None => "Snowflake".to_string(), - Some(driver) => driver.to_string(), - }; - - conn_hashmap.insert("Driver".to_string(), driver); - conn_hashmap.insert("Server".to_string(), config.server); - conn_hashmap.insert( - "Port".to_string(), - config.port.unwrap_or_else(|| "443".to_string()), - ); - conn_hashmap.insert("Uid".to_string(), config.user); - conn_hashmap.insert("Pwd".to_string(), config.password); - if let Some(schema) = config.schema { - conn_hashmap.insert("Schema".to_string(), schema); - } - conn_hashmap.insert("Warehouse".to_string(), config.warehouse); - if let Some(database) = config.database { - conn_hashmap.insert("Database".to_string(), database); - } - if let Some(role) = config.role { - conn_hashmap.insert("Role".to_string(), role); - } - - let mut parts = vec![]; - conn_hashmap.keys().for_each(|k| { - parts.push(format!("{}={}", k, conn_hashmap.get(k).unwrap())); - }); - - let conn_string = parts.join(";"); - - debug!("Snowflake conn string: {:?}", conn_string); - let name = rand::thread_rng() - .sample_iter(&Alphanumeric) - .take(7) - .map(char::from) - .collect(); - let pool = Pool::new(env, conn_string); - Self { pool, name } - } - - pub fn get_name(&self) -> String { - self.name.clone() - } - - pub fn exec(&self, query: &str) -> Result<(), SnowflakeError> { - exec_drop(&self.pool, query, &[]).map_err(SnowflakeError::QueryError) - } - - pub fn exec_with_params( - &self, - query: &str, - params: &[OdbcValue], - ) -> Result<(), SnowflakeError> { - exec_drop(&self.pool, query, params).map_err(SnowflakeError::QueryError) - } - - pub fn exec_stream_creation(&self, query: String) -> Result { - let result = exec_drop(&self.pool, &query, &[]); - result.map_or_else( - |e| { - if e.get_native_error() == 2203 { - Ok(false) - } else if e.get_native_error() == 707 { - Err(SnowflakeError::SnowflakeStreamError( - SnowflakeStreamError::TimeTravelNotAvailableError, - )) - } else { - Err(SnowflakeError::QueryError(e)) - } - }, - |_| Ok(true), - ) - } - - pub fn parse_stream_creation_error(e: Box) -> Result { - if e.get_native_error() == 2203 { - Ok(false) - } else { - Err(SnowflakeError::QueryError(e)) - } - } - - fn parse_not_exist_error(e: Box) -> Result { - if e.get_native_error() == 2003 { - Ok(false) - } else { - Err(SnowflakeError::QueryError(e)) - } - } - - pub fn stream_exist(&self, stream_name: &String) -> Result { - let query = format!("SHOW STREAMS LIKE '{stream_name}';"); - - exec_first_exists(&self.pool, &query).map_or_else(Self::parse_not_exist_error, Ok) - } - - pub fn drop_stream(&self, stream_name: &String) -> Result { - let query = format!("DROP STREAM IF EXISTS {stream_name}"); - - exec_first_exists(&self.pool, &query).map_or_else(Self::parse_not_exist_error, Ok) - } - - pub fn fetch(&self, query: String) -> Result, SnowflakeError> { - exec_iter(self.pool.clone(), query, vec![]) - } - - #[allow(clippy::type_complexity)] - pub fn fetch_tables( - &self, - tables_indexes: Option>, - keys: HashMap>, - schema_name: String, - ) -> Result>, SnowflakeError> { - let tables_condition = tables_indexes.as_ref().map_or("".to_string(), |tables| { - let mut buf = String::new(); - buf.write_str(" AND TABLE_NAME IN(").unwrap(); - for (idx, table_name) in tables.keys().enumerate() { - if idx > 0 { - buf.write_char(',').unwrap(); - } - buf.write_str(&format!("\'{}\'", table_name)).unwrap(); - } - buf.write_char(')').unwrap(); - buf - }); - - let query = format!( - "SELECT TABLE_SCHEMA, TABLE_NAME, COLUMN_NAME, DATA_TYPE, IS_NULLABLE, NUMERIC_SCALE - FROM INFORMATION_SCHEMA.COLUMNS - WHERE TABLE_SCHEMA = '{schema_name}' {tables_condition} - ORDER BY TABLE_NAME, ORDINAL_POSITION" - ); - - let results = exec_iter(self.pool.clone(), query, vec![])?; - let mut schemas: IndexMap)> = - IndexMap::new(); - for (idx, result) in results.enumerate() { - let row_data = result?; - let empty = "".to_string(); - let table_name = if let Field::String(table_name) = &row_data.get(1).unwrap() { - table_name - } else { - &empty - }; - let field_name = if let Field::String(field_name) = &row_data.get(2).unwrap() { - field_name - } else { - &empty - }; - let type_name = if let Field::String(type_name) = &row_data.get(3).unwrap() { - type_name - } else { - &empty - }; - let nullable = if let Field::String(b) = &row_data.get(4).unwrap() { - let is_nullable = b == "NO"; - if is_nullable { - &true - } else { - &false - } - } else { - &false - }; - let scale = if let Field::Int(scale) = &row_data.get(5).unwrap() { - Some(*scale) - } else { - None - }; - - let table_index = match &tables_indexes { - None => idx, - Some(indexes) => *indexes.get(table_name).unwrap_or(&idx), - }; - - match SchemaHelper::map_schema_type(type_name, scale) { - Ok(typ) => { - if let Ok(schema) = schemas - .entry(table_name.clone()) - .or_insert(( - table_index, - Ok(Schema { - fields: vec![], - primary_index: vec![], - }), - )) - .1 - .as_mut() - { - schema.fields.push(FieldDefinition { - name: field_name.clone(), - typ, - nullable: *nullable, - source: SourceDefinition::Dynamic, - description: None, - }); - } - } - Err(e) => { - schemas.insert(table_name.clone(), (table_index, Err(e))); - } - } - } - - schemas.sort_by(|_, a, _, b| a.0.cmp(&b.0)); - - Ok(schemas - .into_iter() - .map(|(name, (_, schema))| match schema { - Ok(mut schema) => { - let mut indexes = vec![]; - if let Some(columns) = keys.get(&name) { - schema.fields.iter().enumerate().for_each(|(idx, f)| { - if columns.contains(&f.name) { - indexes.push(idx); - } - }); - } - - let cdc_type = if indexes.is_empty() { - CdcType::Nothing - } else { - CdcType::FullChanges - }; - - schema.primary_index = indexes; - - Ok((name, SourceSchema::new(schema, cdc_type))) - } - Err(e) => Err(SnowflakeError::SnowflakeSchemaError(e)), - }) - .collect()) - } - - pub fn fetch_keys(&self) -> Result>, SnowflakeError> { - 'retry: loop { - let query = "SHOW PRIMARY KEYS IN SCHEMA".to_string(); - let results = exec_iter(self.pool.clone(), query, vec![])?; - let mut keys: HashMap> = HashMap::new(); - for result in results { - let row_data = match result { - Err(SnowflakeError::NonResumableQuery(_)) => continue 'retry, - result => result?, - }; - let empty = "".to_string(); - let table_name = row_data.get(3).map_or(empty.clone(), |v| match v { - Field::String(v) => v.clone(), - _ => empty.clone(), - }); - let column_name = row_data.get(4).map_or(empty.clone(), |v| match v { - Field::String(v) => v.clone(), - _ => empty.clone(), - }); - - keys.entry(table_name).or_default().push(column_name); - } - break Ok(keys); - } - } -} - -impl From for ClientConfig { - fn from( - SnowflakeConfig { - server, - port, - user, - password, - database, - schema, - warehouse, - driver, - role, - .. - }: SnowflakeConfig, - ) -> Self { - Self { - server, - port: Some(port), - user, - password, - role: Some(role), - driver, - warehouse, - database: Some(database), - schema: Some(schema), - } - } -} - -impl From for ClientConfig { - fn from( - SnowflakeSinkConfig { - server, - port, - user, - password, - driver, - role, - warehouse, - }: SnowflakeSinkConfig, - ) -> Self { - Self { - server, - port, - user, - password, - role, - driver, - warehouse, - database: None, - schema: None, - } - } -} - -macro_rules! retry { - ($operation:expr $(, $label:tt)? $(,)?) => { - match $operation { - Err(err) if is_network_failure(&err) => continue $($label)?, - result => result, - } - }; -} - -fn add_query_offset(query: &str, offset: u64) -> Result { - if offset == 0 { - Ok(query.into()) - } else { - let resumable = query - .trim_start() - .get(0..7) - .map(|s| s.to_uppercase() == "SELECT ") - .unwrap_or(false); - - if resumable { - Ok(format!( - "{query} LIMIT 18446744073709551615 OFFSET {offset}" - )) - } else { - Err(SnowflakeError::NonResumableQuery(query.to_string())) - } - } -} - -fn get_fields_from_cursor( - mut cursor: Cursor, - cols: i16, - schema: &[ColumnDescriptor], -) -> Result, SnowflakeError> { - let mut values = vec![]; - for i in 1..(cols + 1) { - let descriptor = schema.get((i - 1) as usize).unwrap(); - let value = convert_data(&mut cursor, i as u16, descriptor)?; - values.push(value); - } - - Ok(values) -} - -fn exec_drop(pool: &Pool, query: &str, params: &[OdbcValue]) -> Result<(), Box> { - let conn = pool.get_conn()?; - { - let _result = exec_helper(&conn, query, params)?; - } - conn.return_(); - Ok(()) -} - -fn exec_first_exists(pool: &Pool, query: &str) -> Result> { - loop { - let conn = pool.get_conn()?; - let result = match exec_helper(&conn, query, &[])? { - Some(mut data) => retry!(data.fetch())?.is_some(), - None => false, - }; - conn.return_(); - break Ok(result); - } -} - -fn exec_iter( - pool: Pool, - query: String, - params: Vec, -) -> Result { - use genawaiter::{ - rc::{gen, Gen}, - yield_, - }; - use ExecIterResult::*; - - let mut generator: Gen = gen!({ - let mut cursor_position = 0u64; - 'retry: loop { - let conn = pool.get_conn().map_err(SnowflakeError::QueryError)?; - { - let mut data = - match exec_helper(&conn, &add_query_offset(&query, cursor_position)?, ¶ms) - .map_err(SnowflakeError::QueryError)? - { - Some(data) => data, - None => break, - }; - let cols = data - .num_result_cols() - .map_err(|e| SnowflakeError::QueryError(e.into()))?; - let mut schema = Vec::new(); - for i in 1..(cols + 1) { - let value = i.try_into(); - let column_descriptor = match value { - Ok(v) => data - .describe_col(v) - .map_err(|e| SnowflakeError::QueryError(e.into()))?, - Err(e) => Err(SnowflakeSchemaError::SchemaConversionError(e))?, - }; - schema.push(column_descriptor) - } - yield_!(Schema(schema.clone())); - - while let Some(cursor) = - retry!(data.fetch(),'retry).map_err(|e| SnowflakeError::QueryError(e.into()))? - { - let fields = get_fields_from_cursor(cursor, cols, &schema)?; - yield_!(Row(fields)); - cursor_position += 1; - } - } - conn.return_(); - break; - } - Ok::<(), SnowflakeError>(()) - }); - - let mut iterator = std::iter::from_fn(move || { - use genawaiter::GeneratorState::*; - match generator.resume() { - Yielded(fields) => Some(Ok(fields)), - Complete(Err(err)) => Some(Err(err)), - Complete(Ok(())) => None, - } - }); - - let schema = match iterator.next() { - Some(Ok(Schema(schema))) => Some(schema), - Some(Err(err)) => Err(err)?, - None => None, - _ => unreachable!(), - }; - - Ok(ExecIter { - iterator: Box::new(iterator), - schema, - }) -} - -enum ExecIterResult { - Schema(Vec), - Row(Vec), -} - -pub struct ExecIter<'env> { - iterator: Box> + 'env>, - schema: Option>, -} - -impl<'env> ExecIter<'env> { - pub fn schema(&self) -> Option<&Vec> { - self.schema.as_ref() - } -} - -impl<'env> Iterator for ExecIter<'env> { - type Item = Result, SnowflakeError>; - - fn next(&mut self) -> Option { - use ExecIterResult::*; - loop { - let result = match self.iterator.next()? { - Ok(Schema(schema)) => { - self.schema = Some(schema); - continue; - } - Ok(Row(row)) => Ok(row), - Err(err) => Err(err), - }; - return Some(result); - } - } -} - -fn exec_helper<'a>( - conn: &'a Conn<'_>, - query: &str, - params: &'a [OdbcValue], -) -> Result>, Box> -{ - loop { - let mut statement = retry!(odbc::Statement::with_parent(conn.deref()))?; - for (i, param) in params.iter().enumerate() { - let parameter_index = i as u16 + 1; - statement = param.bind(statement, parameter_index)? - } - let result = retry!(statement.exec_direct(query))?; - break match result { - odbc::ResultSetState::Data(data) => Ok(Some(data)), - odbc::ResultSetState::NoData(_) => Ok(None), - }; - } -} diff --git a/dozer-ingestion/snowflake/src/connection/helpers.rs b/dozer-ingestion/snowflake/src/connection/helpers.rs deleted file mode 100644 index 3cf74ec66d..0000000000 --- a/dozer-ingestion/snowflake/src/connection/helpers.rs +++ /dev/null @@ -1,14 +0,0 @@ -pub fn is_network_failure(err: &odbc::DiagnosticRecord) -> bool { - // Reference for ODBC error codes: - // https://learn.microsoft.com/en-us/sql/odbc/reference/appendixes/appendix-a-odbc-error-codes?view=sql-server-ver16 - let sqlstate = &err.get_raw_state()[0..5]; - matches!( - sqlstate, - b"01002" /* Disconnect error */ | - b"08001" /* Client unable to establish connection */ | - b"08007" /* Connection failure during transaction */ | - b"08S01" /* Communication link failure */ | - b"HY000" /* General error */ - if memchr::memmem::find(err.get_raw_message(), b"Timeout was reached").is_some() - ) -} diff --git a/dozer-ingestion/snowflake/src/connection/mod.rs b/dozer-ingestion/snowflake/src/connection/mod.rs deleted file mode 100644 index 6a6edbbe76..0000000000 --- a/dozer-ingestion/snowflake/src/connection/mod.rs +++ /dev/null @@ -1,4 +0,0 @@ -pub mod client; -pub mod helpers; -pub mod params; -pub mod pool; diff --git a/dozer-ingestion/snowflake/src/connection/params.rs b/dozer-ingestion/snowflake/src/connection/params.rs deleted file mode 100644 index c1fb79349a..0000000000 --- a/dozer-ingestion/snowflake/src/connection/params.rs +++ /dev/null @@ -1,164 +0,0 @@ -use odbc::{odbc_safe::AutocommitMode, DiagnosticRecord, Statement}; - -#[derive(Debug, Clone)] -pub enum OdbcValue { - Binary(Vec), - String(String), - Decimal(String), - Timestamp(String), - Date(String), - U8(u8), - I8(i8), - I16(i16), - U16(u16), - I32(i32), - U32(u32), - I64(i64), - U64(u64), - F32(f32), - F64(f64), - Bool(bool), - Null, -} - -impl OdbcValue { - pub fn bind<'a, 'b, 'c, S, R, AC: AutocommitMode>( - &'c self, - statement: Statement<'a, 'b, S, R, AC>, - parameter_index: u16, - ) -> Result, Box> - where - 'b: 'c, - { - let statement = match self { - Self::Binary(value) => statement.bind_parameter(parameter_index, value)?, - Self::String(value) => statement.bind_parameter(parameter_index, value)?, - Self::Decimal(value) => statement.bind_parameter(parameter_index, value)?, - Self::Timestamp(value) => statement.bind_parameter(parameter_index, value)?, - Self::Date(value) => statement.bind_parameter(parameter_index, value)?, - Self::U8(value) => statement.bind_parameter(parameter_index, value)?, - Self::I8(value) => statement.bind_parameter(parameter_index, value)?, - Self::I16(value) => statement.bind_parameter(parameter_index, value)?, - Self::U16(value) => statement.bind_parameter(parameter_index, value)?, - Self::I32(value) => statement.bind_parameter(parameter_index, value)?, - Self::U32(value) => statement.bind_parameter(parameter_index, value)?, - Self::I64(value) => statement.bind_parameter(parameter_index, value)?, - Self::U64(value) => statement.bind_parameter(parameter_index, value)?, - Self::F32(value) => statement.bind_parameter(parameter_index, value)?, - Self::F64(value) => statement.bind_parameter(parameter_index, value)?, - Self::Bool(value) => statement.bind_parameter(parameter_index, value)?, - Self::Null => statement.bind_parameter(parameter_index, &None::)?, - }; - Ok(statement) - } - - pub fn as_sql(&self) -> String { - match self { - Self::Binary(b) => format!("TO_BINARY('{}', 'BASE64')", base64_encode(b)), - Self::String(s) => { - format!( - "TO_VARCHAR(TO_BINARY('{}', 'BASE64'), 'UTF-8')", - base64_encode(s.as_bytes()) - ) - } - Self::Decimal(s) => s.clone(), - Self::Timestamp(s) => format!("'{}'", s), - Self::Date(s) => format!("'{}'", s), - Self::U8(u) => u.to_string(), - Self::I8(i) => i.to_string(), - Self::I16(i) => i.to_string(), - Self::U16(u) => u.to_string(), - Self::I32(i) => i.to_string(), - Self::U32(u) => u.to_string(), - Self::I64(i) => i.to_string(), - Self::U64(u) => u.to_string(), - Self::F32(f) => f.to_string(), - Self::F64(f) => f.to_string(), - Self::Bool(b) => b.to_string(), - Self::Null => "null".to_string(), - } - } -} - -impl From> for OdbcValue { - fn from(value: Vec) -> Self { - Self::Binary(value) - } -} - -impl From for OdbcValue { - fn from(value: String) -> Self { - Self::String(value) - } -} - -impl From for OdbcValue { - fn from(value: u8) -> Self { - Self::U8(value) - } -} - -impl From for OdbcValue { - fn from(value: i8) -> Self { - Self::I8(value) - } -} - -impl From for OdbcValue { - fn from(value: i16) -> Self { - Self::I16(value) - } -} - -impl From for OdbcValue { - fn from(value: u16) -> Self { - Self::U16(value) - } -} - -impl From for OdbcValue { - fn from(value: i32) -> Self { - Self::I32(value) - } -} - -impl From for OdbcValue { - fn from(value: u32) -> Self { - Self::U32(value) - } -} - -impl From for OdbcValue { - fn from(value: i64) -> Self { - Self::I64(value) - } -} - -impl From for OdbcValue { - fn from(value: u64) -> Self { - Self::U64(value) - } -} - -impl From for OdbcValue { - fn from(value: f32) -> Self { - Self::F32(value) - } -} - -impl From for OdbcValue { - fn from(value: f64) -> Self { - Self::F64(value) - } -} - -impl From for OdbcValue { - fn from(value: bool) -> Self { - Self::Bool(value) - } -} - -fn base64_encode(bytes: &[u8]) -> String { - use base64::Engine; - base64::engine::general_purpose::STANDARD.encode(bytes) -} diff --git a/dozer-ingestion/snowflake/src/connection/pool.rs b/dozer-ingestion/snowflake/src/connection/pool.rs deleted file mode 100644 index 9e3d0600ba..0000000000 --- a/dozer-ingestion/snowflake/src/connection/pool.rs +++ /dev/null @@ -1,89 +0,0 @@ -use std::{ - cell::RefCell, - collections::LinkedList, - ops::{Deref, DerefMut}, - rc::Rc, -}; - -use dozer_ingestion_connector::{blocking_retry_on_network_failure, dozer_types}; -use odbc::{ - odbc_safe::{AutocommitOn, Odbc3}, - Connection, DiagnosticRecord, Environment, -}; - -use super::helpers::is_network_failure; - -const MAX_POOL_SIZE: usize = 16; - -#[derive(Debug, Clone)] -pub struct Pool<'env> { - inner: Rc>>, -} - -impl<'env> Pool<'env> { - pub fn new(env: &'env Environment, conn_string: String) -> Self { - Self { - inner: Rc::new(RefCell::new(Inner { - env, - conn_string, - connections: Default::default(), - })), - } - } - - pub fn get_conn(&self) -> Result, Box> { - let mut inner = self.inner.borrow_mut(); - let inner = inner.deref_mut(); - let conn = if let Some(conn) = inner.connections.pop_front() { - conn - } else { - blocking_retry_on_network_failure!( - "connect_with_connection_string", - inner.env.connect_with_connection_string(&inner.conn_string), - is_network_failure, - )? - }; - Ok(Conn { - pool: self.clone(), - inner: conn, - }) - } - - fn return_conn(&self, conn: Connection<'env, AutocommitOn>) { - let mut inner = self.inner.borrow_mut(); - let inner = inner.deref_mut(); - if inner.connections.len() < MAX_POOL_SIZE { - inner.connections.push_back(conn); - } - } -} - -#[derive(Debug)] -struct Inner<'env> { - env: &'env Environment, - conn_string: String, - connections: LinkedList>, -} - -#[derive(Debug)] -pub struct Conn<'env> { - pool: Pool<'env>, - inner: Connection<'env, AutocommitOn>, -} - -impl<'env> Conn<'env> { - /// Returns the connection to the pool. - /// Currently, connections have to be manually returned to the pool - /// because odbc::Connection does not know if it has been disconnected. - pub fn return_(self) { - self.pool.return_conn(self.inner) - } -} - -impl<'env> Deref for Conn<'env> { - type Target = Connection<'env, AutocommitOn>; - - fn deref(&self) -> &Self::Target { - &self.inner - } -} diff --git a/dozer-ingestion/snowflake/src/connector/mod.rs b/dozer-ingestion/snowflake/src/connector/mod.rs deleted file mode 100644 index 4c719cecdd..0000000000 --- a/dozer-ingestion/snowflake/src/connector/mod.rs +++ /dev/null @@ -1,2 +0,0 @@ -mod snowflake; -pub use snowflake::SnowflakeConnector; diff --git a/dozer-ingestion/snowflake/src/connector/snowflake.rs b/dozer-ingestion/snowflake/src/connector/snowflake.rs deleted file mode 100644 index 6b844568bf..0000000000 --- a/dozer-ingestion/snowflake/src/connector/snowflake.rs +++ /dev/null @@ -1,209 +0,0 @@ -use dozer_ingestion_connector::{ - async_trait, - dozer_types::{ - errors::internal::BoxedError, - log::{info, warn}, - models::ingestion_types::{default_snowflake_poll_interval, SnowflakeConfig}, - node::OpIdentifier, - types::FieldType, - }, - tokio, Connector, Ingestor, SourceSchema, SourceSchemaResult, TableIdentifier, TableInfo, -}; -use odbc::create_environment_v3; - -use crate::{ - connection::client::Client, schema_helper::SchemaHelper, stream_consumer::StreamConsumer, - SnowflakeError, SnowflakeStreamError, -}; - -#[derive(Debug)] -pub struct SnowflakeConnector { - name: String, - config: SnowflakeConfig, -} - -impl SnowflakeConnector { - pub fn new(name: String, config: SnowflakeConfig) -> Self { - Self { name, config } - } - - async fn get_schemas_async( - &self, - table_names: Option>, - ) -> Result>, SnowflakeError> { - let config = self.config.clone(); - spawn_blocking(move || SchemaHelper::get_schema(config, table_names.as_deref())).await - } -} - -#[async_trait] -impl Connector for SnowflakeConnector { - fn types_mapping() -> Vec<(String, Option)> - where - Self: Sized, - { - todo!() - } - - async fn validate_connection(&mut self) -> Result<(), BoxedError> { - self.get_schemas_async(None).await?; - Ok(()) - } - - async fn list_tables(&mut self) -> Result, BoxedError> { - let schemas = self.get_schemas_async(None).await?; - let mut tables = vec![]; - for schema in schemas { - tables.push(TableIdentifier::from_table_name(schema?.0)); - } - Ok(tables) - } - - async fn validate_tables(&mut self, tables: &[TableIdentifier]) -> Result<(), BoxedError> { - let table_names = tables - .iter() - .map(|table| table.name.clone()) - .collect::>(); - let schemas = self.get_schemas_async(Some(table_names)).await?; - for schema in schemas { - schema?; - } - Ok(()) - } - - async fn list_columns( - &mut self, - tables: Vec, - ) -> Result, BoxedError> { - let table_names = tables - .iter() - .map(|table| table.name.clone()) - .collect::>(); - let schemas = self.get_schemas_async(Some(table_names)).await?; - let mut result = vec![]; - for schema in schemas { - let (name, schema) = schema?; - let column_names = schema - .schema - .fields - .into_iter() - .map(|field| field.name) - .collect(); - result.push(TableInfo { - schema: None, - name, - column_names, - }); - } - Ok(result) - } - - async fn get_schemas( - &mut self, - table_infos: &[TableInfo], - ) -> Result, BoxedError> { - warn!("TODO: respect `column_names` in `table_infos`"); - let table_names = table_infos - .iter() - .map(|table_info| table_info.name.clone()) - .collect::>(); - Ok(self - .get_schemas_async(Some(table_names)) - .await? - .into_iter() - .map(|schema_result| schema_result.map(|(_, schema)| schema).map_err(Into::into)) - .collect()) - } - - async fn serialize_state(&self) -> Result, BoxedError> { - Ok(vec![]) - } - - async fn start( - &mut self, - ingestor: &Ingestor, - tables: Vec, - last_checkpoint: Option, - ) -> Result<(), BoxedError> { - spawn_blocking({ - let name = self.name.clone(); - let config = self.config.clone(); - let ingestor = ingestor.clone(); - move || run(name, config, tables, last_checkpoint, ingestor) - }) - .await - .map_err(Into::into) - } -} - -fn run( - name: String, - config: SnowflakeConfig, - tables: Vec, - last_checkpoint: Option, - ingestor: Ingestor, -) -> Result<(), SnowflakeError> { - // SNAPSHOT part - run it when stream table doesn't exist - let env = create_environment_v3().unwrap(); - let interval = config - .poll_interval_seconds - .unwrap_or_else(default_snowflake_poll_interval); - let stream_client = Client::new(config.into(), &env); - - let mut consumer = StreamConsumer::new(); - let mut iteration = 0; - loop { - for (idx, table) in tables.iter().enumerate() { - // We only check stream status on first iteration - if iteration == 0 { - let state = - last_checkpoint.map(|checkpoint| (checkpoint.txid, checkpoint.seq_in_tx)); - match state { - None | Some((0, _)) => { - info!("[{}][{}] Creating new stream", name, table.name); - StreamConsumer::drop_stream(&stream_client, &table.name)?; - StreamConsumer::create_stream(&stream_client, &table.name)?; - } - Some((iteration, index)) => { - info!( - "[{}][{}] Continuing ingestion from {}/{}", - name, table.name, iteration, index - ); - if let Ok(false) = - StreamConsumer::is_stream_created(&stream_client, &table.name) - { - return Err(SnowflakeError::SnowflakeStreamError( - SnowflakeStreamError::StreamNotFound, - )); - } - } - } - } - - info!("[{}][{}] Reading from changes stream", name, table.name); - - consumer.consume_stream(&stream_client, &table.name, &ingestor, idx, iteration)?; - - std::thread::sleep(interval); - } - - iteration += 1; - } -} - -async fn spawn_blocking(f: F) -> T -where - F: FnOnce() -> T + Send + 'static, - T: Send + 'static, -{ - tokio::task::spawn_blocking(f) - .await - .unwrap_or_else(|join_err| { - let msg = format!("{join_err}"); - if join_err.is_panic() { - panic!("{msg}; panic: {:?}", join_err.into_panic()) - } else { - panic!("{msg}") - } - }) -} diff --git a/dozer-ingestion/snowflake/src/flow.png b/dozer-ingestion/snowflake/src/flow.png deleted file mode 100644 index 46d3fb4899..0000000000 Binary files a/dozer-ingestion/snowflake/src/flow.png and /dev/null differ diff --git a/dozer-ingestion/snowflake/src/lib.rs b/dozer-ingestion/snowflake/src/lib.rs deleted file mode 100644 index 4a449e05c0..0000000000 --- a/dozer-ingestion/snowflake/src/lib.rs +++ /dev/null @@ -1,73 +0,0 @@ -use std::num::TryFromIntError; - -use dozer_ingestion_connector::dozer_types::{ - rust_decimal, - thiserror::{self, Error}, -}; -use odbc::DiagnosticRecord; - -pub mod connection; -pub mod connector; -mod schema_helper; -pub mod stream_consumer; -pub mod test_utils; - -#[cfg(test)] -mod tests; - -#[derive(Error, Debug)] -pub enum SnowflakeError { - #[error("Failed to parse checkpoint state")] - CorruptedState, - - #[error("Snowflake query error")] - QueryError(#[source] Box), - - #[error("Snowflake connection error")] - ConnectionError(#[source] Box), - - #[error(transparent)] - SnowflakeSchemaError(#[from] SnowflakeSchemaError), - - #[error(transparent)] - SnowflakeStreamError(#[from] SnowflakeStreamError), - - #[error("A network error occurred, but this query is not resumable. query: {0}")] - NonResumableQuery(String), -} - -#[derive(Error, Debug)] -pub enum SnowflakeSchemaError { - #[error("Column type {0} not supported")] - ColumnTypeNotSupported(String), - - #[error("Value conversion Error")] - ValueConversionError(#[source] Box), - - #[error("Invalid date")] - InvalidDateError, - - #[error("Invalid time")] - InvalidTimeError, - - #[error("Schema conversion Error: {0}")] - SchemaConversionError(#[source] TryFromIntError), - - #[error("Decimal convert error")] - DecimalConvertError(#[source] rust_decimal::Error), -} - -#[derive(Error, Debug)] -pub enum SnowflakeStreamError { - #[error("Time travel not available for table")] - TimeTravelNotAvailableError, - - #[error("Unsupported \"{0}\" action in stream")] - UnsupportedActionInStream(String), - - #[error("Cannot determine action")] - CannotDetermineAction, - - #[error("Stream not found")] - StreamNotFound, -} diff --git a/dozer-ingestion/snowflake/src/schema_helper.rs b/dozer-ingestion/snowflake/src/schema_helper.rs deleted file mode 100644 index bf1716932a..0000000000 --- a/dozer-ingestion/snowflake/src/schema_helper.rs +++ /dev/null @@ -1,61 +0,0 @@ -use dozer_ingestion_connector::{ - dozer_types::{models::ingestion_types::SnowflakeConfig, types::FieldType}, - SourceSchema, -}; -use odbc::create_environment_v3; -use std::collections::HashMap; - -use crate::{connection::client::Client, SnowflakeError, SnowflakeSchemaError}; - -pub struct SchemaHelper {} - -impl SchemaHelper { - #[allow(clippy::type_complexity)] - pub fn get_schema( - config: SnowflakeConfig, - table_names: Option<&[String]>, - ) -> Result>, SnowflakeError> { - let env = create_environment_v3().map_err(|e| e.unwrap()).unwrap(); - let client = Client::new(config.clone().into(), &env); - - let keys = client.fetch_keys()?; - - let tables_indexes = table_names.map(|table_names| { - let mut result = HashMap::new(); - for (idx, table_name) in table_names.iter().enumerate() { - result.insert(table_name.clone(), idx); - } - - result - }); - - client.fetch_tables(tables_indexes, keys, config.schema) - } - - pub fn map_schema_type( - type_name: &str, - scale: Option, - ) -> Result { - match type_name { - "NUMBER" => scale.map_or(Ok(FieldType::Int), |scale| { - if scale > 0 { - Ok(FieldType::Decimal) - } else { - Ok(FieldType::Int) - } - }), - "FLOAT" => Ok(FieldType::Float), - "TEXT" => Ok(FieldType::String), - "BINARY" => Ok(FieldType::Binary), - "BOOLEAN" => Ok(FieldType::Boolean), - "DATE" => Ok(FieldType::Date), - "TIMESTAMP_LTZ" | "TIMESTAMP_NTZ" | "TIMESTAMP_TZ" => Ok(FieldType::Timestamp), - // TODO: proper type handling for VARIANT and TIME - "VARIANT" => Ok(FieldType::String), - "TIME" => Ok(FieldType::String), - _ => Err(SnowflakeSchemaError::ColumnTypeNotSupported(format!( - "{type_name:?}" - ))), - } - } -} diff --git a/dozer-ingestion/snowflake/src/stream_consumer.rs b/dozer-ingestion/snowflake/src/stream_consumer.rs deleted file mode 100644 index 811ca6d96d..0000000000 --- a/dozer-ingestion/snowflake/src/stream_consumer.rs +++ /dev/null @@ -1,154 +0,0 @@ -use dozer_ingestion_connector::{ - dozer_types::{ - models::ingestion_types::{IngestionMessage, TransactionInfo}, - node::OpIdentifier, - types::{Field, Operation, Record}, - }, - Ingestor, -}; - -use crate::{connection::client::Client, SnowflakeError, SnowflakeStreamError}; - -#[derive(Default)] -pub struct StreamConsumer {} - -impl StreamConsumer { - pub fn new() -> Self { - Self {} - } - - pub fn get_stream_table_name(table_name: &str, client_name: &str) -> String { - format!("dozer_{table_name}_{client_name}_stream") - } - - pub fn get_stream_temp_table_name(table_name: &str, client_name: &str) -> String { - format!("dozer_{table_name}_{client_name}_stream_temp") - } - - pub fn is_stream_created(client: &Client, table_name: &str) -> Result { - client.stream_exist(&Self::get_stream_table_name(table_name, &client.get_name())) - } - - pub fn drop_stream(client: &Client, table_name: &str) -> Result<(), SnowflakeError> { - let query = format!( - "DROP STREAM IF EXISTS {}", - Self::get_stream_table_name(table_name, &client.get_name()), - ); - - client.exec(&query) - } - - pub fn create_stream(client: &Client, table_name: &String) -> Result<(), SnowflakeError> { - let query = format!( - "CREATE STREAM {} on table {} SHOW_INITIAL_ROWS = TRUE", - Self::get_stream_table_name(table_name, &client.get_name()), - table_name, - ); - - let result = client.exec_stream_creation(query)?; - - if !result { - let query = format!( - "CREATE STREAM {} on view {} SHOW_INITIAL_ROWS = TRUE", - Self::get_stream_table_name(table_name, &client.get_name()), - table_name - ); - client.exec(&query)?; - } - - Ok(()) - } - - fn get_operation( - row: Vec, - action_idx: usize, - used_columns_for_schema: usize, - ) -> Result { - if let Field::String(action) = row.get(action_idx).unwrap() { - let mut row_mut = row.clone(); - let insert_action = &"INSERT"; - let delete_action = &"DELETE"; - - row_mut.truncate(used_columns_for_schema); - - if insert_action == action { - Ok(Operation::Insert { - new: Record::new(row_mut), - }) - } else if delete_action == action { - Ok(Operation::Delete { - old: Record::new(row_mut), - }) - } else { - Err(SnowflakeError::SnowflakeStreamError( - SnowflakeStreamError::UnsupportedActionInStream(action.clone()), - )) - } - } else { - Err(SnowflakeError::SnowflakeStreamError( - SnowflakeStreamError::CannotDetermineAction, - )) - } - } - - pub fn consume_stream( - &mut self, - client: &Client, - table_name: &str, - ingestor: &Ingestor, - table_index: usize, - iteration: u64, - ) -> Result<(), SnowflakeError> { - let temp_table_name = Self::get_stream_temp_table_name(table_name, &client.get_name()); - let stream_name = Self::get_stream_table_name(table_name, &client.get_name()); - - let query = format!( - "CREATE TEMP TABLE IF NOT EXISTS {temp_table_name} AS - SELECT * FROM {stream_name} ORDER BY METADATA$ACTION;" - ); - - client.exec(&query)?; - - let rows = client.fetch(format!("SELECT * FROM {temp_table_name};"))?; - if let Some(schema) = rows.schema() { - let schema_len = schema.len(); - let mut truncated_schema = schema.clone(); - truncated_schema.truncate(schema_len - 3); - - let columns_length = schema_len; - let used_columns_for_schema = columns_length - 3; - let action_idx = used_columns_for_schema; - - for (idx, result) in rows.enumerate() { - let row = result?; - let op = Self::get_operation(row, action_idx, used_columns_for_schema)?; - if ingestor - .blocking_handle_message(IngestionMessage::OperationEvent { - table_index, - op, - id: Some(OpIdentifier::new(iteration, idx as u64)), - }) - .is_err() - { - // If receiver is dropped, we can stop processing - return Ok(()); - } - if ingestor - .blocking_handle_message(IngestionMessage::TransactionInfo( - TransactionInfo::Commit { - id: Some(OpIdentifier::new(iteration, idx as u64)), - source_time: None, - }, - )) - .is_err() - { - return Ok(()); - } - } - } - - let query = format!("DROP TABLE {temp_table_name};"); - - client.exec(&query) - } -} diff --git a/dozer-ingestion/snowflake/src/test_utils.rs b/dozer-ingestion/snowflake/src/test_utils.rs deleted file mode 100644 index a8e0e85198..0000000000 --- a/dozer-ingestion/snowflake/src/test_utils.rs +++ /dev/null @@ -1,17 +0,0 @@ -use dozer_ingestion_connector::dozer_types::models::ingestion_types::SnowflakeConfig; -use odbc::create_environment_v3; - -use crate::{connection::client::Client, stream_consumer::StreamConsumer, SnowflakeError}; - -pub fn remove_streams( - connection: SnowflakeConfig, - table_name: &str, -) -> Result { - let env = create_environment_v3().unwrap(); - let client = Client::new(connection.into(), &env); - - client.drop_stream(&StreamConsumer::get_stream_table_name( - table_name, - &client.get_name(), - )) -} diff --git a/dozer-ingestion/snowflake/src/tests/dozer-config.yaml b/dozer-ingestion/snowflake/src/tests/dozer-config.yaml deleted file mode 100644 index 663b9f970a..0000000000 --- a/dozer-ingestion/snowflake/src/tests/dozer-config.yaml +++ /dev/null @@ -1,28 +0,0 @@ -app_name: snowflake-test -version: 1 -connections: - - config: !Snowflake - server: "{{SN_SERVER}}" - port: 443 - user: "{{SN_USER}}" - password: "{{SN_PASSWORD}}" - database: "{{SN_DATABASE}}" - schema: PUBLIC - warehouse: "{{SN_WAREHOUSE}}" - driver: "{{SN_DRIVER}}" - name: sn_data - -sources: - - name: customers - connection: sn_data - table_name: CUSTOMERS - - -sql: | - SELECT O_ORDERKEY, O_CUSTKEY, O_ORDERSTATUS, O_TOTALPRICE, O_ORDERDATE, O_ORDERPRIORITY, O_CLERK, O_SHIPPRIORITY, O_COMMENT - INTO orders_data - FROM orders; - -sinks: - - table_name: customers_data - config: !Dummy diff --git a/dozer-ingestion/snowflake/src/tests/mod.rs b/dozer-ingestion/snowflake/src/tests/mod.rs deleted file mode 100644 index a7fda4c5f5..0000000000 --- a/dozer-ingestion/snowflake/src/tests/mod.rs +++ /dev/null @@ -1,206 +0,0 @@ -use std::time::Duration; - -use dozer_ingestion_connector::{ - dozer_types::{ - models::connection::ConnectionConfig, - types::FieldType::{Binary, Boolean, Date, Decimal, Float, Int, String, Timestamp}, - }, - test_util::{create_test_runtime, load_test_connection_config, spawn_connector}, - tokio, Connector, TableIdentifier, -}; -use odbc::create_environment_v3; -use rand::Rng; - -use crate::{ - connection::client::Client, connector::SnowflakeConnector, stream_consumer::StreamConsumer, - test_utils::remove_streams, -}; - -const TABLE_NAME: &str = "CUSTOMERS"; - -#[tokio::test] -#[ignore] -async fn test_disabled_connector_and_read_from_stream() { - let config = load_test_connection_config(); - let ConnectionConfig::Snowflake(connection) = config else { - panic!("Snowflake config expected"); - }; - - let env = create_environment_v3().map_err(|e| e.unwrap()).unwrap(); - let client = Client::new(connection.clone().into(), &env); - - let mut rng = rand::thread_rng(); - let table_name = format!("CUSTOMER_TEST_{}", rng.gen::()); - - client - .exec(&format!( - "CREATE TABLE {table_name} LIKE SNOWFLAKE_SAMPLE_DATA.TPCH_SF1000.CUSTOMER;" - )) - .unwrap(); - client.exec(&format!("ALTER TABLE PUBLIC.{table_name} ADD CONSTRAINT {table_name}_PK PRIMARY KEY (C_CUSTKEY);")).unwrap(); - client.exec(&format!("INSERT INTO {table_name} SELECT * FROM SNOWFLAKE_SAMPLE_DATA.TPCH_SF1000.CUSTOMER LIMIT 100")).unwrap(); - - remove_streams(connection.clone(), TABLE_NAME).unwrap(); - - let runtime = create_test_runtime(); - let mut connector = SnowflakeConnector::new("snowflake".to_string(), connection.clone()); - let tables = runtime - .block_on( - connector.list_columns(vec![TableIdentifier::from_table_name(table_name.clone())]), - ) - .unwrap(); - - let (mut iterator, _) = spawn_connector(runtime, connector, tables); - - let mut i = 0; - while i < 100 { - iterator.next_timeout(Duration::from_secs(10)).await; - i += 1; - } - - assert_eq!(100, i); - - client.exec(&format!("INSERT INTO {table_name} SELECT * FROM SNOWFLAKE_SAMPLE_DATA.TPCH_SF1000.CUSTOMER LIMIT 100 OFFSET 100")).unwrap(); - - let mut i = 0; - while i < 100 { - iterator.next_timeout(Duration::from_secs(10)).await; - i += 1; - } - - assert_eq!(100, i); -} - -#[tokio::test] -#[ignore] -async fn test_disabled_connector_get_schemas_test() { - let config = load_test_connection_config(); - let ConnectionConfig::Snowflake(connection) = config else { - panic!("Snowflake config expected"); - }; - let mut connector = SnowflakeConnector::new("snowflake".to_string(), connection.clone()); - let env = create_environment_v3().map_err(|e| e.unwrap()).unwrap(); - let client = Client::new(connection.into(), &env); - - let mut rng = rand::thread_rng(); - let table_name = format!("SCHEMA_MAPPING_TEST_{}", rng.gen::()); - - client - .exec(&format!( - "create table {table_name} - ( - integer_column integer, - float_column float, - text_column varchar, - binary_column binary, - boolean_column boolean, - date_column date, - datetime_column datetime, - decimal_column decimal(5, 2) - ) - data_retention_time_in_days = 0; - - " - )) - .unwrap(); - - let table_infos = connector - .list_columns(vec![TableIdentifier::from_table_name(table_name.clone())]) - .await - .unwrap(); - let schemas = connector.get_schemas(&table_infos).await.unwrap(); - - let source_schema = schemas[0].as_ref().unwrap(); - - for field in &source_schema.schema.fields { - let expected_type = match field.name.as_str() { - "INTEGER_COLUMN" => Int, - "FLOAT_COLUMN" => Float, - "TEXT_COLUMN" => String, - "BINARY_COLUMN" => Binary, - "BOOLEAN_COLUMN" => Boolean, - "DATE_COLUMN" => Date, - "DATETIME_COLUMN" => Timestamp, - "DECIMAL_COLUMN" => Decimal, - _ => { - panic!("Unexpected column: {}", field.name) - } - }; - - assert_eq!(expected_type, field.typ); - } - - client.exec(&format!("DROP TABLE {table_name};")).unwrap(); -} - -#[tokio::test] -#[ignore] -async fn test_disabled_connector_missing_table_validator() { - let config = load_test_connection_config(); - let ConnectionConfig::Snowflake(connection) = config else { - panic!("Snowflake config expected"); - }; - let mut connector = SnowflakeConnector::new("snowflake".to_string(), connection.clone()); - - let not_existing_table = "not_existing_table".to_string(); - let result = connector - .list_columns(vec![TableIdentifier::from_table_name(not_existing_table)]) - .await; - - assert!(result - .unwrap_err() - .to_string() - .starts_with("table not found")); - - let table_infos = connector - .list_columns(vec![TableIdentifier::from_table_name( - TABLE_NAME.to_string(), - )]) - .await - .unwrap(); - let result = connector.get_schemas(&table_infos).await.unwrap(); - - assert!(result[0].is_ok()); -} - -#[tokio::test] -#[ignore] -async fn test_disabled_connector_is_stream_created() { - let config = load_test_connection_config(); - let ConnectionConfig::Snowflake(connection) = config else { - panic!("Snowflake config expected"); - }; - - let env = create_environment_v3().map_err(|e| e.unwrap()).unwrap(); - let client = Client::new(connection.into(), &env); - - let mut rng = rand::thread_rng(); - let table_name = format!("STREAM_EXIST_TEST_{}", rng.gen::()); - - client - .exec(&format!( - "CREATE TABLE {table_name} (id INTEGER) - data_retention_time_in_days = 0; " - )) - .unwrap(); - - let result = StreamConsumer::is_stream_created(&client, &table_name).unwrap(); - assert!( - !result, - "Stream was not created yet, so result of check should be false" - ); - - StreamConsumer::create_stream(&client, &table_name).unwrap(); - let result = StreamConsumer::is_stream_created(&client, &table_name).unwrap(); - assert!( - result, - "Stream is created, so result of check should be true" - ); - - StreamConsumer::drop_stream(&client, &table_name).unwrap(); - let result = StreamConsumer::is_stream_created(&client, &table_name).unwrap(); - assert!( - !result, - "Stream was dropped, so result of check should be false" - ); -} diff --git a/dozer-ingestion/src/errors.rs b/dozer-ingestion/src/errors.rs deleted file mode 100644 index 8668bc6e20..0000000000 --- a/dozer-ingestion/src/errors.rs +++ /dev/null @@ -1,41 +0,0 @@ -use dozer_ingestion_connector::dozer_types::thiserror::{self, Error}; - -#[derive(Error, Debug)] -pub enum ConnectorError { - #[error("Unsupported grpc adapter: {0} {1:?}")] - UnsupportedGrpcAdapter(String, Option), - - #[cfg(feature = "mongodb")] - #[error("mongodb config error: {0}")] - MongodbConfig(#[from] dozer_ingestion_mongodb::MongodbConnectorError), - - #[error("mysql config error: {0}")] - MysqlConfig(#[from] dozer_ingestion_mysql::MySQLConnectorError), - - #[error("postgres config error: {0}")] - PostgresConfig(#[from] dozer_ingestion_postgres::PostgresConnectorError), - - #[error("snowflake feature is not enabled")] - SnowflakeFeatureNotEnabled, - - #[error("kafka feature is not enabled")] - KafkaFeatureNotEnabled, - - #[error("ethereum feature is not enabled")] - EthereumFeatureNotEnabled, - - #[error("mongodb feature is not enabled")] - MongodbFeatureNotEnabled, - - #[error("datafusion feature is not enabled")] - DatafusionFeatureNotEnabled, - - #[error("javascript feature is not enabled")] - JavascrtiptFeatureNotEnabled, - - #[error("{0}: This feature is only avaialble in enteprise. Please contact us.")] - FeatureNotEnabled(String), - - #[error("{0} is not supported as a source connector")] - Unsupported(String), -} diff --git a/dozer-ingestion/src/lib.rs b/dozer-ingestion/src/lib.rs deleted file mode 100644 index 934429f0d3..0000000000 --- a/dozer-ingestion/src/lib.rs +++ /dev/null @@ -1,176 +0,0 @@ -#[cfg(feature = "ethereum")] -use dozer_ingestion_connector::dozer_types::models::ingestion_types::EthProviderConfig; -use dozer_ingestion_connector::dozer_types::{ - event::EventHub, - log::debug, - models::{ - connection::{Connection, ConnectionConfig}, - ingestion_types::default_grpc_adapter, - }, - prettytable::Table, -}; -#[cfg(feature = "datafusion")] -use dozer_ingestion_deltalake::DeltaLakeConnector; -#[cfg(feature = "ethereum")] -use dozer_ingestion_ethereum::{EthLogConnector, EthTraceConnector}; -use dozer_ingestion_grpc::{connector::GrpcConnector, ArrowAdapter, DefaultAdapter}; -#[cfg(feature = "javascript")] -use dozer_ingestion_javascript::JavaScriptConnector; -#[cfg(feature = "kafka")] -use dozer_ingestion_kafka::connector::KafkaConnector; -#[cfg(feature = "mongodb")] -use dozer_ingestion_mongodb::MongodbConnector; -use dozer_ingestion_mysql::connector::{mysql_connection_opts_from_url, MySQLConnector}; -#[cfg(feature = "datafusion")] -use dozer_ingestion_object_store::connector::ObjectStoreConnector; -use dozer_ingestion_postgres::{ - connection::helper::map_connection_config, - connector::{PostgresConfig, PostgresConnector}, -}; -#[cfg(feature = "snowflake")] -use dozer_ingestion_snowflake::connector::SnowflakeConnector; -use dozer_ingestion_webhook::connector::WebhookConnector; -use errors::ConnectorError; -use std::sync::Arc; -use tokio::runtime::Runtime; - -pub mod errors; -pub use dozer_ingestion_connector::*; - -const DEFAULT_POSTGRES_SNAPSHOT_BATCH_SIZE: u32 = 100_000; - -#[allow(unused_variables)] -pub fn get_connector( - runtime: Arc, - event_hub: EventHub, - connection: Connection, - state: Option>, -) -> Result, ConnectorError> { - let config = connection.config; - match config.clone() { - ConnectionConfig::Postgres(c) => { - let config = map_connection_config(&config)?; - let postgres_config = PostgresConfig { - name: connection.name, - config, - schema: c.schema, - batch_size: c.batch_size.unwrap_or(DEFAULT_POSTGRES_SNAPSHOT_BATCH_SIZE) as usize, - }; - - if let Some(dbname) = postgres_config.config.get_dbname() { - debug!("Connecting to postgres database - {}", dbname.to_string()); - } - Ok(Box::new(PostgresConnector::new(postgres_config, state)?)) - } - #[cfg(feature = "ethereum")] - ConnectionConfig::Ethereum(eth_config) => match eth_config.provider { - EthProviderConfig::Log(log_config) => { - Ok(Box::new(EthLogConnector::new(log_config, connection.name))) - } - EthProviderConfig::Trace(trace_config) => Ok(Box::new(EthTraceConnector::new( - trace_config, - connection.name, - ))), - }, - #[cfg(not(feature = "ethereum"))] - ConnectionConfig::Ethereum(_) => Err(ConnectorError::EthereumFeatureNotEnabled), - ConnectionConfig::Grpc(grpc_config) => { - match grpc_config - .adapter - .clone() - .unwrap_or_else(default_grpc_adapter) - .as_str() - { - "arrow" => Ok(Box::new(GrpcConnector::::new( - connection.name, - grpc_config, - ))), - "default" => Ok(Box::new(GrpcConnector::::new( - connection.name, - grpc_config, - ))), - _ => Err(ConnectorError::UnsupportedGrpcAdapter( - connection.name, - grpc_config.adapter, - )), - } - } - #[cfg(feature = "snowflake")] - ConnectionConfig::Snowflake(snowflake) => { - let snowflake_config = snowflake; - - Ok(Box::new(SnowflakeConnector::new( - connection.name, - snowflake_config, - ))) - } - #[cfg(not(feature = "snowflake"))] - ConnectionConfig::Snowflake(_) => Err(ConnectorError::SnowflakeFeatureNotEnabled), - #[cfg(feature = "kafka")] - ConnectionConfig::Kafka(kafka_config) => Ok(Box::new(KafkaConnector::new(kafka_config))), - #[cfg(not(feature = "kafka"))] - ConnectionConfig::Kafka(_) => Err(ConnectorError::KafkaFeatureNotEnabled), - #[cfg(feature = "datafusion")] - ConnectionConfig::S3Storage(object_store_config) => { - Ok(Box::new(ObjectStoreConnector::new(object_store_config))) - } - #[cfg(feature = "datafusion")] - ConnectionConfig::LocalStorage(object_store_config) => { - Ok(Box::new(ObjectStoreConnector::new(object_store_config))) - } - #[cfg(feature = "datafusion")] - ConnectionConfig::DeltaLake(delta_lake_config) => { - Ok(Box::new(DeltaLakeConnector::new(delta_lake_config))) - } - #[cfg(not(feature = "datafusion"))] - ConnectionConfig::DeltaLake(_) => Err(ConnectorError::DatafusionFeatureNotEnabled), - #[cfg(not(feature = "datafusion"))] - ConnectionConfig::LocalStorage(_) => Err(ConnectorError::DatafusionFeatureNotEnabled), - #[cfg(not(feature = "datafusion"))] - ConnectionConfig::S3Storage(_) => Err(ConnectorError::DatafusionFeatureNotEnabled), - #[cfg(feature = "mongodb")] - ConnectionConfig::MongoDB(mongodb_config) => { - let connection_string = mongodb_config.connection_string; - Ok(Box::new(MongodbConnector::new(connection_string)?)) - } - #[cfg(not(feature = "mongodb"))] - ConnectionConfig::MongoDB(_) => Err(ConnectorError::MongodbFeatureNotEnabled), - ConnectionConfig::MySQL(mysql_config) => { - let opts = mysql_connection_opts_from_url(&mysql_config.url)?; - Ok(Box::new(MySQLConnector::new( - mysql_config.url, - opts, - mysql_config.server_id, - ))) - } - ConnectionConfig::Webhook(webhook_config) => { - Ok(Box::new(WebhookConnector::new(webhook_config))) - } - #[cfg(not(feature = "javascript"))] - ConnectionConfig::JavaScript(_) => Err(ConnectorError::JavascrtiptFeatureNotEnabled), - #[cfg(feature = "javascript")] - ConnectionConfig::JavaScript(javascript_config) => Ok(Box::new(JavaScriptConnector::new( - runtime, - javascript_config, - ))), - ConnectionConfig::Aerospike(_) => { - Err(ConnectorError::FeatureNotEnabled("Aerospike".to_string())) - } - ConnectionConfig::Oracle(_) => Err(ConnectorError::FeatureNotEnabled("Oracle".to_string())), - } -} - -pub fn get_connector_info_table(connection: &Connection) -> Option
{ - match &connection.config { - ConnectionConfig::Postgres(config) => match config.replenish() { - Ok(conf) => Some(conf.convert_to_table()), - Err(_) => None, - }, - ConnectionConfig::Ethereum(config) => Some(config.convert_to_table()), - ConnectionConfig::Snowflake(config) => Some(config.convert_to_table()), - ConnectionConfig::Kafka(config) => Some(config.convert_to_table()), - ConnectionConfig::S3Storage(config) => Some(config.convert_to_table()), - ConnectionConfig::LocalStorage(config) => Some(config.convert_to_table()), - _ => None, - } -} diff --git a/dozer-ingestion/src/tests/README.md b/dozer-ingestion/src/tests/README.md deleted file mode 100644 index 4ef8f1e0b5..0000000000 --- a/dozer-ingestion/src/tests/README.md +++ /dev/null @@ -1,30 +0,0 @@ -### Connectors tests - -As connectors connects to external databases, we need to have running databases instances. -To do that, we will create docker containers using this command. - -```bash -docker-compose -f dozer-ingestion/src/tests/connections/postgres/docker-compose.yaml up -``` - -After we have running containers, tests can be executed with this command -```bash -cargo test test_connector_ -- --ignored -``` - -### Snowflake tests - -As snowflake is cloud based database, we cannot create container for it, so to execute tests we need to have credentials. -Credentials can be set with such command - -```bash -export SN_SERVER={sn_server} -export SN_USER={sn_user} -export SN_PASSWORD={sn_password} -export SN_DATABASE={sn_database} -export SN_WAREHOUSE={sn_warehouse} -``` - -After settings is properly set, tests can be executed with such command - -```cargo test snowflake -- --ignored``` \ No newline at end of file diff --git a/dozer-ingestion/src/tests/connections/mysql/docker-compose.yaml b/dozer-ingestion/src/tests/connections/mysql/docker-compose.yaml deleted file mode 100644 index 4959b99461..0000000000 --- a/dozer-ingestion/src/tests/connections/mysql/docker-compose.yaml +++ /dev/null @@ -1,26 +0,0 @@ -version: '2.4' -services: - dozer-wait-for-connections-healthy: - image: alpine - command: echo 'All connections are healthy' - depends_on: - mysql-dozer-tests-db: - condition: service_healthy - mysql-dozer-tests-db: - container_name: mysql-dozer-tests-db - build: - context: . - ports: - - target: 3306 - published: 3306 - environment: - - MYSQL_ROOT_PASSWORD=mysql - - MYSQL_ROOT_HOST=% - - MYSQL_DATABASE=test - healthcheck: - test: - - CMD-SHELL - - mysqladmin ping -uroot --password=$$MYSQL_ROOT_PASSWORD - interval: 5s - timeout: 5s - retries: 5 diff --git a/dozer-ingestion/src/tests/connections/postgres/docker-compose.yaml b/dozer-ingestion/src/tests/connections/postgres/docker-compose.yaml deleted file mode 100644 index a5d1073168..0000000000 --- a/dozer-ingestion/src/tests/connections/postgres/docker-compose.yaml +++ /dev/null @@ -1,27 +0,0 @@ -version: '2.4' -services: - dozer-wait-for-connections-healthy: - image: alpine - command: echo 'All connections are healthy' - depends_on: - postgres-dozer-tests-db: - condition: service_healthy - postgres-dozer-tests-db: - container_name: postgres-dozer-tests-db - image: debezium/postgres:13 - ports: - - target: 5432 - published: 5434 - environment: - - POSTGRES_DB=dozer_test - - POSTGRES_USER=postgres - - POSTGRES_PASSWORD=postgres - - ALLOW_IP_RANGE=0.0.0.0/0 - command: postgres -c hba_file=/var/lib/stock-sample/pg_hba.conf - healthcheck: - test: - - CMD-SHELL - - pg_isready -U postgres -h 0.0.0.0 -d dozer_test - interval: 5s - timeout: 5s - retries: 5 diff --git a/dozer-ingestion/tests/test_connectors.rs b/dozer-ingestion/tests/test_connectors.rs deleted file mode 100644 index 307fa4f29e..0000000000 --- a/dozer-ingestion/tests/test_connectors.rs +++ /dev/null @@ -1,55 +0,0 @@ -use std::sync::Arc; - -use dozer_ingestion::test_util::create_test_runtime; -use test_suite::{ - run_test_suite_basic_cud, run_test_suite_basic_data_ready, run_test_suite_basic_insert_only, -}; -use tokio::runtime::Runtime; - -#[cfg(feature = "datafusion")] -#[test] -fn test_local_storage() { - let runtime = create_test_runtime(); - runtime.block_on(test_local_storage_impl(runtime.clone())); -} -#[cfg(feature = "datafusion")] -async fn test_local_storage_impl(runtime: Arc) { - let _ = env_logger::builder().is_test(true).try_init(); - - run_test_suite_basic_data_ready::( - runtime.clone(), - ) - .await; - run_test_suite_basic_insert_only::(runtime) - .await; -} - -#[test] -fn test_postgres() { - let runtime = create_test_runtime(); - runtime.block_on(test_postgres_impl(runtime.clone())); -} - -async fn test_postgres_impl(runtime: Arc) { - let _ = env_logger::builder().is_test(true).try_init(); - - run_test_suite_basic_data_ready::(runtime.clone()).await; - run_test_suite_basic_insert_only::(runtime.clone()).await; - run_test_suite_basic_cud::(runtime).await; -} - -#[cfg(feature = "mongodb")] -#[test] -fn test_mongodb() { - let runtime = create_test_runtime(); - runtime.block_on(test_mongodb_impl(runtime.clone())); -} - -#[cfg(feature = "mongodb")] -async fn test_mongodb_impl(runtime: Arc) { - let _ = env_logger::builder().is_test(true).try_init(); - - run_test_suite_basic_data_ready::(runtime).await; -} - -mod test_suite; diff --git a/dozer-ingestion/tests/test_suite/basic.rs b/dozer-ingestion/tests/test_suite/basic.rs deleted file mode 100644 index 2f1e30f883..0000000000 --- a/dozer-ingestion/tests/test_suite/basic.rs +++ /dev/null @@ -1,347 +0,0 @@ -use std::{sync::Arc, time::Duration}; - -use dozer_ingestion_connector::{ - dozer_types::{ - log::warn, - models::ingestion_types::IngestionMessage, - types::{Field, FieldDefinition, FieldType, Operation, Record, Schema}, - }, - test_util::spawn_connector, - CdcType, Connector, SourceSchema, TableIdentifier, TableInfo, -}; -use tokio::runtime::Runtime; - -use crate::test_suite::data::reorder; - -use super::{ - data, - records::{Operation as RecordsOperation, Records}, - CudConnectorTest, DataReadyConnectorTest, InsertOnlyConnectorTest, -}; - -pub async fn run_test_suite_basic_data_ready(runtime: Arc) { - let (_connector_test, mut connector) = T::new().await; - - // List tables. - let tables = connector.list_tables().await.unwrap(); - connector.validate_tables(&tables).await.unwrap(); - - // List columns. - let tables = connector.list_columns(tables).await.unwrap(); - - // Get schemas. - let schemas = connector.get_schemas(&tables).await.unwrap(); - let schemas = schemas - .into_iter() - .map(|schema| schema.expect("Failed to get schema")) - .collect::>(); - - // Run connector. - let (mut iterator, abort_handle) = spawn_connector(runtime, connector, tables); - - // Loop over messages until timeout. - let mut num_operations = 0; - while let Some(message) = iterator.next_timeout(Duration::from_secs(1)).await { - // Check message identifier. - if let IngestionMessage::OperationEvent { - table_index, op, .. - } = &message - { - num_operations += 1; - // Check record schema consistency. - match op { - Operation::Insert { new } => { - assert_record_matches_source_schema(new, &schemas[*table_index], true); - } - Operation::Update { old, new } => { - assert_record_matches_source_schema(old, &schemas[*table_index], false); - assert_record_matches_source_schema(new, &schemas[*table_index], true); - } - Operation::Delete { old } => { - assert_record_matches_source_schema(old, &schemas[*table_index], false); - } - Operation::BatchInsert { new } => { - for op in new { - assert_record_matches_source_schema(op, &schemas[*table_index], true); - } - } - } - } - } - - // There should be at least one message. - assert!(num_operations > 0); - abort_handle.abort() -} - -pub async fn run_test_suite_basic_insert_only(runtime: Arc) { - let table_name = "test_table".to_string(); - for data_fn in [ - data::records_with_primary_key, - data::records_without_primary_key, - ] { - // Load test data. - let ((fields, primary_index), records) = data_fn(); - - // Create connector. - let schema_name = None; - let Some((_connector_test, mut connector, (actual_fields, actual_primary_index))) = T::new( - schema_name.clone(), - table_name.clone(), - (fields.clone(), primary_index.clone()), - records.clone(), - ) - .await - else { - warn!("Connector does not support schema name {schema_name:?} or primary index {primary_index:?}."); - continue; - }; - for field in &fields { - if !actual_fields - .iter() - .any(|actual_field| actual_field == field) - { - warn!("Field {:?} is not supported by the connector.", field) - } - } - - // Validate connection. - connector.validate_connection().await.unwrap(); - - // Validate tables. - connector - .validate_tables(&[TableIdentifier::new( - schema_name.clone(), - table_name.clone(), - )]) - .await - .unwrap(); - - // List columns. - let tables = connector - .list_columns(vec![TableIdentifier::new(schema_name, table_name.clone())]) - .await - .unwrap(); - assert_eq!(tables.len(), 1); - assert_eq!(tables[0].name, table_name); - assert_eq!( - tables[0].column_names, - actual_fields - .iter() - .map(|field| field.name.clone()) - .collect::>() - ); - - // Validate schemas. - let schemas = connector.get_schemas(&tables).await.unwrap(); - assert_eq!(schemas.len(), 1); - let actual_schema = &schemas[0].as_ref().unwrap().schema; - assert_eq!(actual_schema.fields, actual_fields); - assert_eq!(actual_schema.primary_index, actual_primary_index); - - // Run the connector and check data is ingested. - let (mut iterator, abort_handle) = spawn_connector(runtime.clone(), connector, tables); - - let mut record_iter = records.iter(); - - while let Some(message) = iterator.next_timeout(Duration::from_secs(1)).await { - // Filter out non-operation events. - let IngestionMessage::OperationEvent { op: operation, .. } = message else { - continue; - }; - - let mut check = |actual_record| { - // Record must match schema. - assert_record_matches_schema(&actual_record, actual_schema, false); - - // Record must match expected record. - assert_records_match( - &actual_record.values, - &actual_fields, - &actual_primary_index, - record_iter - .next() - .expect("Connector sent more records than expected"), - &fields, - false, - ); - }; - - // Operation must be insert or batch insert. - match operation { - Operation::Insert { new } => check(new), - Operation::BatchInsert { new } => { - for new in new { - check(new); - } - } - _ => panic!( - "Expected an insert or batch insert event, but got {:?}", - operation - ), - } - } - - assert!( - record_iter.next().is_none(), - "Connector sent less records than expected." - ); - abort_handle.abort(); - } -} - -pub async fn run_test_suite_basic_cud(runtime: Arc) { - // Load test data. - let ((fields, primary_index), operations) = data::cud_operations(); - - // Create connector. - let schema_name = None; - let table_name = "test_table".to_string(); - let (connector_test, mut connector, (actual_fields, actual_primary_index)) = T::new( - schema_name.clone(), - table_name.clone(), - (fields.clone(), primary_index), - vec![], - ) - .await - .unwrap(); - - let ((reordered_fields, _reordered_primary_index), reordered_operations) = - reorder(&actual_fields, &actual_primary_index, &operations); - - // Create schema. - let tables = vec![TableInfo { - schema: schema_name, - name: table_name, - column_names: reordered_fields - .into_iter() - .map(|field| field.name) - .collect(), - }]; - let mut schemas = connector.get_schemas(&tables).await.unwrap(); - let actual_schema = schemas.remove(0).unwrap().schema; - - // Feed data to connector. - connector_test.start_cud(operations.clone()).await; - - // Run the connector. - let (mut iterator, abort_handle) = spawn_connector(runtime, connector, tables); - - // Check data schema consistency. - let mut records = Records::new(actual_primary_index.clone()); - while let Some(message) = iterator.next_timeout(Duration::from_secs(1)).await { - // Filter out non-operation events. - let IngestionMessage::OperationEvent { op: operation, .. } = message else { - continue; - }; - - // Record must match schema. - match operation { - Operation::Insert { new } => { - assert_record_matches_schema(&new, &actual_schema, false); - records.append_operation(RecordsOperation::Insert { new: new.values }); - } - Operation::Update { old, new } => { - assert_record_matches_schema(&old, &actual_schema, false); - assert_record_matches_schema(&new, &actual_schema, false); - records.append_operation(RecordsOperation::Update { - old: old.values, - new: new.values, - }); - } - Operation::Delete { old } => { - assert_record_matches_schema(&old, &actual_schema, false); - records.append_operation(RecordsOperation::Delete { old: old.values }); - } - Operation::BatchInsert { new } => { - for new in new { - assert_record_matches_schema(&new, &actual_schema, false); - records.append_operation(RecordsOperation::Insert { new: new.values }); - } - } - } - } - - // We can't check operation exact match because the connector may have batched some of them, - // so we check that the final state is the same. - let mut expected_records = Records::new(actual_primary_index); - for operation in reordered_operations { - expected_records.append_operation(operation); - } - assert_eq!(records, expected_records); - - abort_handle.abort(); -} - -fn assert_record_matches_schema(record: &Record, schema: &Schema, only_match_pk: bool) { - assert_eq!(record.values.len(), schema.fields.len()); - for (index, (field, value)) in schema.fields.iter().zip(record.values.iter()).enumerate() { - // If `only_match_pk` is true, we only check primary key fields. - if only_match_pk && !schema.primary_index.iter().any(|i| i == &index) { - continue; - } - if field.nullable && value == &Field::Null { - continue; - } - match field.typ { - FieldType::UInt => { - assert!(value.as_uint().is_some()) - } - FieldType::U128 => { - assert!(value.as_u128().is_some()) - } - FieldType::Int => { - assert!(value.as_int().is_some()) - } - FieldType::Int8 => { - assert!(value.as_int().is_some()) - } - FieldType::I128 => { - assert!(value.as_i128().is_some()) - } - FieldType::Float => { - assert!(value.as_float().is_some()) - } - FieldType::Boolean => assert!(value.as_boolean().is_some()), - FieldType::String => assert!(value.as_string().is_some()), - FieldType::Text => assert!(value.as_text().is_some()), - FieldType::Binary => assert!(value.as_binary().is_some()), - FieldType::Decimal => assert!(value.as_decimal().is_some()), - FieldType::Timestamp => assert!(value.as_timestamp().is_some()), - FieldType::Date => assert!(value.as_date().is_some()), - FieldType::Json => assert!(value.as_json().is_some()), - FieldType::Point => assert!(value.as_point().is_some()), - FieldType::Duration => assert!(value.as_duration().is_some()), - } - } -} - -fn assert_record_matches_source_schema(record: &Record, schema: &SourceSchema, full_match: bool) { - let only_match_pk = !full_match && schema.cdc_type != CdcType::FullChanges; - assert_record_matches_schema(record, &schema.schema, only_match_pk); -} - -fn assert_records_match( - partial_record: &[Field], - partial_fields: &[FieldDefinition], - partial_primary_index: &[usize], - record: &[Field], - fields: &[FieldDefinition], - only_match_pk: bool, -) { - let partial_index_to_index = partial_fields - .iter() - .map(|field| fields.iter().position(|f| f.name == field.name).unwrap()) - .collect::>(); - - for (partial_index, partial_value) in partial_record.iter().enumerate() { - // If `only_match_pk` is true, we only check primary key fields. - if only_match_pk && !partial_primary_index.iter().any(|i| i == &partial_index) { - continue; - } - assert_eq!( - partial_value, - &record[partial_index_to_index[partial_index]] - ); - } -} diff --git a/dozer-ingestion/tests/test_suite/connectors/mod.rs b/dozer-ingestion/tests/test_suite/connectors/mod.rs deleted file mode 100644 index 51b27e079c..0000000000 --- a/dozer-ingestion/tests/test_suite/connectors/mod.rs +++ /dev/null @@ -1,13 +0,0 @@ -#[cfg(feature = "datafusion")] -mod object_store; -mod postgres; -mod sql; - -#[cfg(feature = "mongodb")] -mod mongodb; - -#[cfg(feature = "mongodb")] -pub use self::mongodb::MongodbConnectorTest; -#[cfg(feature = "datafusion")] -pub use self::object_store::LocalStorageObjectStoreConnectorTest; -pub use self::postgres::PostgresConnectorTest; diff --git a/dozer-ingestion/tests/test_suite/connectors/mongodb.rs b/dozer-ingestion/tests/test_suite/connectors/mongodb.rs deleted file mode 100644 index 38c45ee48c..0000000000 --- a/dozer-ingestion/tests/test_suite/connectors/mongodb.rs +++ /dev/null @@ -1,138 +0,0 @@ -use dozer_ingestion_connector::async_trait; -use dozer_ingestion_mongodb::{ - bson::{self, doc}, - mongodb::{ - self, - options::{ClientOptions, InsertOneOptions, WriteConcern}, - }, - MongodbConnector, -}; -use dozer_utils::{process::run_docker_compose, Cleanup}; -use tempfile::TempDir; - -use crate::test_suite::DataReadyConnectorTest; - -pub struct MongodbConnectorTest { - _cleanup: Cleanup, - _temp_dir: TempDir, -} - -#[async_trait] -impl DataReadyConnectorTest for MongodbConnectorTest { - type Connector = MongodbConnector; - - async fn new() -> (Self, Self::Connector) { - let (db, connector_test, connector) = create_mongodb_server().await; - create_collection_with_all_supported_data_types(&db, "test_collection").await; - (connector_test, connector) - } -} - -async fn create_collection( - database: &mongodb::Database, - name: &str, -) -> mongodb::Collection { - database - .run_command( - doc! { - "create": name, - "changeStreamPreAndPostImages": {"enabled": true} - }, - None, - ) - .await - .expect("Failed to create collection"); - database.collection(name) -} - -async fn create_collection_with_all_supported_data_types(database: &mongodb::Database, name: &str) { - let collection = create_collection(database, name).await; - collection - .insert_one( - doc! { - "null": null, - "bool": true, - "float": 3., - "int": 1, - "string": "string", - "array": [1, {"a": 1}, [1, 2], "b"], - "object": {"a": 1, "b": [1, 2]} - }, - Some( - InsertOneOptions::builder() - .write_concern(Some(WriteConcern::MAJORITY)) - .build(), - ), - ) - .await - .expect("Failed to insert into collection"); -} - -async fn create_mongodb_server() -> (mongodb::Database, MongodbConnectorTest, MongodbConnector) { - let database = "testdb".to_owned(); - - let temp_dir = TempDir::new().expect("Failed to create temp dir"); - let docker_compose_path = temp_dir.path().join("docker-compose.yaml"); - std::fs::write(&docker_compose_path, DOCKER_COMPOSE_YAML) - .expect("Failed to write docker compose file"); - - let cleanup = run_docker_compose(&docker_compose_path, "dozer-wait-for-connections-healthy"); - - let connection_string = format!("mongodb://localhost:27018/{database}"); - - let connection_options = ClientOptions::parse(&connection_string).await.unwrap(); - let mut initiate_connection_options = connection_options.clone(); - initiate_connection_options.direct_connection = Some(true); - // We need a direct connection, as our singular node is not yet initialized as a replSet, - // but it was started as one - let client = mongodb::Client::with_options(initiate_connection_options).unwrap(); - - client - .database("admin") - .run_command( - doc! { - "replSetInitiate": { - "_id": "rs0", - "members": [ - {"_id": 0, "host": "localhost:27018"} - ] - }, - }, - None, - ) - .await - .expect("Failed to initialize replSet"); - - let client = mongodb::Client::with_options(connection_options.clone()).unwrap(); - let db = client.default_database().unwrap(); - let connector = MongodbConnector::new(connection_string).unwrap(); - let test = MongodbConnectorTest { - _cleanup: cleanup, - _temp_dir: temp_dir, - }; - (db, test, connector) -} - -const DOCKER_COMPOSE_YAML: &str = r#"version: '2.4' -services: - mongodb: - container_name: dozer-connectors-mongodb - image: mongodb/mongodb-community-server:6.0.7-ubi8 - ports: - - target: 27018 - published: 27018 - command: mongod --replSet rs0 --port 27018 --bind_ip localhost,mongodb - healthcheck: - test: - - CMD-SHELL - - mongosh --host "localhost:27018" --eval 'print("ready")' - interval: 5s - timeout: 5s - retries: 5 - dozer-wait-for-connections-healthy: - image: alpine - command: echo 'All connections are healthy' - depends_on: - mongodb: - condition: service_healthy -"#; diff --git a/dozer-ingestion/tests/test_suite/connectors/object_store/arrow.rs b/dozer-ingestion/tests/test_suite/connectors/object_store/arrow.rs deleted file mode 100644 index 5376a15df7..0000000000 --- a/dozer-ingestion/tests/test_suite/connectors/object_store/arrow.rs +++ /dev/null @@ -1,490 +0,0 @@ -use std::sync::Arc; - -use dozer_ingestion_connector::dozer_types::{ - arrow::{ - self, - array::{ - Time32MillisecondArray, Time32SecondArray, Time64MicrosecondArray, - Time64NanosecondArray, - }, - }, - chrono::Datelike, - json_types::json_to_string, - types::{Field, FieldDefinition, FieldType}, -}; - -use crate::test_suite::FieldsAndPk; - -pub fn record_batch_with_all_supported_data_types() -> arrow::record_batch::RecordBatch { - use arrow::datatypes::{DataType, Field, TimeUnit}; - - let schema = arrow::datatypes::Schema::new(vec![ - Field::new("bool", DataType::Boolean, false), - Field::new("bool_null", DataType::Boolean, true), - Field::new("int8", DataType::Int8, false), - Field::new("int8_null", DataType::Int8, true), - Field::new("int16", DataType::Int16, false), - Field::new("int16_null", DataType::Int16, true), - Field::new("int32", DataType::Int32, false), - Field::new("int32_null", DataType::Int32, true), - Field::new("int64", DataType::Int64, false), - Field::new("int64_null", DataType::Int64, true), - Field::new("uint8", DataType::UInt8, false), - Field::new("uint8_null", DataType::UInt8, true), - Field::new("uint16", DataType::UInt16, false), - Field::new("uint16_null", DataType::UInt16, true), - Field::new("uint32", DataType::UInt32, false), - Field::new("uint32_null", DataType::UInt32, true), - Field::new("uint64", DataType::UInt64, false), - Field::new("uint64_null", DataType::UInt64, true), - // `Float16` not supported by parquet writer. - // Field::new("float16", DataType::Float16, false), - // Field::new("float16_null", DataType::Float16, true), - Field::new("float32", DataType::Float32, false), - Field::new("float32_null", DataType::Float32, true), - Field::new("float64", DataType::Float64, false), - Field::new("float64_null", DataType::Float64, true), - Field::new( - "timestamp_second", - DataType::Timestamp(TimeUnit::Second, None), - false, - ), - Field::new( - "timestamp_second_null", - DataType::Timestamp(TimeUnit::Second, None), - true, - ), - Field::new( - "timestamp_millisecond", - DataType::Timestamp(TimeUnit::Millisecond, None), - false, - ), - Field::new( - "timestamp_millisecond_null", - DataType::Timestamp(TimeUnit::Millisecond, None), - true, - ), - Field::new( - "timestamp_microsecond", - DataType::Timestamp(TimeUnit::Microsecond, None), - false, - ), - Field::new( - "timestamp_microsecond_null", - DataType::Timestamp(TimeUnit::Microsecond, None), - true, - ), - Field::new( - "timestamp_nanosecond", - DataType::Timestamp(TimeUnit::Nanosecond, None), - false, - ), - Field::new( - "timestamp_nanosecond_null", - DataType::Timestamp(TimeUnit::Nanosecond, None), - true, - ), - Field::new("date32", DataType::Date32, false), - Field::new("date32_null", DataType::Date32, true), - Field::new("date64", DataType::Date64, false), - Field::new("date64_null", DataType::Date64, true), - Field::new("time32_second", DataType::Time32(TimeUnit::Second), false), - Field::new( - "time32_second_null", - DataType::Time32(TimeUnit::Second), - true, - ), - Field::new( - "time32_millisecond", - DataType::Time32(TimeUnit::Millisecond), - false, - ), - Field::new( - "time32_millisecond_null", - DataType::Time32(TimeUnit::Millisecond), - true, - ), - Field::new( - "time64_microsecond", - DataType::Time64(TimeUnit::Microsecond), - false, - ), - Field::new( - "time64_microsecond_null", - DataType::Time64(TimeUnit::Microsecond), - true, - ), - Field::new( - "time64_nanosecond", - DataType::Time64(TimeUnit::Nanosecond), - false, - ), - Field::new( - "time64_nanosecond_null", - DataType::Time64(TimeUnit::Nanosecond), - true, - ), - // `Duration` not supported by parquet writer. - // Field::new( - // "duration_second", - // DataType::Duration(TimeUnit::Second), - // false, - // ), - // Field::new( - // "duration_second_null", - // DataType::Duration(TimeUnit::Second), - // true, - // ), - // Field::new( - // "duration_millisecond", - // DataType::Duration(TimeUnit::Millisecond), - // false, - // ), - // Field::new( - // "duration_millisecond_null", - // DataType::Duration(TimeUnit::Millisecond), - // true, - // ), - // Field::new( - // "duration_microsecond", - // DataType::Duration(TimeUnit::Microsecond), - // false, - // ), - // Field::new( - // "duration_microsecond_null", - // DataType::Duration(TimeUnit::Microsecond), - // true, - // ), - // Field::new( - // "duration_nanosecond", - // DataType::Duration(TimeUnit::Nanosecond), - // false, - // ), - // Field::new( - // "duration_nanosecond_null", - // DataType::Duration(TimeUnit::Nanosecond), - // true, - // ), - Field::new("binary", DataType::Binary, false), - Field::new("binary_null", DataType::Binary, true), - Field::new("fixed_size_binary_1", DataType::FixedSizeBinary(1), false), - Field::new( - "fixed_size_binary_1_null", - DataType::FixedSizeBinary(1), - true, - ), - Field::new("large_binary", DataType::LargeBinary, false), - Field::new("large_binary_null", DataType::LargeBinary, true), - Field::new("utf8", DataType::Utf8, false), - Field::new("utf8_null", DataType::Utf8, true), - Field::new("large_utf8", DataType::LargeUtf8, false), - Field::new("large_utf8_null", DataType::LargeUtf8, true), - Field::new("json", DataType::Utf8, false), - Field::new("json_null", DataType::Utf8, true), - ]); - - use arrow::array::{ - Array, BinaryArray, BooleanArray, Date32Array, Date64Array, FixedSizeBinaryArray, - Float32Array, Float64Array, Int16Array, Int32Array, Int64Array, Int8Array, - LargeBinaryArray, LargeStringArray, StringArray, TimestampMicrosecondArray, - TimestampMillisecondArray, TimestampNanosecondArray, TimestampSecondArray, UInt16Array, - UInt32Array, UInt64Array, UInt8Array, - }; - - let columns: Vec> = vec![ - Arc::new(BooleanArray::from_iter([Some(true), Some(false)])), - Arc::new(BooleanArray::from_iter([Some(true), None])), - Arc::new(Int8Array::from_iter_values([0, 1])), - Arc::new(Int8Array::from_iter([Some(0), None])), - Arc::new(Int16Array::from_iter_values([0, 1])), - Arc::new(Int16Array::from_iter([Some(0), None])), - Arc::new(Int32Array::from_iter_values([0, 1])), - Arc::new(Int32Array::from_iter([Some(0), None])), - Arc::new(Int64Array::from_iter_values([0, 1])), - Arc::new(Int64Array::from_iter([Some(0), None])), - Arc::new(UInt8Array::from_iter_values([0, 1])), - Arc::new(UInt8Array::from_iter([Some(0), None])), - Arc::new(UInt16Array::from_iter_values([0, 1])), - Arc::new(UInt16Array::from_iter([Some(0), None])), - Arc::new(UInt32Array::from_iter_values([0, 1])), - Arc::new(UInt32Array::from_iter([Some(0), None])), - Arc::new(UInt64Array::from_iter_values([0, 1])), - Arc::new(UInt64Array::from_iter([Some(0), None])), - // Arc::new(Float16Array::from_iter_values([0i8.into(), 1i8.into()])), - // Arc::new(Float16Array::from_iter([Some(0i8.into()), None])), - Arc::new(Float32Array::from_iter_values([0.0, 1.0])), - Arc::new(Float32Array::from_iter([Some(0.0), None])), - Arc::new(Float64Array::from_iter_values([0.0, 1.0])), - Arc::new(Float64Array::from_iter([Some(0.0), None])), - Arc::new(TimestampSecondArray::from_iter_values([0, 1])), - Arc::new(TimestampSecondArray::from_iter([Some(0), None])), - Arc::new(TimestampMillisecondArray::from_iter_values([0, 1])), - Arc::new(TimestampMillisecondArray::from_iter([Some(0), None])), - Arc::new(TimestampMicrosecondArray::from_iter_values([0, 1])), - Arc::new(TimestampMicrosecondArray::from_iter([Some(0), None])), - Arc::new(TimestampNanosecondArray::from_iter_values([0, 1])), - Arc::new(TimestampNanosecondArray::from_iter([Some(0), None])), - Arc::new(Date32Array::from_iter_values([0, 1])), - Arc::new(Date32Array::from_iter([Some(0), None])), - Arc::new(Date64Array::from_iter_values([0, 1])), - Arc::new(Date64Array::from_iter([Some(0), None])), - Arc::new(Time32SecondArray::from_iter_values([0, 1])), - Arc::new(Time32SecondArray::from_iter([Some(0), None])), - Arc::new(Time32MillisecondArray::from_iter_values([0, 1])), - Arc::new(Time32MillisecondArray::from_iter([Some(0), None])), - Arc::new(Time64MicrosecondArray::from_iter_values([0, 1])), - Arc::new(Time64MicrosecondArray::from_iter([Some(0), None])), - Arc::new(Time64NanosecondArray::from_iter_values([0, 1])), - Arc::new(Time64NanosecondArray::from_iter([Some(0), None])), - // Arc::new(DurationSecondArray::from_iter_values([0, 1])), - // Arc::new(DurationSecondArray::from_iter([Some(0), None])), - // Arc::new(DurationMillisecondArray::from_iter_values([0, 1])), - // Arc::new(DurationMillisecondArray::from_iter([Some(0), None])), - // Arc::new(DurationMicrosecondArray::from_iter_values([0, 1])), - // Arc::new(DurationMicrosecondArray::from_iter([Some(0), None])), - // Arc::new(DurationNanosecondArray::from_iter_values([0, 1])), - // Arc::new(DurationNanosecondArray::from_iter([Some(0), None])), - // Arc::new(IntervalMonthDayNanoArray::from_iter_values([0, 1])), - // Arc::new(IntervalMonthDayNanoArray::from_iter([Some(0), None])), - Arc::new(BinaryArray::from_iter_values([ - [].as_slice(), - [1].as_slice(), - ])), - Arc::new(BinaryArray::from_iter([Some([]), None])), - Arc::new(FixedSizeBinaryArray::try_from_iter([[0], [1]].into_iter()).unwrap()), - Arc::new( - FixedSizeBinaryArray::try_from_sparse_iter_with_size([Some([0]), None].into_iter(), 1) - .unwrap(), - ), - Arc::new(LargeBinaryArray::from_iter_values([ - [].as_slice(), - [1].as_slice(), - ])), - Arc::new(LargeBinaryArray::from_iter([Some([]), None])), - Arc::new(StringArray::from_iter_values(["", "1"])), - Arc::new(StringArray::from_iter([Some(""), None])), - Arc::new(LargeStringArray::from_iter_values(["", "1"])), - Arc::new(LargeStringArray::from_iter([Some(""), None])), - Arc::new(StringArray::from_iter_values(["[1, 2, 3]", "1"])), - Arc::new(StringArray::from_iter([Some("[1, 2, 3]"), None])), - ]; - - arrow::record_batch::RecordBatch::try_new(Arc::new(schema), columns) - .expect("BUG in record_batch_with_all_supported_data_types") -} - -fn field_type_to_arrow(field_type: FieldType) -> Option { - match field_type { - FieldType::UInt => Some(arrow::datatypes::DataType::UInt64), - FieldType::U128 => None, - FieldType::Int => Some(arrow::datatypes::DataType::Int64), - FieldType::Int8 => Some(arrow::datatypes::DataType::Int64), - FieldType::I128 => None, - FieldType::Float => Some(arrow::datatypes::DataType::Float64), - FieldType::Boolean => Some(arrow::datatypes::DataType::Boolean), - FieldType::String => Some(arrow::datatypes::DataType::Utf8), - FieldType::Text => Some(arrow::datatypes::DataType::LargeUtf8), - FieldType::Binary => Some(arrow::datatypes::DataType::LargeBinary), - FieldType::Decimal => None, - FieldType::Timestamp => Some(arrow::datatypes::DataType::Timestamp( - arrow::datatypes::TimeUnit::Nanosecond, - None, - )), - FieldType::Date => Some(arrow::datatypes::DataType::Date32), - FieldType::Json => Some(arrow::datatypes::DataType::Utf8), - FieldType::Point => None, - FieldType::Duration => Some(arrow::datatypes::DataType::Duration( - arrow::datatypes::TimeUnit::Nanosecond, - )), - } -} - -fn field_definition_to_arrow(field_definition: FieldDefinition) -> Option { - field_type_to_arrow(field_definition.typ).map(|data_type| { - arrow::datatypes::Field::new(field_definition.name, data_type, field_definition.nullable) - }) -} - -pub fn schema_to_arrow(fields: Vec) -> (arrow::datatypes::Schema, FieldsAndPk) { - let arrow_fields: Vec<_> = fields - .iter() - .cloned() - .filter_map(field_definition_to_arrow) - .collect(); - let arrow_schema = arrow::datatypes::Schema::new(arrow_fields); - - let fields = fields - .into_iter() - .filter_map(|field| field_type_to_arrow(field.typ).map(|_| field)) - .collect(); - - (arrow_schema, (fields, vec![])) -} - -fn fields_to_arrow<'a, F: IntoIterator>( - fields: F, - count: usize, - field_type: FieldType, -) -> Arc { - match field_type { - FieldType::UInt => { - let mut builder = arrow::array::UInt64Array::builder(count); - for field in fields { - match field { - Field::UInt(value) => builder.append_value(*value), - Field::Null => builder.append_null(), - _ => panic!("Unexpected field type"), - } - } - Arc::new(builder.finish()) - } - FieldType::U128 => panic!("Unexpected field type"), - FieldType::Int => { - let mut builder = arrow::array::Int64Array::builder(count); - for field in fields { - match field { - Field::Int(value) => builder.append_value(*value), - Field::Null => builder.append_null(), - _ => panic!("Unexpected field type"), - } - } - Arc::new(builder.finish()) - } - FieldType::Int8 => { - let mut builder = arrow::array::Int64Array::builder(count); - for field in fields { - match field { - Field::Int(value) => builder.append_value(*value), - Field::Int8(value) => builder.append_value(*value as i64), - Field::Null => builder.append_null(), - _ => panic!("Unexpected field type"), - } - } - Arc::new(builder.finish()) - } - FieldType::I128 => panic!("Unexpected field type"), - FieldType::Float => { - let mut builder = arrow::array::Float64Array::builder(count); - for field in fields { - match field { - Field::Float(value) => builder.append_value(value.0), - Field::Null => builder.append_null(), - _ => panic!("Unexpected field type"), - } - } - Arc::new(builder.finish()) - } - FieldType::Boolean => { - let mut builder = arrow::array::BooleanArray::builder(count); - for field in fields { - match field { - Field::Boolean(value) => builder.append_value(*value), - Field::Null => builder.append_null(), - _ => panic!("Unexpected field type"), - } - } - Arc::new(builder.finish()) - } - FieldType::String => { - let mut builder = arrow::array::StringBuilder::new(); - for field in fields { - match field { - Field::String(value) => builder.append_value(value), - Field::Null => builder.append_null(), - _ => panic!("Unexpected field type"), - } - } - Arc::new(builder.finish()) - } - FieldType::Text => { - let mut builder = arrow::array::LargeStringBuilder::new(); - for field in fields { - match field { - Field::Text(value) => builder.append_value(value), - Field::Null => builder.append_null(), - _ => panic!("Unexpected field type"), - } - } - Arc::new(builder.finish()) - } - FieldType::Binary => { - let mut builder = arrow::array::LargeBinaryBuilder::new(); - for field in fields { - match field { - Field::Binary(value) => builder.append_value(value), - Field::Null => builder.append_null(), - _ => panic!("Unexpected field type"), - } - } - Arc::new(builder.finish()) - } - FieldType::Decimal => panic!("Decimal not supported"), - FieldType::Timestamp => { - let mut builder = arrow::array::TimestampNanosecondArray::builder(count); - for field in fields { - match field { - Field::Timestamp(value) => builder - .append_value(value.timestamp_nanos_opt().expect( - "value can not be represented in a timestamp with nanosecond precision.", - )), - Field::Null => builder.append_null(), - _ => panic!("Unexpected field type"), - } - } - Arc::new(builder.finish()) - } - FieldType::Date => { - let mut builder = arrow::array::Date32Array::builder(count); - for field in fields { - match field { - Field::Date(value) => builder.append_value(value.num_days_from_ce()), - Field::Null => builder.append_null(), - _ => panic!("Unexpected field type"), - } - } - Arc::new(builder.finish()) - } - FieldType::Json => { - let mut builder = arrow::array::StringBuilder::new(); - for field in fields { - match field { - Field::Json(value) => builder.append_value(json_to_string(value)), - Field::Null => builder.append_null(), - _ => panic!("Unexpected field type"), - } - } - Arc::new(builder.finish()) - } - FieldType::Point => panic!("Point not supported"), - FieldType::Duration => { - let mut builder = arrow::array::DurationNanosecondArray::builder(count); - for field in fields { - match field { - Field::Duration(value) => builder.append_value(value.0.as_nanos() as i64), - Field::Null => builder.append_null(), - _ => panic!("Unexpected field type"), - } - } - Arc::new(builder.finish()) - } - } -} - -pub fn records_to_arrow( - records: &[Vec], - fields: Vec, -) -> arrow::record_batch::RecordBatch { - let mut columns = vec![]; - for (index, field) in fields.iter().enumerate() { - if field_type_to_arrow(field.typ).is_some() { - let fields = records.iter().map(|record| &record[index]); - let column = fields_to_arrow(fields, records.len(), field.typ); - columns.push(column); - } - } - - let (schema, _) = schema_to_arrow(fields); - - arrow::record_batch::RecordBatch::try_new(Arc::new(schema), columns) - .expect("BUG in records_to_arrow") -} diff --git a/dozer-ingestion/tests/test_suite/connectors/object_store/local_storage.rs b/dozer-ingestion/tests/test_suite/connectors/object_store/local_storage.rs deleted file mode 100644 index 6a21882a6a..0000000000 --- a/dozer-ingestion/tests/test_suite/connectors/object_store/local_storage.rs +++ /dev/null @@ -1,104 +0,0 @@ -use dozer_ingestion_object_store::connector::ObjectStoreConnector; - -use dozer_ingestion_connector::{ - async_trait, - dozer_types::{ - arrow, - models::ingestion_types::{LocalDetails, LocalStorage, ParquetConfig, Table, TableConfig}, - types::Field, - }, -}; -use tempfile::TempDir; - -use crate::test_suite::{DataReadyConnectorTest, FieldsAndPk, InsertOnlyConnectorTest}; - -use super::arrow::{record_batch_with_all_supported_data_types, records_to_arrow, schema_to_arrow}; - -pub struct LocalStorageObjectStoreConnectorTest { - _temp_dir: TempDir, -} - -#[async_trait] -impl DataReadyConnectorTest for LocalStorageObjectStoreConnectorTest { - type Connector = ObjectStoreConnector; - - async fn new() -> (Self, Self::Connector) { - let record_batch = record_batch_with_all_supported_data_types(); - let (temp_dir, connector) = create_connector("sample".to_string(), &record_batch); - ( - Self { - _temp_dir: temp_dir, - }, - connector, - ) - } -} - -#[async_trait] -impl InsertOnlyConnectorTest for LocalStorageObjectStoreConnectorTest { - type Connector = ObjectStoreConnector; - - async fn new( - schema_name: Option, - table_name: String, - schema: FieldsAndPk, - records: Vec>, - ) -> Option<(Self, Self::Connector, FieldsAndPk)> { - if schema_name.is_some() { - return None; - } - - let record_batch = records_to_arrow(&records, schema.0.clone()); - - let (temp_dir, connector) = create_connector(table_name, &record_batch); - - let (_, schema) = schema_to_arrow(schema.0); - - Some(( - Self { - _temp_dir: temp_dir, - }, - connector, - schema, - )) - } -} - -fn create_connector( - table_name: String, - record_batch: &arrow::record_batch::RecordBatch, -) -> (TempDir, ObjectStoreConnector) { - let temp_dir = TempDir::new().expect("Failed to create temp dir"); - let path = temp_dir.path().join(&table_name); - std::fs::create_dir_all(&path).expect("Failed to create dir"); - - let file = std::fs::File::create(path.join("0.parquet")).expect("Failed to create file"); - let props = parquet::file::properties::WriterProperties::builder().build(); - let mut writer = parquet::arrow::arrow_writer::ArrowWriter::try_new( - file, - record_batch.schema(), - Some(props), - ) - .expect("Failed to create writer"); - writer - .write(record_batch) - .expect("Failed to write record batch"); - writer.close().expect("Failed to close writer"); - - let local_storage = LocalStorage { - details: LocalDetails { - path: temp_dir.path().to_str().expect("Non-UTF8 path").to_string(), - }, - tables: vec![Table { - config: TableConfig::Parquet(ParquetConfig { - path: table_name.to_string(), - extension: ".parquet".to_string(), - marker_extension: None, - }), - name: table_name, - }], - }; - let connector = ObjectStoreConnector::new(local_storage); - - (temp_dir, connector) -} diff --git a/dozer-ingestion/tests/test_suite/connectors/object_store/mod.rs b/dozer-ingestion/tests/test_suite/connectors/object_store/mod.rs deleted file mode 100644 index aee220ddf6..0000000000 --- a/dozer-ingestion/tests/test_suite/connectors/object_store/mod.rs +++ /dev/null @@ -1,4 +0,0 @@ -mod arrow; -mod local_storage; - -pub use local_storage::LocalStorageObjectStoreConnectorTest; diff --git a/dozer-ingestion/tests/test_suite/connectors/postgres.rs b/dozer-ingestion/tests/test_suite/connectors/postgres.rs deleted file mode 100644 index 681cc99b95..0000000000 --- a/dozer-ingestion/tests/test_suite/connectors/postgres.rs +++ /dev/null @@ -1,187 +0,0 @@ -use dozer_ingestion_connector::{async_trait, dozer_types::types::Field}; -use dozer_ingestion_postgres::{ - connection::{client::Client, helper::connect}, - connector::{PostgresConfig, PostgresConnector}, - tokio_postgres, -}; -use dozer_utils::{process::run_docker_compose, Cleanup}; -use tempfile::TempDir; - -use crate::test_suite::{ - records::Operation, CudConnectorTest, DataReadyConnectorTest, FieldsAndPk, - InsertOnlyConnectorTest, -}; - -use super::sql::{ - create_schema, create_table, create_table_with_all_supported_data_types, insert_record, - operation_to_sql, schema_to_sql, -}; - -pub struct PostgresConnectorTest { - config: tokio_postgres::Config, - schema_name: Option, - table_name: String, - schema: FieldsAndPk, - _cleanup: Cleanup, - _temp_dir: TempDir, -} - -#[async_trait] -impl DataReadyConnectorTest for PostgresConnectorTest { - type Connector = PostgresConnector; - - async fn new() -> (Self, Self::Connector) { - let (mut client, connector_test, connector) = create_postgres_server().await; - client - .batch_execute(&create_table_with_all_supported_data_types("test_table")) - .await - .unwrap(); - - (connector_test, connector) - } -} - -#[async_trait] -impl InsertOnlyConnectorTest for PostgresConnectorTest { - type Connector = PostgresConnector; - - async fn new( - schema_name: Option, - table_name: String, - schema: FieldsAndPk, - records: Vec>, - ) -> Option<(Self, Self::Connector, FieldsAndPk)> { - let (mut client, mut connector_test, connector) = create_postgres_server().await; - - let (actual_schema, _) = schema_to_sql(schema.clone()); - - if let Some(schema_name) = &schema_name { - client - .batch_execute(&create_schema(schema_name)) - .await - .expect("Failed to create schema"); - } - - let query = create_table(schema_name.as_deref(), &table_name, &actual_schema); - client - .batch_execute(&query) - .await - .expect("Failed to create table"); - - for record in records { - let query = insert_record(schema_name.as_deref(), &table_name, &record, &schema.0); - client - .batch_execute(&query) - .await - .expect("Failed to insert record"); - } - - connector_test.schema_name = schema_name; - connector_test.table_name = table_name; - connector_test.schema = schema; - - Some((connector_test, connector, actual_schema)) - } -} - -#[async_trait] -impl CudConnectorTest for PostgresConnectorTest { - async fn start_cud(&self, operations: Vec) { - let mut client = connect(self.config.clone()).await.unwrap(); - let schema_name = self.schema_name.clone(); - let table_name = self.table_name.clone(); - let schema = self.schema.clone(); - tokio::spawn(async move { - for operation in operations { - client - .batch_execute(&operation_to_sql( - schema_name.as_deref(), - &table_name, - &operation, - &schema, - )) - .await - .unwrap(); - } - }); - } -} - -async fn create_postgres_server() -> (Client, PostgresConnectorTest, PostgresConnector) { - let host = "localhost"; - let port = 5432; - let user = "postgres"; - let password = "postgres"; - let dbname = "dozer-test"; - - let temp_dir = tempfile::Builder::new() - .prefix("postgres") - .tempdir() - .expect("Failed to create temp dir"); - let docker_compose_path = temp_dir.path().join("docker-compose.yaml"); - std::fs::write(&docker_compose_path, DOCKER_COMPOSE_YAML) - .expect("Failed to write docker compose file"); - let cleanup = run_docker_compose(&docker_compose_path, "dozer-wait-for-connections-healthy"); - - let mut config = tokio_postgres::Config::default(); - config - .host(host) - .port(port) - .user(user) - .password(password) - .dbname(dbname); - - let connector = PostgresConnector::new( - PostgresConfig { - name: "postgres_connector_test".to_string(), - config: config.clone(), - schema: None, - batch_size: 1000, - }, - None, - ) - .unwrap(); - - let client = connect(config.clone()).await.unwrap(); - - ( - client, - PostgresConnectorTest { - config, - schema_name: Default::default(), - table_name: Default::default(), - schema: Default::default(), - _cleanup: cleanup, - _temp_dir: temp_dir, - }, - connector, - ) -} - -const DOCKER_COMPOSE_YAML: &str = r#"version: '2.4' -services: - postgres: - container_name: postgres - image: debezium/postgres:13 - ports: - - target: 5432 - published: 5432 - environment: - - POSTGRES_DB=dozer-test - - POSTGRES_USER=postgres - - POSTGRES_PASSWORD=postgres - - ALLOW_IP_RANGE=0.0.0.0/0 - healthcheck: - test: - - CMD-SHELL - - pg_isready -U postgres -h 0.0.0.0 -d dozer-test - interval: 5s - timeout: 5s - retries: 5 - dozer-wait-for-connections-healthy: - image: alpine - command: echo 'All connections are healthy' - depends_on: - postgres: - condition: service_healthy -"#; diff --git a/dozer-ingestion/tests/test_suite/connectors/sql.rs b/dozer-ingestion/tests/test_suite/connectors/sql.rs deleted file mode 100644 index 3608f1214c..0000000000 --- a/dozer-ingestion/tests/test_suite/connectors/sql.rs +++ /dev/null @@ -1,359 +0,0 @@ -use dozer_ingestion_connector::dozer_types::{ - json_types::json_to_string, - types::{Field, FieldDefinition, FieldType}, -}; - -use crate::test_suite::{records::Operation, FieldsAndPk}; - -pub fn create_table_with_all_supported_data_types(table_name: &str) -> String { - format!( - r#" - CREATE TABLE {table_name} ( - boolean BOOLEAN NOT NULL, - boolean_null BOOLEAN, - int2 INT2 NOT NULL, - int2_null INT2, - int4 INT4 NOT NULL, - int4_null INT4, - int8 INT8 NOT NULL, - int8_null INT8, - char CHAR NOT NULL, - char_null CHAR, - text TEXT NOT NULL, - text_null TEXT, - varchar VARCHAR NOT NULL, - varchar_null VARCHAR, - bpchar BPCHAR NOT NULL, - bpchar_null BPCHAR, - float4 FLOAT4 NOT NULL, - float4_null FLOAT4, - float8 FLOAT8 NOT NULL, - float8_null FLOAT8, - bytea BYTEA NOT NULL, - bytea_null BYTEA, - timestamp TIMESTAMP NOT NULL, - timestamp_null TIMESTAMP, - timestamptz TIMESTAMPTZ NOT NULL, - timestamptz_null TIMESTAMPTZ, - numeric NUMERIC NOT NULL, - numeric_null NUMERIC, - json JSON NOT NULL, - json_null JSON, - jsonb JSONB NOT NULL, - jsonb_null JSONB, - json_array JSON[] NOT NULL, - json_array_null JSON[], - jsonb_array JSONB[] NOT NULL, - jsonb_array_null JSONB[], - date DATE NOT NULL, - date_null DATE, - point POINT NOT NULL, - point_null POINT, - uuid UUID NOT NULL, - uuid_null UUID - ); - INSERT INTO {table_name} VALUES ( - false, - false, - 0, - 0, - 0, - 0, - 0, - 0, - '', - '', - '', - '', - '', - '', - '', - '', - 0.0, - 0.0, - 0.0, - 0.0, - '', - '', - '1970-01-01 00:00:00', - '1970-01-01 00:00:00', - '1970-01-01 00:00:00', - '1970-01-01 00:00:00', - 0, - 0, - '{{}}'::json, - '{{}}'::json, - '{{}}'::jsonb, - '{{}}'::jsonb, - ARRAY[]::json[], - ARRAY[]::json[], - ARRAY[]::jsonb[], - ARRAY[]::jsonb[], - '1970-01-01', - '1970-01-01', - '(0,0)', - '(0,0)', - 'a0eebc99-9c0b-4ef8-bb6d-6bb9bd380a11'::UUID, - 'a0eebc99-9c0b-4ef8-bb6d-6bb9bd380a11'::UUID - ); - INSERT INTO {table_name} VALUES ( - true, - null, - 1, - null, - 1, - null, - 1, - null, - '1', - null, - '1', - null, - '1', - null, - '1', - null, - 1.0, - null, - 1.0, - null, - '1', - null, - '1970-01-01 00:00:00', - null, - '1970-01-01 00:00:00', - null, - 1, - null, - '{{ "1": 1 }}'::json, - null, - '{{ "1": 1 }}'::jsonb, - null, - ARRAY['{{ "1": 1 }}']::json[], - null, - ARRAY['{{ "1": 1 }}']::jsonb[], - null, - '1970-01-01', - null, - '(1,1)', - null, - 'a0eebc99-9c0b-4ef8-bb6d-6bb9bd380a11'::UUID, - null - ); - "#, - ) -} - -pub fn create_schema(name: &str) -> String { - format!( - r#" - CREATE SCHEMA {name}; - "#, - name = name - ) -} - -fn field_type_to_sql(field_type: FieldType) -> Option { - match field_type { - FieldType::UInt => None, - FieldType::U128 => None, - FieldType::Int => Some("INT8".to_string()), - FieldType::Int8 => Some("INT8".to_string()), - FieldType::I128 => None, - FieldType::Float => Some("FLOAT8".to_string()), - FieldType::Boolean => Some("BOOLEAN".to_string()), - FieldType::String => Some("TEXT".to_string()), - FieldType::Text => None, - FieldType::Binary => Some("BYTEA".to_string()), - FieldType::Decimal => Some("NUMERIC".to_string()), - FieldType::Timestamp => Some("TIMESTAMP".to_string()), - FieldType::Date => Some("DATE".to_string()), - FieldType::Json => Some("JSONB".to_string()), - FieldType::Point => Some("POINT".to_string()), - FieldType::Duration => Some("DURATION".to_string()), - } -} - -fn field_definition_to_sql(field_definition: &FieldDefinition) -> Option { - let field_type = field_type_to_sql(field_definition.typ)?; - let nullable = if field_definition.nullable { - "" - } else { - " NOT NULL" - }; - Some(format!( - "{} {}{}", - field_definition.name, field_type, nullable - )) -} - -pub fn schema_to_sql((fields, primary_index): FieldsAndPk) -> (FieldsAndPk, String) { - let mut actual_fields = vec![]; - let mut actual_primary_index = vec![]; - - let mut fields_sql = vec![]; - for (index, field) in fields.into_iter().enumerate() { - let is_primary_key = primary_index.iter().any(|i| *i == index); - - let Some(mut field_sql) = field_definition_to_sql(&field) else { - continue; - }; - if is_primary_key { - actual_primary_index.push(actual_fields.len()); - field_sql.push_str(" PRIMARY KEY"); - } - actual_fields.push(field); - fields_sql.push(field_sql); - } - ((actual_fields, actual_primary_index), fields_sql.join(", ")) -} - -fn full_table_name(schema_name: Option<&str>, table_name: &str) -> String { - match schema_name { - Some(schema_name) => format!("{}.{}", schema_name, table_name), - None => table_name.to_string(), - } -} - -pub fn create_table(schema_name: Option<&str>, table_name: &str, schema: &FieldsAndPk) -> String { - format!( - r#" - CREATE TABLE {table_name} ({fields}); - "#, - table_name = full_table_name(schema_name, table_name), - fields = schema_to_sql(schema.clone()).1 - ) -} - -fn field_to_sql(field: &Field) -> String { - match field { - Field::UInt(i) => i.to_string(), - Field::U128(i) => i.to_string(), - Field::Int(i) => i.to_string(), - Field::Int8(i) => i.to_string(), - Field::I128(i) => i.to_string(), - Field::Float(f) => f.to_string(), - Field::Boolean(b) => b.to_string(), - Field::String(s) => s.to_string(), - Field::Text(s) => s.to_string(), - Field::Binary(b) => format!("'\\x{}'", hex::encode(b)), - Field::Decimal(d) => d.to_string(), - Field::Timestamp(t) => format!("'{}'", t), - Field::Date(d) => format!("'{}'", d), - Field::Json(b) => format!("'{}'::jsonb", json_to_string(b)), - Field::Point(p) => format!("'({},{})'", p.0.x(), p.0.y()), - Field::Duration(_) => field.to_string(), - Field::Null => "NULL".to_string(), - } -} - -pub fn insert_record( - schema_name: Option<&str>, - table_name: &str, - record: &[Field], - fields: &[FieldDefinition], -) -> String { - let mut values_sql = vec![]; - for (field, value) in fields.iter().zip(record.iter()) { - if field_type_to_sql(field.typ).is_none() { - continue; - } - values_sql.push(field_to_sql(value)); - } - - format!( - r#" - INSERT INTO {table_name} VALUES ({values}); - "#, - table_name = full_table_name(schema_name, table_name), - values = values_sql.join(", "), - ) -} - -fn update_record( - schema_name: Option<&str>, - table_name: &str, - old: &[Field], - new: &[Field], - (fields, primary_key): &FieldsAndPk, -) -> String { - let mut set = vec![]; - let mut where_ = vec![]; - for (index, ((old_field, new_field), field_definition)) in - old.iter().zip(new).zip(fields).enumerate() - { - if field_type_to_sql(field_definition.typ).is_none() { - continue; - } - - set.push(format!( - "{} = {}", - field_definition.name, - field_to_sql(new_field) - )); - - let is_primary_key = primary_key.iter().any(|i| *i == index); - - if is_primary_key { - where_.push(format!( - "{} = {}", - field_definition.name, - field_to_sql(old_field) - )); - } - } - - format!( - r#" - UPDATE {table_name} SET {set} WHERE {where}; - "#, - table_name = full_table_name(schema_name, table_name), - set = set.join(", "), - where = where_.join(" AND "), - ) -} - -fn delete_record( - schema_name: Option<&str>, - table_name: &str, - record: &[Field], - (fields, primary_key): &FieldsAndPk, -) -> String { - let mut where_ = vec![]; - for (index, (field, field_definition)) in record.iter().zip(fields).enumerate() { - if field_type_to_sql(field_definition.typ).is_none() { - continue; - } - - let is_primary_key = primary_key.iter().any(|i| *i == index); - - if is_primary_key { - where_.push(format!( - "{} = {}", - field_definition.name, - field_to_sql(field) - )); - } - } - - format!( - r#" - DELETE FROM {table_name} WHERE {where}; - "#, - table_name = full_table_name(schema_name, table_name), - where = where_.join(" AND "), - ) -} - -pub fn operation_to_sql( - schema_name: Option<&str>, - table_name: &str, - operation: &Operation, - schema: &FieldsAndPk, -) -> String { - match operation { - Operation::Insert { new } => insert_record(schema_name, table_name, new, &schema.0), - Operation::Update { new, old } => update_record(schema_name, table_name, old, new, schema), - Operation::Delete { old } => delete_record(schema_name, table_name, old, schema), - } -} diff --git a/dozer-ingestion/tests/test_suite/data.rs b/dozer-ingestion/tests/test_suite/data.rs deleted file mode 100644 index fa3c9d6ceb..0000000000 --- a/dozer-ingestion/tests/test_suite/data.rs +++ /dev/null @@ -1,71 +0,0 @@ -use dozer_ingestion_connector::dozer_types::types::{Field, FieldDefinition, FieldType}; - -use super::{records::Operation, FieldsAndPk}; - -pub fn records_without_primary_key() -> (FieldsAndPk, Vec>) { - let fields = vec![ - FieldDefinition { - name: "int".to_string(), - typ: FieldType::Int, - nullable: false, - source: Default::default(), - description: None, - }, - FieldDefinition { - name: "uint".to_string(), - typ: FieldType::UInt, - nullable: false, - source: Default::default(), - description: None, - }, - ]; - - let records = vec![vec![Field::Int(0), Field::UInt(0)]]; - - ((fields, vec![]), records) -} - -pub fn records_with_primary_key() -> (FieldsAndPk, Vec>) { - let ((fields, _), records) = records_without_primary_key(); - ((fields, vec![0]), records) -} - -pub fn cud_operations() -> (FieldsAndPk, Vec) { - let (schema, records) = records_with_primary_key(); - let updated_record = vec![Field::Int(1), Field::UInt(1)]; - let operations = vec![ - Operation::Insert { - new: records[0].clone(), - }, - Operation::Update { - old: records[0].clone(), - new: updated_record.clone(), - }, - Operation::Delete { - old: updated_record, - }, - ]; - (schema, operations) -} - -pub fn reorder( - fields: &[FieldDefinition], - pk: &[usize], - operations: &[Operation], -) -> (FieldsAndPk, Vec) { - let reversed_fields = fields.iter().rev().cloned().collect(); - let reversed_pk = pk.iter().map(|pk| fields.len() - 1 - pk).collect(); - - let mut reversed_operations = operations.to_vec(); - for op in reversed_operations.iter_mut() { - match op { - Operation::Insert { new } => new.reverse(), - Operation::Update { old, new } => { - old.reverse(); - new.reverse(); - } - Operation::Delete { old } => old.reverse(), - } - } - ((reversed_fields, reversed_pk), reversed_operations) -} diff --git a/dozer-ingestion/tests/test_suite/mod.rs b/dozer-ingestion/tests/test_suite/mod.rs deleted file mode 100644 index 58ee1af7e5..0000000000 --- a/dozer-ingestion/tests/test_suite/mod.rs +++ /dev/null @@ -1,56 +0,0 @@ -use dozer_ingestion_connector::{ - async_trait, - dozer_types::types::{Field, FieldDefinition}, - Connector, -}; - -#[async_trait] -pub trait DataReadyConnectorTest: Send + Sized + 'static { - type Connector: Connector; - - async fn new() -> (Self, Self::Connector); -} - -pub type FieldsAndPk = (Vec, Vec); - -#[async_trait] -pub trait InsertOnlyConnectorTest: Send + Sized + 'static { - type Connector: Connector; - - /// Creates a connector which contains a table whose name is `table_name`. - /// - /// The test should try its best to create a table with the given `schema_name`, `table_name` and `schema`. - /// If any of the field in `schema` is not supported, it can skip that field. - /// - /// If `schema_name` or `schema.primary_index` is not supported in this connector, `None` should be returned. - /// - /// The actually created schema should be returned. - async fn new( - schema_name: Option, - table_name: String, - schema: FieldsAndPk, - records: Vec>, - ) -> Option<(Self, Self::Connector, FieldsAndPk)>; -} - -#[async_trait] -pub trait CudConnectorTest: InsertOnlyConnectorTest { - /// Spawns a thread to feed cud operations to connector. - async fn start_cud(&self, operations: Vec); -} - -mod basic; -mod connectors; -mod data; -mod records; - -pub use basic::{ - run_test_suite_basic_cud, run_test_suite_basic_data_ready, run_test_suite_basic_insert_only, -}; - -#[cfg(feature = "mongodb")] -pub use connectors::MongodbConnectorTest; - -#[cfg(feature = "datafusion")] -pub use connectors::LocalStorageObjectStoreConnectorTest; -pub use connectors::PostgresConnectorTest; diff --git a/dozer-ingestion/tests/test_suite/records.rs b/dozer-ingestion/tests/test_suite/records.rs deleted file mode 100644 index 8b709efa2d..0000000000 --- a/dozer-ingestion/tests/test_suite/records.rs +++ /dev/null @@ -1,48 +0,0 @@ -use std::collections::HashMap; - -use dozer_ingestion_connector::dozer_types::types::Field; - -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum Operation { - Insert { new: Vec }, - Update { old: Vec, new: Vec }, - Delete { old: Vec }, -} - -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct Records { - primary_index: Vec, - data: HashMap, Vec>, -} - -impl Records { - pub fn new(primary_index: Vec) -> Self { - Self { - primary_index, - data: HashMap::new(), - } - } - - pub fn append_operation(&mut self, operation: Operation) { - match operation { - Operation::Insert { new } => { - let primary_key = get_primary_key(&new, &self.primary_index); - assert!(self.data.insert(primary_key, new).is_none()); - } - Operation::Update { old, new } => { - let old_primary_key = get_primary_key(&old, &self.primary_index); - assert!(self.data.remove(&old_primary_key).is_some()); - let new_primary_key = get_primary_key(&new, &self.primary_index); - assert!(self.data.insert(new_primary_key, new).is_none()); - } - Operation::Delete { old } => { - let primary_key = get_primary_key(&old, &self.primary_index); - assert!(self.data.remove(&primary_key).is_some()); - } - } - } -} - -fn get_primary_key(record: &[Field], primary_index: &[usize]) -> Vec { - primary_index.iter().map(|i| record[*i].clone()).collect() -} diff --git a/dozer-ingestion/webhook/Cargo.toml b/dozer-ingestion/webhook/Cargo.toml deleted file mode 100644 index ddf467775a..0000000000 --- a/dozer-ingestion/webhook/Cargo.toml +++ /dev/null @@ -1,16 +0,0 @@ -[package] -name = "dozer-ingestion-webhook" -version = "0.1.0" -edition = "2021" -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-ingestion-connector = { path = "../connector" } -actix-web = "4.4.1" -env_logger = "0.11.1" - -[dev-dependencies] -tokio = { version = "1.0", features = ["full", "test-util"] } -reqwest = { version = "0.11.20", features = ["json", "blocking"] } diff --git a/dozer-ingestion/webhook/src/connector.rs b/dozer-ingestion/webhook/src/connector.rs deleted file mode 100644 index 84d9f08764..0000000000 --- a/dozer-ingestion/webhook/src/connector.rs +++ /dev/null @@ -1,157 +0,0 @@ -use crate::{server::WebhookServer, util::extract_source_schema, Error}; -use dozer_ingestion_connector::{ - async_trait, - dozer_types::{ - self, errors::internal::BoxedError, models::ingestion_types::WebhookConfig, - node::OpIdentifier, - }, - utils::TableNotFound, - Connector, Ingestor, SourceSchema, SourceSchemaResult, TableIdentifier, TableInfo, -}; -use std::{collections::HashMap, sync::Arc, vec}; - -#[derive(Debug)] -pub struct WebhookConnector { - pub config: WebhookConfig, -} - -impl WebhookConnector { - pub fn new(config: WebhookConfig) -> Self { - Self { config } - } - - fn get_all_schemas(&self) -> Result, Error> { - let mut result: HashMap = HashMap::new(); - let config = &self.config; - for endpoint in &config.endpoints { - let schemas = extract_source_schema(endpoint.schema.clone()); - for (key, value) in schemas { - result.insert(key, value); - } - } - - Ok(result) - } -} - -#[async_trait] -impl Connector for WebhookConnector { - fn types_mapping() -> Vec<(String, Option)> - where - Self: Sized, - { - todo!() - } - - async fn validate_connection(&mut self) -> Result<(), BoxedError> { - Ok(()) - } - - async fn list_tables(&mut self) -> Result, BoxedError> { - Ok(self - .get_all_schemas()? - .into_keys() - .map(TableIdentifier::from_table_name) - .collect()) - } - - async fn validate_tables(&mut self, tables: &[TableIdentifier]) -> Result<(), BoxedError> { - let schemas = self.get_all_schemas()?; - for table in tables { - if !schemas - .iter() - .any(|(name, _)| name == &table.name && table.schema.is_none()) - { - return Err(TableNotFound { - schema: table.schema.clone(), - name: table.name.clone(), - } - .into()); - } - } - Ok(()) - } - - async fn list_columns( - &mut self, - tables: Vec, - ) -> Result, BoxedError> { - let schemas = self.get_all_schemas()?; - let mut result: Vec = vec![]; - for table in tables { - let source_schema_option = schemas.get(table.name.as_str()); - match source_schema_option { - Some(source_schema) => { - let column_names = source_schema - .schema - .fields - .iter() - .map(|field| field.name.clone()) - .collect(); - if result - .iter() - .any(|table_info| table_info.name == table.name) - { - continue; - } - result.push(TableInfo { - schema: table.schema, - name: table.name, - column_names, - }) - } - None => { - return Err(TableNotFound { - schema: table.schema.clone(), - name: table.name.clone(), - } - .into()); - } - } - } - Ok(result) - } - - async fn get_schemas( - &mut self, - table_infos: &[TableInfo], - ) -> Result, BoxedError> { - let schemas = self.get_all_schemas()?; - let mut result = vec![]; - for table in table_infos { - let table_name = table.name.clone(); - let schema = schemas.get(table_name.as_str()); - match schema { - Some(schema) => { - result.push(Ok(schema.clone())); - } - None => { - result.push(Err(TableNotFound { - schema: table.schema.clone(), - name: table.name.clone(), - } - .into())); - } - } - } - Ok(result) - } - - async fn serialize_state(&self) -> Result, BoxedError> { - Ok(vec![]) - } - - async fn start( - &mut self, - ingestor: &Ingestor, - tables: Vec, - _last_checkpoint: Option, - ) -> Result<(), BoxedError> { - let config = self.config.clone(); - let server = WebhookServer::new(config); - server - .start(Arc::new(ingestor.to_owned()), tables) - .await - .map_err(Into::into) - } -} diff --git a/dozer-ingestion/webhook/src/lib.rs b/dozer-ingestion/webhook/src/lib.rs deleted file mode 100644 index ee1a902034..0000000000 --- a/dozer-ingestion/webhook/src/lib.rs +++ /dev/null @@ -1,27 +0,0 @@ -use std::{net::AddrParseError, path::PathBuf}; - -use dozer_ingestion_connector::dozer_types::{ - serde_json, - thiserror::{self, Error}, -}; -pub mod connector; -mod server; -#[cfg(test)] -mod tests; -mod util; -#[derive(Debug, Error)] - -pub enum Error { - #[error("cannot read file {0:?}: {1}")] - CannotReadFile(PathBuf, #[source] std::io::Error), - #[error("serde json error: {0}")] - SerdeJson(#[from] serde_json::Error), - #[error("arrow error: {0}")] - AddrParse(#[from] AddrParseError), - #[error("default adapter cannot handle arrow ingest message")] - SchemaNotFound(String), - #[error("field {0} not found in schema")] - FieldNotFound(String), - #[error("actix web start error: {0}")] - ActixWebStartError(#[from] std::io::Error), -} diff --git a/dozer-ingestion/webhook/src/server.rs b/dozer-ingestion/webhook/src/server.rs deleted file mode 100644 index b226c3cc31..0000000000 --- a/dozer-ingestion/webhook/src/server.rs +++ /dev/null @@ -1,190 +0,0 @@ -use crate::{ - util::{extract_source_schema, map_record}, - Error, -}; -use actix_web::{ - web::{self, Data}, - App, HttpRequest, HttpServer, Responder, -}; -use dozer_ingestion_connector::{ - dozer_types::{ - models::ingestion_types::{IngestionMessage, WebhookConfig, WebhookVerb}, - serde_json, - types::{Operation, Record}, - }, - Ingestor, SourceSchema, TableInfo, -}; -use std::{collections::HashMap, sync::Arc}; - -pub(crate) struct WebhookServer { - config: WebhookConfig, -} - -impl WebhookServer { - pub(crate) fn new(config: WebhookConfig) -> Self { - Self { config } - } - - pub(crate) async fn start( - &self, - ingestor: Arc, - tables: Vec, - ) -> Result<(), Error> { - let config = self.config.clone(); - // Clone or extract necessary data from `self` - let server = HttpServer::new(move || { - let mut app = App::new(); - - for endpoint in config.endpoints.iter() { - let endpoint_data = endpoint.clone(); - let source_schema_dict = extract_source_schema(endpoint_data.to_owned().schema); - let tables = tables.clone(); - let mut app_resource = web::resource(endpoint_data.path) - .app_data(web::Data::new(Arc::clone(&ingestor))) - .app_data(web::Data::new(source_schema_dict)) - .app_data(web::Data::new(tables)); - for verb in &endpoint.verbs { - app_resource = match verb { - WebhookVerb::POST => app_resource.route(web::post().to(Self::post_handler)), - WebhookVerb::DELETE => { - app_resource.route(web::delete().to(Self::delete_handler)) - } - _ => app_resource.route(web::route().to(Self::other_handler)), - }; - } - app = app.service(app_resource); - } - app - }); - - let host = config - .host - .clone() - .unwrap_or_else(|| "127.0.0.1".to_string()); - let port = config.port.unwrap_or(8080); - let address = format!("{}:{}", host, port); // Format host and port into a single string - let server = server.bind(address)?.run(); - server.await.map_err(Into::into) - } - - fn common_handler( - tables: Data>, - schema_dict: Data>, - info: web::Json, - ) -> actix_web::Result)>, actix_web::error::Error> { - let source_schema_dict = schema_dict.get_ref(); - let info = &info.into_inner(); - - let mut result: Vec<(usize, Vec)> = vec![]; - if let serde_json::Value::Object(object) = info { - for (schema_name, values) in object.iter() { - let schema = match source_schema_dict.get(schema_name) { - Some(schema) => schema, - None => return Err(actix_web::error::ErrorBadRequest("Invalid schema name")), - }; - let records = match values.as_array() { - Some(values_arr) => values_arr - .iter() - .map(|value_element| { - let value = value_element.as_object().ok_or_else(|| { - actix_web::error::ErrorBadRequest("Invalid value") - })?; - map_record(value.to_owned(), &schema.schema) - .map_err(actix_web::error::ErrorBadRequest) - }) - .collect::, _>>(), - None => { - let value = values - .as_object() - .ok_or_else(|| actix_web::error::ErrorBadRequest("Invalid value"))?; - map_record(value.to_owned(), &schema.schema) - .map(|e| vec![e]) - .map_err(|e| actix_web::error::ErrorBadRequest(e.to_string())) - } - }?; - let table_idx = tables - .iter() - .position(|table| table.name.as_str() == schema_name) - .ok_or_else(|| actix_web::error::ErrorBadRequest("Invalid table name"))?; - result.push((table_idx, records)); - } - } else { - return Err(actix_web::error::ErrorBadRequest("Invalid JSON")); - } - Ok(result) - } - - async fn post_handler( - ingestor: Data>, - schema_dict: Data>, - tables: Data>, - info: web::Json, - ) -> actix_web::Result { - let ingestor = ingestor.get_ref(); - let records = Self::common_handler(tables, schema_dict, info)?; - - for (table_idx, records) in records { - let op: IngestionMessage = if records.len() == 1 { - IngestionMessage::OperationEvent { - table_index: table_idx, - op: Operation::Insert { - new: records[0].clone(), - }, - id: None, - } - } else { - IngestionMessage::OperationEvent { - table_index: table_idx, - op: Operation::BatchInsert { new: records }, - id: None, - } - }; - ingestor - .handle_message(op) - .await - .map_err(|e| actix_web::error::ErrorInternalServerError(format!("Error: {}", e)))?; - } - - let json_response = serde_json::json!({ - "status": "ok" - }); - - Ok(web::Json(json_response)) - } - - async fn delete_handler( - ingestor: Data>, - schema_dict: Data>, - tables: Data>, - info: web::Json, - ) -> actix_web::Result { - let ingestor = ingestor.get_ref(); - let records = Self::common_handler(tables, schema_dict, info)?; - for (table_idx, records) in records { - for record in records { - let op: IngestionMessage = IngestionMessage::OperationEvent { - table_index: table_idx, - op: Operation::Delete { old: record }, - id: None, - }; - ingestor.handle_message(op).await.map_err(|e| { - actix_web::error::ErrorInternalServerError(format!("Error: {}", e)) - })?; - } - } - - let json_response = serde_json::json!({ - "status": "ok" - }); - - Ok(web::Json(json_response)) - } - async fn other_handler(req: HttpRequest) -> actix_web::Result { - // get VERB from request - let verb = req.method().as_str(); - let json_response = serde_json::json!({ - "status": format!("{} not supported", verb) - }); - Ok(web::Json(json_response)) - } -} diff --git a/dozer-ingestion/webhook/src/tests.rs b/dozer-ingestion/webhook/src/tests.rs deleted file mode 100644 index 00020b711a..0000000000 --- a/dozer-ingestion/webhook/src/tests.rs +++ /dev/null @@ -1,222 +0,0 @@ -use crate::connector::WebhookConnector; -use dozer_ingestion_connector::{ - dozer_types::{ - json_types::json_from_str, - models::ingestion_types::{ - IngestionMessage, WebhookConfig, WebhookConfigSchemas, WebhookEndpoint, WebhookVerb, - }, - serde_json::{self, json}, - types::{Field, Record}, - }, - test_util::{create_test_runtime, spawn_connector_all_tables}, - tokio::runtime::Runtime, - IngestionIterator, -}; -use std::sync::Arc; - -fn ingest_webhook( - runtime: Arc, - port: u32, -) -> ( - IngestionIterator, - dozer_ingestion_connector::futures::future::AbortHandle, -) { - let user_schema = r#" - { - "users": { - "schema": { - "fields": [ - { - "name": "id", - "typ": "Int", - "nullable": false - }, - { - "name": "name", - "typ": "String", - "nullable": true - }, - { - "name": "json", - "typ": "Json", - "nullable": true - } - ] - } - } - } - "#; - let customer_schema = r#" - { - "customers": { - "schema": { - "fields": [ - { - "name": "id", - "typ": "Int", - "nullable": false - }, - { - "name": "name", - "typ": "String", - "nullable": true - }, - { - "name": "json", - "typ": "Json", - "nullable": true - } - ] - } - } - } - "#; - let webhook_connector = WebhookConnector::new(WebhookConfig { - port: Some(port), - host: None, - endpoints: vec![ - WebhookEndpoint { - path: "/customers".to_string(), - verbs: vec![WebhookVerb::POST, WebhookVerb::DELETE], - schema: WebhookConfigSchemas::Inline(customer_schema.to_string()), - }, - WebhookEndpoint { - path: "/users".to_string(), - verbs: vec![WebhookVerb::POST, WebhookVerb::DELETE], - schema: WebhookConfigSchemas::Inline(user_schema.to_string()), - }, - ], - }); - spawn_connector_all_tables(runtime.clone(), webhook_connector) -} - -#[test] -fn ingest_webhook_batch_insert() { - let runtime = create_test_runtime(); - let port = 58883; - let result: (IngestionIterator, _) = ingest_webhook(runtime.clone(), port); - // call http request to webhook endpoint - let client = reqwest::blocking::Client::new(); - let post_value = json!({ - "users": [ - { - "id": 1, - "name": "John Doe", - "json": { - "key": "value" - } - }, - { - "id": 2, - "name": "Jane Doe", - "json": { - "key": "value" - } - } - ] - }); - let http_result = client - .post(format!("http://127.0.0.1:{:}/users", port)) - .json(&post_value) - .send(); - assert!(http_result.is_ok()); - let response = http_result.unwrap(); - assert!(response.status().is_success()); - let mut iterator = result.0; - let msg = iterator.next().unwrap(); - - let ivalue_str = - json_from_str(&serde_json::to_string(&json!({"key": "value"})).unwrap()).unwrap(); - - let expected_fields1: Vec = vec![ - Field::Int(1), - Field::String("John Doe".to_string()), - Field::Json(ivalue_str.clone()), - ]; - let expected_record1 = Record { - values: expected_fields1, - lifetime: None, - }; - - let expected_fields2: Vec = vec![ - Field::Int(2), - Field::String("Jane Doe".to_string()), - Field::Json(ivalue_str), - ]; - let expected_record2 = Record { - values: expected_fields2, - lifetime: None, - }; - - if let IngestionMessage::OperationEvent { - table_index: _, - op, - id: _, - } = msg - { - assert_eq!( - op, - dozer_ingestion_connector::dozer_types::types::Operation::BatchInsert { - new: vec![expected_record1, expected_record2], - } - ); - } else { - panic!("Expected operation event"); - } -} - -#[test] -fn ingest_webhook_delete() { - let runtime = create_test_runtime(); - let port = 58884; - let result: (IngestionIterator, _) = ingest_webhook(runtime.clone(), port); - // call http request to webhook endpoint - let client = reqwest::blocking::Client::new(); - let delete_value = json!({ - "users": [ - { - "id": 1, - "name": "John Doe", - "json": { - "key": "value" - } - } - ] - }); - let http_result = client - .delete(format!("http://127.0.0.1:{:}/users", port)) - .json(&delete_value) - .send(); - assert!(http_result.is_ok()); - let response = http_result.unwrap(); - assert!(response.status().is_success()); - let mut iterator = result.0; - let msg = iterator.next().unwrap(); - let ivalue_str = - json_from_str(&serde_json::to_string(&json!({"key": "value"})).unwrap()).unwrap(); - - let expected_fields1: Vec = vec![ - Field::Int(1), - Field::String("John Doe".to_string()), - Field::Json(ivalue_str.clone()), - ]; - let expected_record1 = Record { - values: expected_fields1.to_owned(), - lifetime: None, - }; - - if let IngestionMessage::OperationEvent { - table_index: _, - op, - id: _, - } = msg - { - if let dozer_ingestion_connector::dozer_types::types::Operation::Delete { old } = op { - assert_eq!(old, expected_record1); - } else { - panic!("Expected delete operation"); - } - } else { - panic!("Expected operation event"); - } -} diff --git a/dozer-ingestion/webhook/src/util.rs b/dozer-ingestion/webhook/src/util.rs deleted file mode 100644 index 333a30c09a..0000000000 --- a/dozer-ingestion/webhook/src/util.rs +++ /dev/null @@ -1,146 +0,0 @@ -use crate::Error; -use dozer_ingestion_connector::{ - dozer_types::{ - chrono::{self, NaiveDate}, - json_types::json_from_str, - models::ingestion_types::WebhookConfigSchemas, - ordered_float::OrderedFloat, - rust_decimal::Decimal, - serde_json, - types::{Field, FieldType, Record, Schema}, - }, - SourceSchema, -}; -use std::{collections::HashMap, fs, path::Path}; - -pub fn extract_source_schema(input: WebhookConfigSchemas) -> HashMap { - match input { - WebhookConfigSchemas::Inline(schema_str) => { - let schemas: HashMap = serde_json::from_str(&schema_str).unwrap(); - schemas - } - WebhookConfigSchemas::Path(path) => { - let path = Path::new(&path); - let schema_str = fs::read_to_string(path).unwrap(); - let schemas: HashMap = serde_json::from_str(&schema_str).unwrap(); - schemas - } - } -} - -pub fn map_record( - rec: serde_json::map::Map, - schema: &Schema, -) -> Result { - let mut values: Vec = vec![]; - let fields = schema.fields.clone(); - for field in fields.into_iter() { - let field_name = field.name.clone(); - let field_value = rec.get(&field_name); - if !field.nullable && field_value.is_none() { - return Err(Error::FieldNotFound(field_name)); - } - match field_value { - Some(value) => match field.typ { - FieldType::String => { - let str_value: String = serde_json::from_value(value.clone())?; - let field = Field::String(str_value); - values.push(field); - } - FieldType::Int => { - let i64_value: i64 = serde_json::from_value(value.clone())?; - let field = Field::Int(i64_value); - values.push(field); - } - FieldType::Int8 => { - let i8_value: i8 = serde_json::from_value(value.clone())?; - let field = Field::Int8(i8_value); - values.push(field); - } - FieldType::Float => { - let float_value: f64 = serde_json::from_value(value.clone())?; - let field = Field::Float(OrderedFloat(float_value)); - values.push(field); - } - FieldType::Boolean => { - let bool_value: bool = serde_json::from_value(value.clone())?; - let field = Field::Boolean(bool_value); - values.push(field); - } - FieldType::Timestamp => { - let i64_value: i64 = serde_json::from_value(value.clone())?; - let timestamp_value = chrono::NaiveDateTime::from_timestamp_millis(i64_value) - .map(|t| { - Field::Timestamp( - chrono::DateTime::::from_naive_utc_and_offset( - t, - chrono::Utc, - ) - .into(), - ) - }) - .unwrap_or(Field::Null); - values.push(timestamp_value); - } - FieldType::Date => { - let i64_value: NaiveDate = serde_json::from_value(value.clone())?; - let field = Field::Date(i64_value); - values.push(field); - } - FieldType::UInt => { - let u64_value: u64 = serde_json::from_value(value.clone())?; - let field = Field::UInt(u64_value); - values.push(field); - } - FieldType::U128 => { - let u128_value: u128 = serde_json::from_value(value.clone())?; - let field = Field::U128(u128_value); - values.push(field); - } - FieldType::I128 => { - let i128_value: i128 = serde_json::from_value(value.clone())?; - let field = Field::I128(i128_value); - values.push(field); - } - FieldType::Text => { - let str_value: String = serde_json::from_value(value.clone())?; - let field = Field::Text(str_value); - values.push(field); - } - FieldType::Binary => { - let str_value: String = serde_json::from_value(value.clone())?; - let field = Field::Binary(str_value.into_bytes()); - values.push(field); - } - FieldType::Decimal => { - let str_value: String = serde_json::from_value(value.clone())?; - let decimal_value: Decimal = - Decimal::from_str_exact(str_value.as_str()).unwrap(); - let field = Field::Decimal(decimal_value); - values.push(field); - } - FieldType::Json => { - let str_value: String = serde_json::to_string(value)?; - let ivalue_str = json_from_str(str_value.as_str()).unwrap(); - let field = Field::Json(ivalue_str); - values.push(field); - } - FieldType::Point => { - values.push(Field::Null); - } - FieldType::Duration => { - values.push(Field::Null); - } - }, - None => { - let field = Field::Null; - values.push(field); - } - } - } - - Ok(Record { - values, - lifetime: None, - }) -} diff --git a/dozer-sink-clickhouse/Cargo.toml b/dozer-sink-clickhouse/Cargo.toml deleted file mode 100644 index 47b93314c9..0000000000 --- a/dozer-sink-clickhouse/Cargo.toml +++ /dev/null @@ -1,15 +0,0 @@ -[package] -name = "dozer-sink-clickhouse" -version = "0.1.0" -edition = "2021" -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-core = { path = "../dozer-core" } -dozer-types = { path = "../dozer-types" } -clickhouse-rs = { git = "https://github.com/getdozer/clickhouse-rs" } -either = "1.10.0" -chrono-tz = "0.8.6" -serde = "1.0.197" diff --git a/dozer-sink-clickhouse/src/client.rs b/dozer-sink-clickhouse/src/client.rs deleted file mode 100644 index 3df34f4c3a..0000000000 --- a/dozer-sink-clickhouse/src/client.rs +++ /dev/null @@ -1,143 +0,0 @@ -#![allow(dead_code)] -use super::ddl::get_create_table_query; -use super::types::ValueWrapper; -use crate::errors::QueryError; -use crate::types::{insert_multi, map_value_wrapper_to_field}; -use clickhouse_rs::types::Query; -use clickhouse_rs::{ClientHandle, Pool}; -use dozer_types::log::{debug, info}; -use dozer_types::models::sink::{ClickhouseSinkConfig, ClickhouseTableOptions}; -use dozer_types::types::{Field, FieldDefinition}; -pub struct SqlResult { - pub rows: Vec>, -} - -#[derive(Clone)] -pub struct ClickhouseClient { - pool: Pool, -} - -impl ClickhouseClient { - pub fn new(config: ClickhouseSinkConfig) -> Self { - let url = Self::construct_url(&config); - let pool = Pool::new(url); - Self { pool } - } - - pub fn construct_url(config: &ClickhouseSinkConfig) -> String { - let user_password = match &config.password { - Some(password) => format!("{}:{}", config.user, password), - None => config.user.to_string(), - }; - - let options = config - .options - .iter() - .map(|(k, v)| format!("{}={}", k, v)) - .collect::>() - .join("&"); - let url = format!( - "{}://{}@{}:{}/{}?{options}", - config.scheme, user_password, config.host, config.port, config.database - ); - debug!("{url}"); - url - } - - pub async fn get_client_handle(&self) -> Result { - let client = self.pool.get_handle().await?; - Ok(client) - } - - pub async fn drop_table(&self, datasource_name: &str) -> Result<(), QueryError> { - let mut client = self.pool.get_handle().await?; - let ddl = format!("DROP TABLE IF EXISTS {}", datasource_name); - info!("#{ddl}"); - client.execute(ddl).await?; - Ok(()) - } - - pub async fn create_table( - &self, - datasource_name: &str, - fields: &[FieldDefinition], - table_options: Option, - query_id: Option, - ) -> Result<(), QueryError> { - let mut client = self.pool.get_handle().await?; - let ddl = get_create_table_query(datasource_name, fields, table_options); - info!("Creating Clickhouse Table"); - info!("{ddl}"); - let query_id = query_id.unwrap_or("".to_string()); - let ddl = Query::new(ddl).id(query_id); - client.execute(ddl).await?; - Ok(()) - } - - pub async fn fetch_all( - &self, - query: &str, - schema: Vec, - query_id: Option, - ) -> Result { - let mut client = self.pool.get_handle().await?; - /* - TODO: Include query_id in RowBinary protocol. - https://github.com/suharev7/clickhouse-rs/issues/176 - https://github.com/ClickHouse/ClickHouse/blob/master/src/Client/Connection.cpp - https://www.propeldata.com/blog/how-to-check-your-clickhouse-version - */ - - let query = Query::new(query).id(query_id.map_or("".to_string(), |q| q.to_string())); - /* - let query = query_id.map_or(query.to_string(), |id| { - format!("{0} settings log_comment = '{1}'", query, id) - }); - */ - - let block = client.query(query).fetch_all().await?; - - let mut rows: Vec> = vec![]; - for row in block.rows() { - let mut row_data = vec![]; - for (idx, field) in schema.clone().into_iter().enumerate() { - let v: ValueWrapper = row.get(idx)?; - row_data.push(map_value_wrapper_to_field(v, field)?); - } - rows.push(row_data); - } - - Ok(SqlResult { rows }) - } - - pub async fn check_table(&self, table_name: &str) -> Result { - let mut client = self.pool.get_handle().await?; - let query = format!("CHECK TABLE {}", table_name); - client.query(query).fetch_all().await?; - - // if error not found, table exists - Ok(true) - } - - pub async fn insert( - &self, - table_name: &str, - fields: &[FieldDefinition], - values: Vec, - query_id: Option, - ) -> Result<(), QueryError> { - let client = self.pool.get_handle().await?; - insert_multi(client, table_name, fields, vec![values], query_id).await - } - - pub async fn insert_multi( - &self, - table_name: &str, - fields: &[FieldDefinition], - rows: Vec>, - query_id: Option, - ) -> Result<(), QueryError> { - let client = self.pool.get_handle().await?; - insert_multi(client, table_name, fields, rows, query_id).await - } -} diff --git a/dozer-sink-clickhouse/src/ddl.rs b/dozer-sink-clickhouse/src/ddl.rs deleted file mode 100644 index aa4da3f1bf..0000000000 --- a/dozer-sink-clickhouse/src/ddl.rs +++ /dev/null @@ -1,79 +0,0 @@ -use dozer_types::models::sink::ClickhouseTableOptions; -use dozer_types::types::FieldDefinition; - -use crate::schema::map_field_to_type; - -const DEFAULT_TABLE_ENGINE: &str = "MergeTree()"; - -pub fn get_create_table_query( - table_name: &str, - fields: &[FieldDefinition], - table_options: Option, -) -> String { - let engine = table_options - .as_ref() - .and_then(|c| c.engine.clone()) - .unwrap_or_else(|| DEFAULT_TABLE_ENGINE.to_string()); - let engine_name = if engine == "CollapsingMergeTree" { - "CollapsingMergeTree(sign)".to_string() - } else { - engine.to_owned() - }; - let mut parts = fields - .iter() - .map(|field| { - let typ = map_field_to_type(field); - format!("{} {}", field.name, typ) - }) - .collect::>(); - if engine == "CollapsingMergeTree" { - parts.push("sign Int8".to_string()); - } - - parts.push( - table_options - .as_ref() - .and_then(|options| options.primary_keys.clone()) - .map_or("".to_string(), |pk| { - format!("PRIMARY KEY ({})", pk.join(", ")) - }), - ); - - let query = parts.join(",\n"); - - let partition_by = table_options - .as_ref() - .and_then(|options| options.partition_by.clone()) - .map_or("".to_string(), |partition_by| { - format!("PARTITION BY {}\n", partition_by) - }); - let sample_by = table_options - .as_ref() - .and_then(|options| options.sample_by.clone()) - .map_or("".to_string(), |partition_by| { - format!("SAMPLE BY {}\n", partition_by) - }); - let order_by = table_options - .as_ref() - .and_then(|options| options.order_by.clone()) - .map_or("".to_string(), |order_by| { - format!("ORDER BY ({})\n", order_by.join(", ")) - }); - let cluster = table_options - .as_ref() - .and_then(|options| options.cluster.clone()) - .map_or("".to_string(), |cluster| { - format!("ON CLUSTER {}\n", cluster) - }); - - format!( - "CREATE TABLE IF NOT EXISTS {table_name} {cluster} ( - {query} - ) - ENGINE = {engine_name} - {order_by} - {partition_by} - {sample_by} - ", - ) -} diff --git a/dozer-sink-clickhouse/src/errors.rs b/dozer-sink-clickhouse/src/errors.rs deleted file mode 100644 index afa0201f8d..0000000000 --- a/dozer-sink-clickhouse/src/errors.rs +++ /dev/null @@ -1,52 +0,0 @@ -use dozer_types::{ - thiserror::{self, Error}, - types::FieldType, -}; - -#[derive(Error, Debug)] -pub enum ClickhouseSinkError { - #[error("Only MergeTree engine is supported for delete operation")] - UnsupportedOperation, - - #[error("Column {0} not found in sink table")] - ColumnNotFound(String), - - #[error("Column {0} has type {1} in dozer schema but type {2} in sink table")] - ColumnTypeMismatch(String, String, String), - - #[error("Clickhouse error: {0:?}")] - ClickhouseError(#[from] clickhouse_rs::errors::Error), - - #[error("Primary key not found")] - PrimaryKeyNotFound, - - #[error("Sink table does not exist and create_table_options is not set")] - SinkTableDoesNotExist, - - #[error("Expected primary key {0:?} but got {1:?}")] - PrimaryKeyMismatch(Vec, Vec), - - #[error("QueryError: {0:?}")] - QueryError(#[from] QueryError), -} - -#[derive(Error, Debug)] -pub enum QueryError { - #[error("Clickhouse error: {0:?}")] - DataFetchError(#[from] clickhouse_rs::errors::Error), - - #[error("Unexpected field type for {field_name:?}, expected {field_type:?}")] - TypeMismatch { - field_name: String, - field_type: FieldType, - }, - - #[error("Decimal overflow")] - DecimalOverflow, - - #[error("Unsupported field type {0:?}")] - UnsupportedFieldType(FieldType), - - #[error("{0:?}")] - CustomError(String), -} diff --git a/dozer-sink-clickhouse/src/lib.rs b/dozer-sink-clickhouse/src/lib.rs deleted file mode 100644 index 426b54ba6c..0000000000 --- a/dozer-sink-clickhouse/src/lib.rs +++ /dev/null @@ -1,10 +0,0 @@ -pub mod client; -pub mod ddl; -pub mod errors; -pub mod schema; -mod sink; -pub use sink::ClickhouseSinkFactory; -pub mod metadata; -#[cfg(test)] -mod tests; -pub mod types; diff --git a/dozer-sink-clickhouse/src/metadata.rs b/dozer-sink-clickhouse/src/metadata.rs deleted file mode 100644 index d5c6075c80..0000000000 --- a/dozer-sink-clickhouse/src/metadata.rs +++ /dev/null @@ -1,49 +0,0 @@ -use dozer_types::types::{FieldDefinition, FieldType, Schema, SourceDefinition}; - -// Replication Metadata Constants -pub const REPLICA_METADATA_TABLE: &str = "__dozer_replication_metadata"; -pub const META_TABLE_COL: &str = "table"; -pub const META_TXN_ID_COL: &str = "txn_id"; - -pub struct ReplicationMetadata { - pub schema: Schema, - pub table_name: String, -} - -impl ReplicationMetadata { - pub fn schema(&self) -> &Schema { - &self.schema - } - - pub fn get_primary_keys(&self) -> Vec { - vec![META_TABLE_COL.to_string()] - } - - pub fn get_metadata() -> ReplicationMetadata { - ReplicationMetadata { - table_name: REPLICA_METADATA_TABLE.to_string(), - schema: Schema::new() - .field( - FieldDefinition { - name: META_TABLE_COL.to_owned(), - typ: FieldType::String, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - true, - ) - .field( - FieldDefinition { - name: META_TXN_ID_COL.to_owned(), - typ: FieldType::UInt, - nullable: false, - source: SourceDefinition::Dynamic, - description: None, - }, - false, - ) - .clone(), - } - } -} diff --git a/dozer-sink-clickhouse/src/schema.rs b/dozer-sink-clickhouse/src/schema.rs deleted file mode 100644 index 14c21f0a30..0000000000 --- a/dozer-sink-clickhouse/src/schema.rs +++ /dev/null @@ -1,178 +0,0 @@ -use crate::client::ClickhouseClient; -use crate::errors::ClickhouseSinkError::{self, SinkTableDoesNotExist}; -use clickhouse_rs::types::Complex; -use clickhouse_rs::{Block, ClientHandle}; -use dozer_types::log::warn; -use dozer_types::models::sink::ClickhouseSinkConfig; -use dozer_types::serde::{Deserialize, Serialize}; -use dozer_types::types::{FieldDefinition, FieldType, Schema}; - -#[derive(Debug, Deserialize, Serialize)] -#[serde(crate = "dozer_types::serde")] -pub struct ClickhouseSchemaColumn { - name: String, - type_: String, -} - -#[derive(Debug, Deserialize, Serialize, Clone)] -#[serde(crate = "dozer_types::serde")] -pub struct ClickhouseTable { - pub database: String, - pub name: String, - pub engine: String, - pub engine_full: String, -} - -#[derive(Debug, Deserialize, Serialize)] -#[serde(crate = "dozer_types::serde")] -pub struct ClickhouseKeyColumnDef { - pub column_name: Option, - pub constraint_name: Option, - pub constraint_schema: String, -} - -pub struct ClickhouseSchema {} - -impl ClickhouseSchema { - pub async fn get_clickhouse_table( - client: ClickhouseClient, - config: &ClickhouseSinkConfig, - ) -> Result { - let mut client = client.get_client_handle().await?; - let query = format!("DESCRIBE TABLE {}", config.sink_table_name); - let block: Block = client.query(&query).fetch_all().await?; - - if block.row_count() == 0 { - Err(SinkTableDoesNotExist) - } else { - Self::fetch_sink_table_info(client, &config.sink_table_name).await - } - } - - pub async fn compare_with_dozer_schema( - client: ClickhouseClient, - schema: &Schema, - table: &ClickhouseTable, - ) -> Result<(), ClickhouseSinkError> { - let mut client = client.get_client_handle().await?; - let block: Block = client - .query(&format!( - "DESCRIBE TABLE {database}.{table_name}", - table_name = table.name, - database = table.database - )) - .fetch_all() - .await?; - - let columns: Vec = block - .rows() - .map(|row| { - let column_name: String = row.get("name").unwrap(); - let column_type: String = row.get("type").unwrap(); - ClickhouseSchemaColumn { - name: column_name, - type_: column_type, - } - }) - .collect(); - - for field in &schema.fields { - let Some(column) = columns.iter().find(|column| column.name == field.name) else { - return Err(ClickhouseSinkError::ColumnNotFound(field.name.clone())); - }; - let expected_type = map_field_to_type(field); - let column_type = column.type_.clone(); - if expected_type != column_type { - return Err(ClickhouseSinkError::ColumnTypeMismatch( - field.name.clone(), - expected_type.to_string(), - column_type.to_string(), - )); - } - } - - Ok(()) - } - - async fn fetch_sink_table_info( - mut handle: ClientHandle, - sink_table_name: &str, - ) -> Result { - let block = handle - .query(format!( - "SELECT database, name, engine, engine_full FROM system.tables WHERE name = '{}'", - sink_table_name - )) - .fetch_all() - .await?; - let row = block.rows().next().unwrap(); - Ok(ClickhouseTable { - database: row.get(0)?, - name: row.get(1)?, - engine: row.get(2)?, - engine_full: row.get(3)?, - }) - } - - #[allow(dead_code)] - async fn fetch_primary_keys( - mut handle: ClientHandle, - sink_table_name: &str, - schema: &str, - ) -> Result, ClickhouseSinkError> { - let block = handle - .query(format!( - r#" - SELECT column_name, constraint_name, constraint_schema - FROM INFORMATION_SCHEMA.key_column_usage - WHERE table_name = '{}' AND constraint_schema = '{}' and column_name is NOT NULL"#, - sink_table_name, schema - )) - .fetch_all() - .await?; - - let mut keys = vec![]; - for r in block.rows() { - let name: Option = r.get(0)?; - if let Some(name) = name { - keys.push(name); - } - } - - Ok(keys) - } -} - -pub fn map_field_to_type(field: &FieldDefinition) -> String { - const DECIMAL_SCALE: u8 = 4; - let decimal = format!("Decimal(10, {})", DECIMAL_SCALE); - let typ: &str = match field.typ { - FieldType::UInt => "UInt64", - FieldType::U128 => "UInt128", - FieldType::Int => "Int64", - FieldType::Int8 => "Int8", - FieldType::I128 => "Int128", - FieldType::Float => "Float64", - FieldType::Boolean => "Boolean", - FieldType::String => "String", - FieldType::Text => "String", - FieldType::Binary => "Array(UInt8)", - FieldType::Decimal => &decimal, - FieldType::Timestamp => "DateTime64(3)", - FieldType::Date => "Date", - FieldType::Json => "JSON", - FieldType::Point => "Point", - FieldType::Duration => unimplemented!(), - }; - - if field.nullable { - if field.typ != FieldType::Binary { - format!("Nullable({})", typ) - } else { - warn!("Binary field cannot be nullable, ignoring nullable flag"); - typ.to_string() - } - } else { - typ.to_string() - } -} diff --git a/dozer-sink-clickhouse/src/sink.rs b/dozer-sink-clickhouse/src/sink.rs deleted file mode 100644 index 06dbd8e1a2..0000000000 --- a/dozer-sink-clickhouse/src/sink.rs +++ /dev/null @@ -1,338 +0,0 @@ -use dozer_core::epoch::Epoch; -use dozer_core::event::EventHub; -use dozer_core::node::{PortHandle, Sink, SinkFactory}; -use dozer_core::tokio::runtime::Runtime; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_types::errors::internal::BoxedError; - -use dozer_types::log::debug; -use dozer_types::models::sink::{ClickhouseSinkConfig, ClickhouseTableOptions}; -use dozer_types::node::OpIdentifier; - -use crate::client::ClickhouseClient; -use crate::errors::ClickhouseSinkError; -use crate::metadata::{ - ReplicationMetadata, META_TABLE_COL, META_TXN_ID_COL, REPLICA_METADATA_TABLE, -}; -use crate::schema::{ClickhouseSchema, ClickhouseTable}; -use dozer_types::tonic::async_trait; -use dozer_types::types::{Field, FieldDefinition, Operation, Schema, TableOperation}; -use std::collections::HashMap; -use std::fmt::Debug; -use std::sync::Arc; - -const BATCH_SIZE: usize = 100; - -#[derive(Debug)] -pub struct ClickhouseSinkFactory { - runtime: Arc, - config: ClickhouseSinkConfig, -} - -impl ClickhouseSinkFactory { - pub fn new(config: ClickhouseSinkConfig, runtime: Arc) -> Self { - Self { config, runtime } - } - - pub async fn create_replication_metadata_table(&self) -> Result<(), BoxedError> { - let client = ClickhouseClient::new(self.config.clone()); - let repl_metadata = ReplicationMetadata::get_metadata(); - - let primary_keys = repl_metadata.get_primary_keys(); - let partition_by = format!("({})", primary_keys.join(",")); - let create_table_options = ClickhouseTableOptions { - engine: Some("ReplacingMergeTree".to_string()), - primary_keys: Some(repl_metadata.get_primary_keys()), - partition_by: Some(partition_by), - // Replaced using this key - order_by: Some(repl_metadata.get_primary_keys()), - cluster: self - .config - .create_table_options - .as_ref() - .and_then(|o| o.cluster.clone()), - sample_by: None, - }; - client - .create_table( - &repl_metadata.table_name, - &repl_metadata.schema.fields, - Some(create_table_options), - None, - ) - .await?; - - Ok(()) - } -} - -#[async_trait] -impl SinkFactory for ClickhouseSinkFactory { - fn type_name(&self) -> String { - "clickhouse".to_string() - } - - fn get_input_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - fn get_input_port_name(&self, _port: &PortHandle) -> String { - self.config.source_table_name.clone() - } - - fn prepare(&self, input_schemas: HashMap) -> Result<(), BoxedError> { - debug_assert!(input_schemas.len() == 1); - Ok(()) - } - - async fn build( - &self, - mut input_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - let schema = input_schemas.remove(&DEFAULT_PORT_HANDLE).unwrap(); - - let client = ClickhouseClient::new(self.config.clone()); - - let config = &self.config; - - // Create Sink Table - self.create_replication_metadata_table().await?; - - // Create Sink Table - if self.config.create_table_options.is_some() { - client - .create_table( - &config.sink_table_name, - &schema.fields, - self.config.create_table_options.clone(), - None, - ) - .await?; - } - let table = ClickhouseSchema::get_clickhouse_table(client.clone(), &self.config).await?; - - ClickhouseSchema::compare_with_dozer_schema(client.clone(), &schema, &table).await?; - let sink = ClickhouseSink::new( - client, - self.config.clone(), - schema, - self.runtime.clone(), - table, - ); - - Ok(Box::new(sink)) - } -} - -pub(crate) struct ClickhouseSink { - pub(crate) client: ClickhouseClient, - pub(crate) runtime: Arc, - pub(crate) schema: Schema, - pub(crate) sink_table_name: String, - pub(crate) table: ClickhouseTable, - batch: Vec>, - metadata: ReplicationMetadata, - latest_txid: Option, -} - -impl Debug for ClickhouseSink { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.debug_struct("ClickhouseSink") - .field("sink_table_name", &self.sink_table_name) - .field("table", &self.table) - .field("schema", &self.schema) - .finish() - } -} - -impl ClickhouseSink { - pub fn new( - client: ClickhouseClient, - config: ClickhouseSinkConfig, - schema: Schema, - runtime: Arc, - table: ClickhouseTable, - ) -> Self { - let mut schema = schema.clone(); - - if table.engine == "CollapsingMergeTree" && !schema.fields.is_empty() { - // get source from any field in schema - let source = schema.fields[0].source.clone(); - schema.fields.push(FieldDefinition { - name: "sign".to_string(), - typ: dozer_types::types::FieldType::Int8, - nullable: false, - description: None, - source, - }); - } - Self { - client, - runtime, - schema, - sink_table_name: config.sink_table_name, - table, - batch: Vec::new(), - latest_txid: None, - metadata: ReplicationMetadata::get_metadata(), - } - } - - pub async fn insert_metadata(&self) -> Result<(), BoxedError> { - debug!( - "[Sink] Inserting metadata record {:?} {}", - self.latest_txid, - self.sink_table_name.clone() - ); - if let Some(txid) = self.latest_txid { - self.client - .insert( - REPLICA_METADATA_TABLE, - &self.metadata.schema.fields, - vec![ - Field::String(self.sink_table_name.clone()), - Field::UInt(txid), - ], - None, - ) - .await?; - } - Ok(()) - } - - fn insert_values(&mut self, values: &[Field]) -> Result<(), BoxedError> { - // add values to batch instead of inserting immediately - self.batch.push(values.to_vec()); - Ok(()) - } - - fn commit_batch(&mut self) -> Result<(), BoxedError> { - let batch = std::mem::take(&mut self.batch); - self.runtime.block_on(async { - //Insert batch - self.client - .insert_multi(&self.sink_table_name, &self.schema.fields, batch, None) - .await?; - - self.insert_metadata().await?; - Ok::<(), BoxedError>(()) - })?; - - Ok(()) - } - - fn _get_latest_op(&mut self) -> Result, BoxedError> { - let op = self.runtime.block_on(async { - let mut client = self.client.get_client_handle().await?; - let table_name = self.sink_table_name.clone(); - let query = format!("SELECT \"{META_TXN_ID_COL}\" FROM \"{REPLICA_METADATA_TABLE}\" WHERE \"{META_TABLE_COL}\" = '\"{table_name}\"' ORDER BY \"{META_TXN_ID_COL}\" LIMIT 1"); - let block = client - .query(query) - .fetch_all() - .await?; - - let row = block.rows().next(); - match row { - Some(row) => { - let txid: u64 = row.get(META_TXN_ID_COL)?; - Ok::, BoxedError>(Some(OpIdentifier { txid, seq_in_tx: 0 })) - }, - None => Ok::, BoxedError>(None), - } - })?; - Ok(op) - } -} - -impl Sink for ClickhouseSink { - fn commit(&mut self, _epoch_details: &Epoch) -> Result<(), BoxedError> { - Ok(()) - } - - fn flush_batch(&mut self) -> Result<(), BoxedError> { - self.commit_batch()?; - Ok(()) - } - - fn process(&mut self, op: TableOperation) -> Result<(), BoxedError> { - self.latest_txid = op.id.map(|id| id.txid); - match op.op { - Operation::Insert { new } => { - if self.table.engine == "CollapsingMergeTree" { - let mut values = new.values; - values.push(Field::Int8(1)); - self.insert_values(&values)?; - } else { - self.insert_values(&new.values)?; - } - - if self.batch.len() > BATCH_SIZE - 1 { - self.commit_batch()?; - } - } - Operation::Delete { old } => { - if self.table.engine != "CollapsingMergeTree" { - return Err(BoxedError::from(ClickhouseSinkError::UnsupportedOperation)); - } - let mut values = old.values; - values.push(Field::Int(-1)); - self.insert_values(&values)?; - } - Operation::Update { new, old } => { - if self.table.engine != "CollapsingMergeTree" { - return Err(BoxedError::from(ClickhouseSinkError::UnsupportedOperation)); - } - let mut values = old.values; - values.push(Field::Int8(-1)); - self.insert_values(&values)?; - - let mut values = new.values; - values.push(Field::Int8(1)); - self.insert_values(&values)?; - } - Operation::BatchInsert { new } => { - for record in new { - let mut values = record.values; - if self.table.engine == "CollapsingMergeTree" { - values.push(Field::Int8(1)); - } - self.insert_values(&values)?; - } - self.commit_batch()?; - } - } - - Ok(()) - } - - fn on_source_snapshotting_started( - &mut self, - _connection_name: String, - ) -> Result<(), BoxedError> { - Ok(()) - } - - fn on_source_snapshotting_done( - &mut self, - _connection_name: String, - id: Option, - ) -> Result<(), BoxedError> { - self.latest_txid = id.map(|opid| opid.txid); - self.commit_batch()?; - Ok(()) - } - - fn set_source_state(&mut self, _source_state: &[u8]) -> Result<(), BoxedError> { - Ok(()) - } - - fn get_source_state(&mut self) -> Result>, BoxedError> { - Ok(None) - } - - fn get_latest_op_id(&mut self) -> Result, BoxedError> { - // self.get_latest_op() - Ok(None) - } -} diff --git a/dozer-sink-clickhouse/src/tests.rs b/dozer-sink-clickhouse/src/tests.rs deleted file mode 100644 index 859eb42458..0000000000 --- a/dozer-sink-clickhouse/src/tests.rs +++ /dev/null @@ -1,105 +0,0 @@ -use crate::client::ClickhouseClient; -use crate::schema::ClickhouseSchema; -use clickhouse_rs::types::Query; -use dozer_core::tokio; -use dozer_types::models::sink::ClickhouseSinkConfig; -use dozer_types::types::{FieldDefinition, FieldType, Schema}; - -fn get_client() -> ClickhouseClient { - ClickhouseClient::new(get_sink_config()) -} - -fn get_sink_config() -> ClickhouseSinkConfig { - ClickhouseSinkConfig { - source_table_name: "source_table".to_string(), - sink_table_name: "sink_table".to_string(), - scheme: "tcp".to_string(), - create_table_options: None, - user: "default".to_string(), - password: None, - database: "default".to_string(), - host: "localhost".to_string(), - port: 9000, - options: vec![], - } -} - -fn _get_dozer_schema() -> Schema { - Schema { - fields: vec![ - FieldDefinition { - name: "id".to_string(), - typ: FieldType::UInt, - nullable: false, - source: Default::default(), - description: None, - }, - FieldDefinition { - name: "data".to_string(), - typ: FieldType::String, - nullable: false, - source: Default::default(), - description: None, - }, - ], - primary_index: vec![0], - } -} - -async fn create_table(table_name: &str) { - let mut client = get_client().get_client_handle().await.unwrap(); - client - .execute(&format!("DROP TABLE IF EXISTS {table_name}")) - .await - .unwrap(); - - client - .execute(&format!("CREATE TABLE {table_name}(id UInt64, data String, PRIMARY KEY id) ENGINE = CollapsingMergeTree ORDER BY id")) - .await - .unwrap(); -} - -#[tokio::test] -#[ignore] -async fn test_get_clickhouse_table() { - let client = get_client(); - let sink_config = get_sink_config(); - create_table(&sink_config.sink_table_name).await; - let clickhouse_table = ClickhouseSchema::get_clickhouse_table(client, &sink_config) - .await - .unwrap(); - assert_eq!(clickhouse_table.name, sink_config.sink_table_name); -} - -use clickhouse_rs::{Block, Pool}; -use std::error::Error; - -#[tokio::test] -#[ignore] -async fn clickhouse_test() -> Result<(), Box> { - let uuid = "248c40d9-d1eb-47c4-8801-943dbab34df9"; - let database_url = "tcp://default@localhost:9000/query_test"; - let ddl = r" - CREATE TABLE IF NOT EXISTS payment ( - customer_id UInt32, - amount UInt32, - account_name Nullable(FixedString(3)) - ) Engine=Memory"; - - let block = Block::new() - .column("customer_id", vec![1_u32, 3, 5, 7, 9]) - .column("amount", vec![2_u32, 4, 6, 8, 10]) - .column( - "account_name", - vec![Some("foo"), None, None, None, Some("bar")], - ); - - let pool = Pool::new(database_url); - - let mut client = pool.get_handle().await?; - client.execute(ddl).await?; - - let table = Query::new("payment").id(uuid); - client.insert(table, block).await?; - Ok(()) -} diff --git a/dozer-sink-clickhouse/src/types.rs b/dozer-sink-clickhouse/src/types.rs deleted file mode 100644 index 1994b6b3c6..0000000000 --- a/dozer-sink-clickhouse/src/types.rs +++ /dev/null @@ -1,319 +0,0 @@ -#![allow(clippy::redundant_closure_call)] -use crate::errors::QueryError; - -use chrono_tz::{Tz, UTC}; -use clickhouse_rs::types::column::ColumnFrom; -use clickhouse_rs::{Block, ClientHandle}; -use dozer_types::chrono::{DateTime, FixedOffset, Offset, TimeZone}; -use dozer_types::ordered_float::OrderedFloat; -use dozer_types::rust_decimal::{self}; -use dozer_types::serde_json; -use dozer_types::types::{Field, FieldDefinition, FieldType}; -use either::Either; - -use clickhouse_rs::types::{FromSql, Query, Value, ValueRef}; - -pub struct ValueWrapper(pub Value); - -impl<'a> FromSql<'a> for ValueWrapper { - fn from_sql(value: ValueRef<'a>) -> clickhouse_rs::errors::Result { - let v = Value::from(value); - Ok(ValueWrapper(v)) - } -} - -pub fn map_value_wrapper_to_field( - value: ValueWrapper, - field: FieldDefinition, -) -> Result { - if field.nullable { - if let clickhouse_rs::types::Value::Nullable(v) = value.0 { - match v { - Either::Left(_) => Ok(Field::Null), - Either::Right(data) => { - let mut fd = field.clone(); - fd.nullable = false; - map_value_wrapper_to_field(ValueWrapper(*data), fd) - } - } - } else { - Err(QueryError::CustomError(format!( - "Field is marked as nullable in Schema but not as per the database {0:?}", - field.name - ))) - } - } else { - let value = value.0; - match field.typ { - FieldType::UInt => match value { - clickhouse_rs::types::Value::UInt8(val) => Ok(Field::UInt(val.into())), - clickhouse_rs::types::Value::UInt16(val) => Ok(Field::UInt(val.into())), - clickhouse_rs::types::Value::UInt32(val) => Ok(Field::UInt(val.into())), - clickhouse_rs::types::Value::UInt64(val) => Ok(Field::UInt(val)), - _ => Err(QueryError::CustomError("Invalid UInt value".to_string())), - }, - FieldType::U128 => match value { - clickhouse_rs::types::Value::UInt128(val) => Ok(Field::U128(val)), - _ => Err(QueryError::CustomError("Invalid U128 value".to_string())), - }, - FieldType::Int => match value { - clickhouse_rs::types::Value::Int8(val) => Ok(Field::Int(val.into())), - clickhouse_rs::types::Value::Int16(val) => Ok(Field::Int(val.into())), - clickhouse_rs::types::Value::Int32(val) => Ok(Field::Int(val.into())), - clickhouse_rs::types::Value::Int64(val) => Ok(Field::Int(val)), - _ => Err(QueryError::CustomError("Invalid Int value".to_string())), - }, - FieldType::I128 => match value { - clickhouse_rs::types::Value::Int128(val) => Ok(Field::I128(val)), - _ => Err(QueryError::CustomError("Invalid I128 value".to_string())), - }, - FieldType::Float => match value { - clickhouse_rs::types::Value::Float64(val) => Ok(Field::Float(OrderedFloat(val))), - _ => Err(QueryError::CustomError("Invalid Float value".to_string())), - }, - FieldType::Boolean => match value { - clickhouse_rs::types::Value::UInt8(val) => Ok(Field::Boolean(val != 0)), - _ => Err(QueryError::CustomError("Invalid Boolean value".to_string())), - }, - FieldType::String => match value { - clickhouse_rs::types::Value::String(_) => Ok(Field::String(value.to_string())), - _ => Err(QueryError::CustomError("Invalid String value".to_string())), - }, - FieldType::Text => match value { - clickhouse_rs::types::Value::String(_) => Ok(Field::String(value.to_string())), - _ => Err(QueryError::CustomError("Invalid String value".to_string())), - }, - FieldType::Binary => match value { - clickhouse_rs::types::Value::String(val) => { - let val = (*val).clone(); - Ok(Field::Binary(val)) - } - _ => Err(QueryError::CustomError("Invalid Binary value".to_string())), - }, - FieldType::Decimal => match value { - clickhouse_rs::types::Value::Decimal(v) => Ok(Field::Decimal( - rust_decimal::Decimal::new(v.internal(), v.scale() as u32), - )), - _ => Err(QueryError::CustomError("Invalid Decimal value".to_string())), - }, - FieldType::Timestamp => { - let v: DateTime = value.into(); - let dt = convert_to_fixed_offset(v); - match dt { - Some(dt) => Ok(Field::Timestamp(dt)), - None => Err(QueryError::CustomError( - "Invalid Timestamp value".to_string(), - )), - } - } - FieldType::Date => Ok(Field::Date(value.into())), - FieldType::Json => match value { - clickhouse_rs::types::Value::String(_) => { - let json = value.to_string(); - let json = serde_json::from_str(&json); - json.map(Field::Json) - .map_err(|e| QueryError::CustomError(e.to_string())) - } - _ => Err(QueryError::CustomError("Invalid Json value".to_string())), - }, - x => Err(QueryError::CustomError(format!( - "Unsupported type {0:?}", - x - ))), - } - } -} - -fn convert_to_fixed_offset(datetime_tz: DateTime) -> Option> { - // Get the offset from UTC in seconds for the specific datetime - let offset_seconds = datetime_tz - .timezone() - .offset_from_utc_datetime(&datetime_tz.naive_utc()) - .fix() - .local_minus_utc(); - - // Create a FixedOffset timezone from the offset in seconds using east_opt() - FixedOffset::east_opt(offset_seconds) - .map(|fixed_offset| fixed_offset.from_utc_datetime(&datetime_tz.naive_utc())) -} - -fn extract_last_column( - rows: &mut [Vec], - mut mapper: Mapper, -) -> Result, QueryError> -where - Mapper: FnMut(Field) -> Result, -{ - rows.iter_mut() - .map(|row| mapper(row.pop().expect("must still have column"))) - .collect() -} - -fn make_nullable_mapper( - mut mapper: Mapper, -) -> impl FnMut(Field) -> Result, QueryError> -where - Mapper: FnMut(Field) -> Result, -{ - move |field| { - if matches!(field, Field::Null) { - Ok(None) - } else { - mapper(field).map(Some) - } - } -} - -/// This is a closure that takes a generic parameter, -/// like C++'s templated labmda, which Rust doesn't support. -/// -/// Saves 4 parameters at every call site. -struct AddLastColumn<'a> { - block: Block, - name: &'a str, - rows: &'a mut [Vec], - nullable: bool, -} - -impl<'a> AddLastColumn<'a> { - fn call( - self, - mapper: Mapper, - ) -> Result, QueryError> - where - Vec: ColumnFrom, - Vec>: ColumnFrom, - Mapper: FnMut(Field) -> Result, - { - Ok(if self.nullable { - self.block.column( - self.name, - extract_last_column(self.rows, make_nullable_mapper(mapper))?, - ) - } else { - self.block - .column(self.name, extract_last_column(self.rows, mapper)?) - }) - } -} - -fn add_last_column_to_block( - block: Block, - name: &str, - rows: &mut [Vec], - field_type: FieldType, - nullable: bool, -) -> Result, QueryError> { - let make_error = || QueryError::TypeMismatch { - field_name: name.to_string(), - field_type, - }; - - macro_rules! trivial_mapper { - ($field_type:path) => { - |field| match field { - $field_type(value) => Ok(value), - _ => Err(make_error()), - } - }; - } - - let add_last_column = AddLastColumn { - block, - name, - rows, - nullable, - }; - - match field_type { - FieldType::UInt => add_last_column.call(trivial_mapper!(Field::UInt)), - FieldType::U128 => add_last_column.call(trivial_mapper!(Field::U128)), - FieldType::Int => add_last_column.call(trivial_mapper!(Field::Int)), - FieldType::Int8 => add_last_column.call(trivial_mapper!(Field::Int8)), - FieldType::I128 => add_last_column.call(trivial_mapper!(Field::I128)), - FieldType::Boolean => add_last_column.call(trivial_mapper!(Field::Boolean)), - FieldType::Float => add_last_column.call(|field| match field { - Field::Float(value) => Ok(value.0), - _ => Err(make_error()), - }), - FieldType::String => add_last_column.call(trivial_mapper!(Field::String)), - FieldType::Text => add_last_column.call(trivial_mapper!(Field::Text)), - FieldType::Binary => add_last_column.call(trivial_mapper!(Field::Binary)), - FieldType::Decimal => add_last_column.call(|field| match field { - Field::Decimal(value) => { - // This is hardcoded in `clickhouse-rs`. - if value.scale() > 18 { - return Err(QueryError::DecimalOverflow); - } - let mantissa: i64 = value - .mantissa() - .try_into() - .map_err(|_| QueryError::DecimalOverflow)?; - Ok(clickhouse_rs::types::Decimal::new( - mantissa, - value.scale() as u8, - )) - } - _ => Err(make_error()), - }), - FieldType::Timestamp => add_last_column.call(|field| match field { - Field::Timestamp(value) => Ok(value.with_timezone(&UTC)), - _ => Err(make_error()), - }), - FieldType::Date => add_last_column.call(trivial_mapper!(Field::Date)), - FieldType::Json => add_last_column.call(|field| match field { - Field::Json(value) => Ok(dozer_types::json_types::json_to_bytes(&value)), - _ => Err(make_error()), - }), - other => Err(QueryError::UnsupportedFieldType(other)), - } -} - -pub async fn insert_multi( - mut client: ClientHandle, - table_name: &str, - fields: &[FieldDefinition], - mut rows: Vec>, - query_id: Option, -) -> Result<(), QueryError> { - let mut block = Block::::new(); - for field in fields.iter().rev() { - block = add_last_column_to_block(block, &field.name, &mut rows, field.typ, field.nullable)?; - } - - let query_id = query_id.unwrap_or("".to_string()); - - let table = Query::new(table_name).id(query_id); - - // Insert the block into the table - client.insert(table, block).await?; - - Ok(()) -} - -mod tests { - #[test] - fn test_add_last_column_to_block() { - use super::*; - use dozer_types::rust_decimal::prelude::ToPrimitive; - let dozer_decimal = dozer_types::rust_decimal::Decimal::new(123, 10); - let mut rows = vec![vec![ - Field::Null, - Field::Text("text".to_string()), - Field::Decimal(dozer_decimal), - ]]; - let mut block = Block::::new(); - block = add_last_column_to_block(block, "decimal", &mut rows, FieldType::Decimal, false) - .unwrap(); - block = add_last_column_to_block(block, "text", &mut rows, FieldType::Text, false).unwrap(); - block = add_last_column_to_block(block, "null", &mut rows, FieldType::UInt, true).unwrap(); - let decimal = block - .get_column("decimal") - .unwrap() - .iter::() - .unwrap() - .next() - .unwrap(); - assert_eq!(Into::::into(decimal), dozer_decimal.to_f64().unwrap()); - } -} diff --git a/dozer-sql/.gitignore b/dozer-sql/.gitignore deleted file mode 100644 index f315ad9fc5..0000000000 --- a/dozer-sql/.gitignore +++ /dev/null @@ -1,14 +0,0 @@ -# Generated by Cargo -# will have compiled files and executables -debug/ -target/ - -# Remove Cargo.lock from gitignore if creating an executable, leave it for libraries -# More information here https://doc.rust-lang.org/cargo/guide/cargo-toml-vs-cargo-lock.html -# Cargo.lock - -# These are backup files generated by rustfmt -**/*.rs.bk - -# MSVC Windows builds of rustc generate these, which store debugging information -*.pdb \ No newline at end of file diff --git a/dozer-sql/Cargo.toml b/dozer-sql/Cargo.toml deleted file mode 100644 index 28bb354f41..0000000000 --- a/dozer-sql/Cargo.toml +++ /dev/null @@ -1,30 +0,0 @@ -[package] -name = "dozer-sql" -version = "0.4.0" -edition = "2021" -authors = ["getdozer/dozer-dev"] -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-types = { path = "../dozer-types" } -dozer-core = { path = "../dozer-core" } -dozer-tracing = { path = "../dozer-tracing" } -dozer-sql-expression = { path = "expression" } - -ahash = "0.8.3" -bincode = { workspace = true } -enum_dispatch = "0.3.12" -linked-hash-map = { version = "0.5.6", features = ["serde_impl"] } -multimap = "0.9.0" -regex = "1.10.2" -tokio = { version = "1", features = ["rt", "macros"] } - -[dev-dependencies] -proptest = "1.3.1" - -[features] -python = ["dozer-sql-expression/python"] -onnx = ["dozer-sql-expression/onnx"] -javascript = ["dozer-sql-expression/javascript"] diff --git a/dozer-sql/expression/Cargo.toml b/dozer-sql/expression/Cargo.toml deleted file mode 100644 index 3423d7904a..0000000000 --- a/dozer-sql/expression/Cargo.toml +++ /dev/null @@ -1,33 +0,0 @@ -[package] -name = "dozer-sql-expression" -version = "0.4.0" -edition = "2021" -authors = ["getdozer/dozer-dev"] -license = "AGPL-3.0-or-later" - -[dependencies] -dozer-types = { path = "../../dozer-types" } -dozer-core = { path = "../../dozer-core" } -num-traits = "0.2.16" -sqlparser = { git = "https://github.com/getdozer/sqlparser-rs.git" } -bigdecimal = { version = "0.3", features = ["serde"], optional = true } -ort = { version = "1.15.2", optional = true } -ndarray = { version = "0.15", optional = true } -half = { version = "2.3.1", optional = true } -like = "0.3.1" -jsonpath = { path = "../jsonpath" } -bincode = { workspace = true } -tokio = "1.34.0" -async-recursion = "1.0.5" - -dozer-deno = { path = "../../dozer-deno", optional = true } -deno_core = { workspace = true, optional = true } - -[dev-dependencies] -proptest = "1.2.0" - -[features] -bigdecimal = ["dep:bigdecimal", "sqlparser/bigdecimal"] -python = ["dozer-types/python-auto-initialize"] -onnx = ["dep:ort", "dep:ndarray", "dep:half"] -javascript = ["dep:dozer-deno", "dep:deno_core"] diff --git a/dozer-sql/expression/src/aggregate.rs b/dozer-sql/expression/src/aggregate.rs deleted file mode 100644 index e194d66372..0000000000 --- a/dozer-sql/expression/src/aggregate.rs +++ /dev/null @@ -1,47 +0,0 @@ -use std::fmt::{Display, Formatter}; - -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Hash, bincode::Encode, bincode::Decode)] -pub enum AggregateFunctionType { - Avg, - Count, - Max, - MaxAppendOnly, - MaxValue, - Min, - MinAppendOnly, - MinValue, - Sum, -} - -impl AggregateFunctionType { - pub(crate) fn new(name: &str) -> Option { - match name { - "avg" => Some(AggregateFunctionType::Avg), - "count" => Some(AggregateFunctionType::Count), - "max" => Some(AggregateFunctionType::Max), - "max_append_only" => Some(AggregateFunctionType::MaxAppendOnly), - "max_value" => Some(AggregateFunctionType::MaxValue), - "min" => Some(AggregateFunctionType::Min), - "min_append_only" => Some(AggregateFunctionType::MinAppendOnly), - "min_value" => Some(AggregateFunctionType::MinValue), - "sum" => Some(AggregateFunctionType::Sum), - _ => None, - } - } -} - -impl Display for AggregateFunctionType { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - match self { - AggregateFunctionType::Avg => f.write_str("AVG"), - AggregateFunctionType::Count => f.write_str("COUNT"), - AggregateFunctionType::Max => f.write_str("MAX"), - AggregateFunctionType::MaxAppendOnly => f.write_str("MAX_APPEND_ONLY"), - AggregateFunctionType::MaxValue => f.write_str("MAX_VALUE"), - AggregateFunctionType::Min => f.write_str("MIN"), - AggregateFunctionType::MinAppendOnly => f.write_str("MIN_APPEND_ONLY"), - AggregateFunctionType::MinValue => f.write_str("MIN_VALUE"), - AggregateFunctionType::Sum => f.write_str("SUM"), - } - } -} diff --git a/dozer-sql/expression/src/arg_utils.rs b/dozer-sql/expression/src/arg_utils.rs deleted file mode 100644 index 5c625b91ed..0000000000 --- a/dozer-sql/expression/src/arg_utils.rs +++ /dev/null @@ -1,127 +0,0 @@ -use std::fmt::Display; -use std::ops::Range; - -use crate::error::Error; -use crate::execution::{Expression, ExpressionType}; -use dozer_types::chrono::{DateTime, FixedOffset}; -use dozer_types::types::{DozerPoint, Field, FieldType, Schema}; - -pub fn validate_one_argument( - args: &[Expression], - schema: &Schema, - function_name: impl Display, -) -> Result { - validate_num_arguments(1..2, args.len(), function_name)?; - args[0].get_type(schema) -} - -pub fn validate_two_arguments( - args: &[Expression], - schema: &Schema, - function_name: impl Display, -) -> Result<(ExpressionType, ExpressionType), Error> { - validate_num_arguments(2..3, args.len(), function_name)?; - let arg1 = args[0].get_type(schema)?; - let arg2 = args[1].get_type(schema)?; - Ok((arg1, arg2)) -} - -pub fn validate_num_arguments( - expected: Range, - actual: usize, - function_name: impl Display, -) -> Result<(), Error> { - if !expected.contains(&actual) { - Err(Error::InvalidNumberOfArguments { - function_name: function_name.to_string(), - expected, - actual, - }) - } else { - Ok(()) - } -} - -pub fn validate_arg_type( - arg: &Expression, - expected: Vec, - schema: &Schema, - function_name: impl Display, - argument_index: usize, -) -> Result { - let arg_t = arg.get_type(schema)?; - if !expected.contains(&arg_t.return_type) { - Err(Error::InvalidFunctionArgumentType { - function_name: function_name.to_string(), - argument_index, - actual: arg_t.return_type, - expected, - }) - } else { - Ok(arg_t) - } -} - -pub fn extract_uint( - field: Field, - function_name: impl Display, - argument_index: usize, -) -> Result { - if let Some(value) = field.to_uint() { - Ok(value) - } else { - Err(Error::InvalidFunctionArgument { - function_name: function_name.to_string(), - argument_index, - argument: field, - }) - } -} - -pub fn extract_float( - field: Field, - function_name: impl Display, - argument_index: usize, -) -> Result { - if let Some(value) = field.to_float() { - Ok(value) - } else { - Err(Error::InvalidFunctionArgument { - function_name: function_name.to_string(), - argument_index, - argument: field, - }) - } -} - -pub fn extract_point( - field: Field, - function_name: impl Display, - argument_index: usize, -) -> Result { - if let Some(value) = field.to_point() { - Ok(value) - } else { - Err(Error::InvalidFunctionArgument { - function_name: function_name.to_string(), - argument_index, - argument: field, - }) - } -} - -pub fn extract_timestamp( - field: Field, - function_name: impl Display, - argument_index: usize, -) -> Result, Error> { - if let Some(value) = field.to_timestamp() { - Ok(value) - } else { - Err(Error::InvalidFunctionArgument { - function_name: function_name.to_string(), - argument_index, - argument: field, - }) - } -} diff --git a/dozer-sql/expression/src/builder.rs b/dozer-sql/expression/src/builder.rs deleted file mode 100644 index cefcda0039..0000000000 --- a/dozer-sql/expression/src/builder.rs +++ /dev/null @@ -1,1064 +0,0 @@ -use std::sync::Arc; - -use crate::aggregate::AggregateFunctionType; -use crate::conditional::ConditionalExpressionType; -use crate::datetime::DateTimeFunctionType; -use crate::error::Error; -use dozer_types::models::udf_config::{UdfConfig, UdfType}; -use dozer_types::types::FieldType; -use dozer_types::{ - ordered_float::OrderedFloat, - types::{Field, FieldDefinition, Schema, SourceDefinition}, -}; -use sqlparser::ast::{ - BinaryOperator as SqlBinaryOperator, DataType, DateTimeField, Expr as SqlExpr, Expr, Function, - FunctionArg, FunctionArgExpr, Ident, Interval, TrimWhereField, - UnaryOperator as SqlUnaryOperator, Value as SqlValue, -}; -use tokio::runtime::Runtime; - -use crate::execution::Expression; -use crate::execution::Expression::{ConditionalExpression, GeoFunction, Now, ScalarFunction}; -use crate::geo::common::GeoFunctionType; -use crate::json_functions::JsonFunctionType; -use crate::operator::{BinaryOperatorType, UnaryOperatorType}; -use crate::scalar::common::ScalarFunctionType; -use crate::scalar::string::TrimType; - -use super::cast::CastOperatorType; - -#[allow(dead_code)] -#[derive(Clone, Debug)] -pub struct ExpressionBuilder { - // Must be an aggregation function - pub aggregations: Vec, - pub offset: usize, - runtime: Arc, -} - -impl ExpressionBuilder { - pub fn new(offset: usize, runtime: Arc) -> Self { - Self { - aggregations: Vec::new(), - offset, - runtime, - } - } - - pub fn from(offset: usize, aggregations: Vec, runtime: Arc) -> Self { - Self { - aggregations, - offset, - runtime, - } - } - - pub async fn build( - &mut self, - parse_aggregations: bool, - sql_expression: &SqlExpr, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - self.parse_sql_expression(parse_aggregations, sql_expression, schema, udfs) - .await - } - - #[async_recursion::async_recursion] - pub async fn parse_sql_expression( - &mut self, - parse_aggregations: bool, - expression: &SqlExpr, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - match expression { - SqlExpr::Trim { - expr, - trim_where, - trim_what, - } => { - self.parse_sql_trim_function( - parse_aggregations, - expr, - trim_where, - trim_what, - schema, - udfs, - ) - .await - } - SqlExpr::Identifier(ident) => Self::parse_sql_column(&[ident.clone()], schema), - SqlExpr::CompoundIdentifier(ident) => Self::parse_sql_column(ident, schema), - SqlExpr::Value(SqlValue::Number(n, _)) => Self::parse_sql_number(n), - SqlExpr::Value(SqlValue::Null) => Ok(Expression::Literal(Field::Null)), - SqlExpr::Value(SqlValue::SingleQuotedString(s) | SqlValue::DoubleQuotedString(s)) => { - Self::parse_sql_string(s) - } - SqlExpr::UnaryOp { expr, op } => { - self.parse_sql_unary_op(parse_aggregations, op, expr, schema, udfs) - .await - } - SqlExpr::BinaryOp { left, op, right } => { - self.parse_sql_binary_op(parse_aggregations, left, op, right, schema, udfs) - .await - } - SqlExpr::Nested(expr) => { - self.parse_sql_expression(parse_aggregations, expr, schema, udfs) - .await - } - SqlExpr::Function(sql_function) => { - self.parse_sql_function(parse_aggregations, sql_function, schema, udfs) - .await - } - SqlExpr::Like { - negated, - expr, - pattern, - escape_char, - } => { - self.parse_sql_like_operator( - parse_aggregations, - negated, - expr, - pattern, - escape_char, - schema, - udfs, - ) - .await - } - SqlExpr::InList { - expr, - list, - negated, - } => { - self.parse_sql_in_list_operator( - parse_aggregations, - expr, - list, - *negated, - schema, - udfs, - ) - .await - } - - SqlExpr::Cast { expr, data_type } => { - self.parse_sql_cast_operator(parse_aggregations, expr, data_type, schema, udfs) - .await - } - SqlExpr::Extract { field, expr } => { - self.parse_sql_extract_operator(parse_aggregations, field, expr, schema, udfs) - .await - } - SqlExpr::Interval(Interval { - value, - leading_field, - leading_precision: _, - last_field: _, - fractional_seconds_precision: _, - }) => { - self.parse_sql_interval_expression( - parse_aggregations, - value, - leading_field, - schema, - udfs, - ) - .await - } - SqlExpr::Case { - operand, - conditions, - results, - else_result, - } => { - self.parse_sql_case_expression( - parse_aggregations, - operand, - conditions, - results, - else_result, - schema, - udfs, - ) - .await - } - SqlExpr::IsNull(expr) => { - self.parse_sql_isnull_operator(parse_aggregations, &false, expr, schema, udfs) - .await - } - SqlExpr::IsNotNull(expr) => { - self.parse_sql_isnull_operator(parse_aggregations, &true, expr, schema, udfs) - .await - } - _ => Err(Error::UnsupportedExpression(expression.clone())), - } - } - - fn parse_sql_column(ident: &[Ident], schema: &Schema) -> Result { - let (src_field, src_table_or_alias, src_connection) = match ident.len() { - 1 => (&ident[0].value, None, None), - 2 => (&ident[1].value, Some(&ident[0].value), None), - 3 => ( - &ident[2].value, - Some(&ident[1].value), - Some(&ident[0].value), - ), - _ => { - return Err(Error::InvalidIdent(ident.to_vec())); - } - }; - - let matching_by_field: Vec<(usize, &FieldDefinition)> = schema - .fields - .iter() - .enumerate() - .filter(|(_idx, f)| &f.name == src_field) - .collect(); - - match matching_by_field.len() { - 1 => Ok(Expression::Column { - index: matching_by_field[0].0, - }), - _ => match src_table_or_alias { - None => Err(Error::InvalidIdent(ident.to_vec())), - Some(src_table_or_alias) => { - let matching_by_table_or_alias: Vec<(usize, &FieldDefinition)> = - matching_by_field - .into_iter() - .filter(|(_idx, field)| match &field.source { - SourceDefinition::Alias { name } => name == src_table_or_alias, - SourceDefinition::Table { - name, - connection: _, - } => name == src_table_or_alias, - _ => false, - }) - .collect(); - - match matching_by_table_or_alias.len() { - 1 => Ok(Expression::Column { - index: matching_by_table_or_alias[0].0, - }), - _ => match src_connection { - None => Err(Error::InvalidIdent(ident.to_vec())), - Some(src_connection) => { - let matching_by_connection: Vec<(usize, &FieldDefinition)> = - matching_by_table_or_alias - .into_iter() - .filter(|(_idx, field)| match &field.source { - SourceDefinition::Table { - name: _, - connection, - } => connection == src_connection, - _ => false, - }) - .collect(); - - match matching_by_connection.len() { - 1 => Ok(Expression::Column { - index: matching_by_connection[0].0, - }), - _ => Err(Error::InvalidIdent(ident.to_vec())), - } - } - }, - } - } - }, - } - } - - async fn parse_sql_trim_function( - &mut self, - parse_aggregations: bool, - expr: &Expr, - trim_where: &Option, - trim_what: &Option>, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - let arg = Box::new( - self.parse_sql_expression(parse_aggregations, expr, schema, udfs) - .await?, - ); - let what = match trim_what { - Some(e) => Some(Box::new( - self.parse_sql_expression(parse_aggregations, e, schema, udfs) - .await?, - )), - _ => None, - }; - let typ = trim_where.as_ref().map(|e| match e { - TrimWhereField::Both => TrimType::Both, - TrimWhereField::Leading => TrimType::Leading, - TrimWhereField::Trailing => TrimType::Trailing, - }); - Ok(Expression::Trim { arg, what, typ }) - } - - async fn aggr_function_check( - &mut self, - function_name: String, - parse_aggregations: bool, - sql_function: &Function, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Option { - if !parse_aggregations { - return None; - } - - let aggr = AggregateFunctionType::new(function_name.as_str())?; - - let mut arg_expr: Vec = Vec::new(); - for arg in &sql_function.args { - let aggregation = self - .parse_sql_function_arg(true, arg, schema, udfs) - .await - .ok()?; - arg_expr.push(aggregation); - } - let measure = Expression::AggregateFunction { - fun: aggr, - args: arg_expr, - }; - let index = match self - .aggregations - .iter() - .enumerate() - .find(|e| e.1 == &measure) - { - Some((index, _existing)) => index, - _ => { - self.aggregations.push(measure); - self.aggregations.len() - 1 - } - }; - Some(Expression::Column { - index: self.offset + index, - }) - } - - async fn scalar_function_check( - &mut self, - function_name: String, - parse_aggregations: bool, - sql_function: &Function, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Option { - let mut function_args: Vec = Vec::new(); - for arg in &sql_function.args { - function_args.push( - self.parse_sql_function_arg(parse_aggregations, arg, schema, udfs) - .await - .ok()?, - ); - } - - let sft = ScalarFunctionType::new(function_name.as_str())?; - Some(ScalarFunction { - fun: sft, - args: function_args, - }) - } - - async fn geo_expr_check( - &mut self, - function_name: String, - parse_aggregations: bool, - sql_function: &Function, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Option { - let mut function_args: Vec = Vec::new(); - for arg in &sql_function.args { - function_args.push( - self.parse_sql_function_arg(parse_aggregations, arg, schema, udfs) - .await - .ok()?, - ); - } - - let gft = GeoFunctionType::new(function_name.as_str())?; - Some(GeoFunction { - fun: gft, - args: function_args, - }) - } - - fn datetime_expr_check(&mut self, function_name: String) -> Option { - let dtf = DateTimeFunctionType::new(function_name.as_str())?; - Some(Now { fun: dtf }) - } - - async fn json_func_check( - &mut self, - function_name: String, - parse_aggregations: bool, - sql_function: &Function, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Option { - let mut function_args: Vec = Vec::new(); - for arg in &sql_function.args { - function_args.push( - self.parse_sql_function_arg(parse_aggregations, arg, schema, udfs) - .await - .ok()?, - ); - } - - let jft = JsonFunctionType::new(function_name.as_str())?; - Some(Expression::Json { - fun: jft, - args: function_args, - }) - } - - async fn conditional_expr_check( - &mut self, - function_name: String, - parse_aggregations: bool, - sql_function: &Function, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Option { - let mut function_args: Vec = Vec::new(); - for arg in &sql_function.args { - function_args.push( - self.parse_sql_function_arg(parse_aggregations, arg, schema, udfs) - .await - .ok()?, - ); - } - - let cet = ConditionalExpressionType::new(function_name.as_str())?; - Some(ConditionalExpression { - fun: cet, - args: function_args, - }) - } - - async fn parse_sql_function( - &mut self, - parse_aggregations: bool, - sql_function: &Function, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - let function_name = sql_function.name.to_string().to_lowercase(); - - #[cfg(feature = "python")] - if function_name.starts_with("py_") { - // The function is from python udf. - let udf_name = function_name.strip_prefix("py_").unwrap(); - return self - .parse_python_udf(udf_name, sql_function, schema, udfs) - .await; - } - - if let Some(aggr_check) = self - .aggr_function_check( - function_name.clone(), - parse_aggregations, - sql_function, - schema, - udfs, - ) - .await - { - return Ok(aggr_check); - } - - if let Some(scalar_check) = self - .scalar_function_check( - function_name.clone(), - parse_aggregations, - sql_function, - schema, - udfs, - ) - .await - { - return Ok(scalar_check); - } - - if let Some(geo_check) = self - .geo_expr_check( - function_name.clone(), - parse_aggregations, - sql_function, - schema, - udfs, - ) - .await - { - return Ok(geo_check); - } - - if let Some(conditional_check) = self - .conditional_expr_check( - function_name.clone(), - parse_aggregations, - sql_function, - schema, - udfs, - ) - .await - { - return Ok(conditional_check); - } - - if let Some(datetime_check) = self.datetime_expr_check(function_name.clone()) { - return Ok(datetime_check); - } - - if let Some(json_check) = self - .json_func_check( - function_name.clone(), - parse_aggregations, - sql_function, - schema, - udfs, - ) - .await - { - return Ok(json_check); - } - - // config check for udfs - let udf_type = udfs.iter().find(|udf| udf.name == function_name); - if let Some(udf_type) = udf_type { - return match &udf_type.config { - UdfType::Onnx(config) => { - #[cfg(feature = "onnx")] - { - self.parse_onnx_udf( - function_name.clone(), - config, - sql_function, - schema, - udfs, - ) - .await - } - - #[cfg(not(feature = "onnx"))] - { - let _ = config; - Err(Error::OnnxNotEnabled) - } - } - - UdfType::JavaScript(config) => { - #[cfg(feature = "javascript")] - { - self.parse_javascript_udf( - function_name.clone(), - config, - sql_function, - schema, - udfs, - ) - .await - } - - #[cfg(not(feature = "javascript"))] - { - let _ = config; - Err(Error::JavaScriptNotEnabled) - } - } - }; - } - - Err(Error::UnknownFunction(function_name.clone())) - } - - async fn parse_sql_function_arg( - &mut self, - parse_aggregations: bool, - argument: &FunctionArg, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - match argument { - FunctionArg::Named { - name: _, - arg: FunctionArgExpr::Expr(arg), - } => { - self.parse_sql_expression(parse_aggregations, arg, schema, udfs) - .await - } - FunctionArg::Named { - name: _, - arg: FunctionArgExpr::Wildcard, - } => Ok(Expression::Literal(Field::Null)), - FunctionArg::Unnamed(FunctionArgExpr::Expr(arg)) => { - self.parse_sql_expression(parse_aggregations, arg, schema, udfs) - .await - } - FunctionArg::Unnamed(FunctionArgExpr::Wildcard) => Ok(Expression::Literal(Field::Null)), - _ => Err(Error::UnsupportedFunctionArg(argument.clone())), - } - } - - #[allow(clippy::too_many_arguments)] - async fn parse_sql_case_expression( - &mut self, - parse_aggregations: bool, - operand: &Option>, - conditions: &[Expr], - results: &[Expr], - else_result: &Option>, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - let op = match operand { - Some(o) => Some(Box::new( - self.parse_sql_expression(parse_aggregations, o, schema, udfs) - .await?, - )), - None => None, - }; - let mut conds = vec![]; - for cond in conditions { - conds.push( - self.parse_sql_expression(parse_aggregations, cond, schema, udfs) - .await?, - ); - } - let mut res = vec![]; - for r in results { - res.push( - self.parse_sql_expression(parse_aggregations, r, schema, udfs) - .await?, - ); - } - let else_res = match else_result { - Some(r) => Some(Box::new( - self.parse_sql_expression(parse_aggregations, r, schema, udfs) - .await?, - )), - None => None, - }; - - Ok(Expression::Case { - operand: op, - conditions: conds, - results: res, - else_result: else_res, - }) - } - - async fn parse_sql_interval_expression( - &mut self, - parse_aggregations: bool, - value: &Expr, - leading_field: &Option, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - let right = self - .parse_sql_expression(parse_aggregations, value, schema, udfs) - .await?; - if let Some(leading_field) = leading_field { - Ok(Expression::DateTimeFunction { - fun: DateTimeFunctionType::Interval { - field: *leading_field, - }, - arg: Box::new(right), - }) - } else { - Err(Error::MissingLeadingFieldInInterval) - } - } - - async fn parse_sql_unary_op( - &mut self, - parse_aggregations: bool, - op: &SqlUnaryOperator, - expr: &SqlExpr, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - let arg = Box::new( - self.parse_sql_expression(parse_aggregations, expr, schema, udfs) - .await?, - ); - let operator = match op { - SqlUnaryOperator::Not => UnaryOperatorType::Not, - SqlUnaryOperator::Plus => UnaryOperatorType::Plus, - SqlUnaryOperator::Minus => UnaryOperatorType::Minus, - _ => return Err(Error::UnsupportedUnaryOperator(*op)), - }; - - Ok(Expression::UnaryOperator { operator, arg }) - } - - async fn parse_sql_binary_op( - &mut self, - parse_aggregations: bool, - left: &SqlExpr, - op: &SqlBinaryOperator, - right: &SqlExpr, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - let left_op = self - .parse_sql_expression(parse_aggregations, left, schema, udfs) - .await?; - let right_op = self - .parse_sql_expression(parse_aggregations, right, schema, udfs) - .await?; - - let operator = match op { - SqlBinaryOperator::Gt => BinaryOperatorType::Gt, - SqlBinaryOperator::GtEq => BinaryOperatorType::Gte, - SqlBinaryOperator::Lt => BinaryOperatorType::Lt, - SqlBinaryOperator::LtEq => BinaryOperatorType::Lte, - SqlBinaryOperator::Eq => BinaryOperatorType::Eq, - SqlBinaryOperator::NotEq => BinaryOperatorType::Ne, - SqlBinaryOperator::Plus => BinaryOperatorType::Add, - SqlBinaryOperator::Minus => BinaryOperatorType::Sub, - SqlBinaryOperator::Multiply => BinaryOperatorType::Mul, - SqlBinaryOperator::Divide => BinaryOperatorType::Div, - SqlBinaryOperator::Modulo => BinaryOperatorType::Mod, - SqlBinaryOperator::And => BinaryOperatorType::And, - SqlBinaryOperator::Or => BinaryOperatorType::Or, - _ => return Err(Error::UnsupportedBinaryOperator(op.clone())), - }; - - Ok(Expression::BinaryOperator { - left: Box::new(left_op), - operator, - right: Box::new(right_op), - }) - } - - #[cfg(not(feature = "bigdecimal"))] - fn parse_sql_number(n: &str) -> Result { - match n.parse::() { - Ok(n) => Ok(Expression::Literal(Field::Int(n))), - Err(_) => match n.parse::() { - Ok(f) => Ok(Expression::Literal(Field::Float(OrderedFloat(f)))), - Err(_) => Err(Error::NotANumber(n.to_string())), - }, - } - } - - #[cfg(feature = "bigdecimal")] - fn parse_sql_number(n: &bigdecimal::BigDecimal) -> Result { - use bigdecimal::ToPrimitive; - if n.is_integer() { - Ok(Expression::Literal(Field::Int(n.to_i64().unwrap()))) - } else { - match n.to_f64() { - Some(f) => Ok(Expression::Literal(Field::Float(OrderedFloat(f)))), - None => Err(Error::NotANumber(n.to_string())), - } - } - } - - #[allow(clippy::too_many_arguments)] - async fn parse_sql_like_operator( - &mut self, - parse_aggregations: bool, - negated: &bool, - expr: &Expr, - pattern: &Expr, - escape_char: &Option, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - let arg = self - .parse_sql_expression(parse_aggregations, expr, schema, udfs) - .await?; - let pattern = self - .parse_sql_expression(parse_aggregations, pattern, schema, udfs) - .await?; - let like_expression = Expression::Like { - arg: Box::new(arg), - pattern: Box::new(pattern), - escape: *escape_char, - }; - if *negated { - Ok(Expression::UnaryOperator { - operator: UnaryOperatorType::Not, - arg: Box::new(like_expression), - }) - } else { - Ok(like_expression) - } - } - - async fn parse_sql_extract_operator( - &mut self, - parse_aggregations: bool, - field: &sqlparser::ast::DateTimeField, - expr: &Expr, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - let right = self - .parse_sql_expression(parse_aggregations, expr, schema, udfs) - .await?; - Ok(Expression::DateTimeFunction { - fun: DateTimeFunctionType::Extract { field: *field }, - arg: Box::new(right), - }) - } - - async fn parse_sql_cast_operator( - &mut self, - parse_aggregations: bool, - expr: &Expr, - data_type: &DataType, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - let expression = self - .parse_sql_expression(parse_aggregations, expr, schema, udfs) - .await?; - let cast_to = match data_type { - DataType::Decimal(_) => CastOperatorType(FieldType::Decimal), - DataType::Binary(_) => CastOperatorType(FieldType::Binary), - DataType::Float(_) => CastOperatorType(FieldType::Float), - DataType::Int(_) => CastOperatorType(FieldType::Int), - DataType::Integer(_) => CastOperatorType(FieldType::Int), - DataType::UnsignedInt(_) => CastOperatorType(FieldType::UInt), - DataType::UnsignedInteger(_) => CastOperatorType(FieldType::UInt), - DataType::Boolean => CastOperatorType(FieldType::Boolean), - DataType::Date => CastOperatorType(FieldType::Date), - DataType::Timestamp(..) => CastOperatorType(FieldType::Timestamp), - DataType::Text => CastOperatorType(FieldType::Text), - DataType::String => CastOperatorType(FieldType::String), - DataType::JSON => CastOperatorType(FieldType::Json), - DataType::Custom(name, ..) => { - if name.to_string().to_lowercase() == "uint" { - CastOperatorType(FieldType::UInt) - } else if name.to_string().to_lowercase() == "u128" { - CastOperatorType(FieldType::U128) - } else if name.to_string().to_lowercase() == "i128" { - CastOperatorType(FieldType::I128) - } else { - return Err(Error::UnsupportedDataType(data_type.clone())); - } - } - _ => Err(Error::UnsupportedDataType(data_type.clone()))?, - }; - Ok(Expression::Cast { - arg: Box::new(expression), - typ: cast_to, - }) - } - - fn parse_sql_string(s: &str) -> Result { - Ok(Expression::Literal(Field::String(s.to_owned()))) - } - - pub fn fullname_from_ident(ident: &[Ident]) -> String { - let mut ident_tokens = vec![]; - for token in ident.iter() { - ident_tokens.push(token.value.clone()); - } - ident_tokens.join(".") - } - - pub fn normalize_ident(id: &Ident) -> String { - match id.quote_style { - Some(_) => id.value.clone(), - None => id.value.clone(), - } - } - - #[cfg(feature = "python")] - async fn parse_python_udf( - &mut self, - name: &str, - function: &Function, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - use crate::python_udf::Error::{FailedToParseReturnType, MissingReturnType}; - - // First, get python function define by name. - // Then, transfer python function to Expression::PythonUDF - let mut args = vec![]; - for argument in &function.args { - let arg = self - .parse_sql_function_arg(false, argument, schema, udfs) - .await?; - args.push(arg); - } - - let return_type = { - let ident = function - .return_type - .as_ref() - .ok_or_else(|| MissingReturnType)?; - - FieldType::try_from(ident.value.as_str()).map_err(FailedToParseReturnType)? - }; - - Ok(Expression::PythonUDF { - name: name.to_string(), - args, - return_type, - }) - } - - #[cfg(feature = "onnx")] - async fn parse_onnx_udf( - &mut self, - name: String, - config: &dozer_types::models::udf_config::OnnxConfig, - function: &Function, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - use crate::error::Error::Onnx; - use crate::onnx::error::Error::OnnxOrtErr; - - // First, get onnx function define by name. - // Then, transfer onnx function to Expression::OnnxUDF - use crate::onnx::utils::{onnx_input_validation, onnx_output_validation}; - use ort::{Environment, GraphOptimizationLevel, LoggingLevel, SessionBuilder}; - use std::path::Path; - - let mut args = vec![]; - for argument in &function.args { - let arg = self - .parse_sql_function_arg(false, argument, schema, udfs) - .await?; - args.push(arg); - } - - let environment = Environment::builder() - .with_name("dozer_onnx") - .with_log_level(LoggingLevel::Verbose) - .build() - .map_err(|e| Onnx(OnnxOrtErr(e)))? - .into_arc(); - - let session = SessionBuilder::new(&environment) - .map_err(|e| Onnx(OnnxOrtErr(e)))? - .with_optimization_level(GraphOptimizationLevel::Level1) - .map_err(|e| Onnx(OnnxOrtErr(e)))? - .with_intra_threads(1) - .map_err(|e| Onnx(OnnxOrtErr(e)))? - .with_model_from_file(Path::new(config.path.as_str())) - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - - // input number, type, shape validation - onnx_input_validation(schema, &args, &session.inputs)?; - // output number, type, shape validation - onnx_output_validation(&session.outputs)?; - - Ok(Expression::OnnxUDF { - name, - session: crate::onnx::DozerSession(session.into()), - args, - }) - } - - #[cfg(feature = "javascript")] - async fn parse_javascript_udf( - &mut self, - name: String, - config: &dozer_types::models::udf_config::JavaScriptConfig, - function: &Function, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - let mut args = vec![]; - for argument in &function.args { - let arg = self - .parse_sql_function_arg(false, argument, schema, udfs) - .await?; - args.push(arg); - } - - use crate::javascript::{validate_args, Udf}; - validate_args(name.clone(), &args, schema)?; - let udf = Udf::new( - self.runtime.clone(), - name, - config.module.clone(), - args.remove(0), - ) - .await?; - Ok(Expression::JavaScriptUdf(udf)) - } - - async fn parse_sql_in_list_operator( - &mut self, - parse_aggregations: bool, - expr: &Expr, - list: &[Expr], - negated: bool, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - let expr = self - .parse_sql_expression(parse_aggregations, expr, schema, udfs) - .await?; - let mut list_expressions = vec![]; - for expr in list { - list_expressions.push( - self.parse_sql_expression(parse_aggregations, expr, schema, udfs) - .await?, - ); - } - let in_list_expression = Expression::InList { - expr: Box::new(expr), - list: list_expressions, - negated, - }; - - Ok(in_list_expression) - } - - async fn parse_sql_isnull_operator( - &mut self, - parse_aggregations: bool, - negated: &bool, - expr: &Expr, - schema: &Schema, - udfs: &[UdfConfig], - ) -> Result { - let arg = self - .parse_sql_expression(parse_aggregations, expr, schema, udfs) - .await?; - - if *negated { - Ok(Expression::IsNotNull { arg: Box::new(arg) }) - } else { - Ok(Expression::IsNull { arg: Box::new(arg) }) - } - } -} - -#[derive(Debug, Clone, Hash, PartialEq, Eq)] -pub struct NameOrAlias(pub String, pub Option); - -pub fn extend_schema_source_def(schema: &Schema, name: &NameOrAlias) -> Schema { - let mut output_schema = schema.clone(); - let mut fields = vec![]; - for mut field in schema.clone().fields.into_iter() { - if let Some(alias) = &name.1 { - field.source = SourceDefinition::Alias { - name: alias.to_string(), - }; - } - - fields.push(field); - } - output_schema.fields = fields; - - output_schema -} diff --git a/dozer-sql/expression/src/case.rs b/dozer-sql/expression/src/case.rs deleted file mode 100644 index 6a09e7a2bf..0000000000 --- a/dozer-sql/expression/src/case.rs +++ /dev/null @@ -1,34 +0,0 @@ -use dozer_types::types::Record; -use dozer_types::types::{Field, Schema}; -use std::iter::zip; - -use crate::error::Error; -use crate::execution::Expression; - -pub fn evaluate_case( - schema: &Schema, - _operand: &Option>, - conditions: &mut [Expression], - results: &mut [Expression], - else_result: &mut Option>, - record: &Record, -) -> Result { - let iter = zip(conditions, results); - for (cond, res) in iter { - let field = cond.evaluate(record, schema)?; - if let Some(cond_match) = field.as_boolean() { - if cond_match { - let then_res = res.evaluate(record, schema)?; - return Ok(then_res); - } else { - continue; - } - } - } - if let Some(else_res) = else_result { - let else_return = else_res.evaluate(record, schema)?; - Ok(else_return) - } else { - Ok(Field::Null) - } -} diff --git a/dozer-sql/expression/src/cast.rs b/dozer-sql/expression/src/cast.rs deleted file mode 100644 index 7c84720676..0000000000 --- a/dozer-sql/expression/src/cast.rs +++ /dev/null @@ -1,382 +0,0 @@ -use std::fmt::{Display, Formatter}; - -use dozer_types::types::Record; -use dozer_types::{ - ordered_float::OrderedFloat, - types::{Field, FieldType, Schema}, -}; - -use crate::arg_utils::validate_arg_type; -use crate::error::Error; - -use super::execution::{Expression, ExpressionType}; - -#[allow(dead_code)] -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Hash)] -pub struct CastOperatorType(pub FieldType); - -impl Display for CastOperatorType { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - match self.0 { - FieldType::UInt => f.write_str("CAST AS UINT"), - FieldType::U128 => f.write_str("CAST AS U128"), - FieldType::Int => f.write_str("CAST AS INT"), - FieldType::Int8 => f.write_str("CAST AS INT8"), - FieldType::I128 => f.write_str("CAST AS I128"), - FieldType::Float => f.write_str("CAST AS FLOAT"), - FieldType::Boolean => f.write_str("CAST AS BOOLEAN"), - FieldType::String => f.write_str("CAST AS STRING"), - FieldType::Text => f.write_str("CAST AS TEXT"), - FieldType::Binary => f.write_str("CAST AS BINARY"), - FieldType::Decimal => f.write_str("CAST AS DECIMAL"), - FieldType::Timestamp => f.write_str("CAST AS TIMESTAMP"), - FieldType::Date => f.write_str("CAST AS DATE"), - FieldType::Json => f.write_str("CAST AS JSON"), - FieldType::Point => f.write_str("CAST AS POINT"), - FieldType::Duration => f.write_str("CAST AS DURATION"), - } - } -} - -impl CastOperatorType { - pub(crate) fn evaluate( - &self, - schema: &Schema, - arg: &mut Expression, - record: &Record, - ) -> Result { - let field = arg.evaluate(record, schema)?; - cast_field(&field, self.0) - } - - pub(crate) fn get_return_type( - &self, - schema: &Schema, - arg: &Expression, - ) -> Result { - let (expected_input_type, return_type) = match self.0 { - FieldType::UInt => ( - vec![ - FieldType::Int, - FieldType::String, - FieldType::UInt, - FieldType::I128, - FieldType::U128, - FieldType::Json, - ], - FieldType::UInt, - ), - FieldType::U128 => ( - vec![ - FieldType::Int, - FieldType::String, - FieldType::UInt, - FieldType::I128, - FieldType::U128, - FieldType::Json, - ], - FieldType::U128, - ), - FieldType::Int => ( - vec![ - FieldType::Int, - FieldType::Int8, - FieldType::String, - FieldType::UInt, - FieldType::I128, - FieldType::U128, - FieldType::Json, - ], - FieldType::Int, - ), - FieldType::Int8 => ( - vec![ - FieldType::Int, - FieldType::Int8, - FieldType::String, - FieldType::UInt, - FieldType::I128, - FieldType::U128, - FieldType::Json, - ], - FieldType::Int, - ), - FieldType::I128 => ( - vec![ - FieldType::Int, - FieldType::String, - FieldType::UInt, - FieldType::I128, - FieldType::U128, - FieldType::Json, - ], - FieldType::I128, - ), - FieldType::Float => ( - vec![ - FieldType::Decimal, - FieldType::Float, - FieldType::Int, - FieldType::I128, - FieldType::String, - FieldType::UInt, - FieldType::U128, - FieldType::Json, - ], - FieldType::Float, - ), - FieldType::Boolean => ( - vec![ - FieldType::Boolean, - FieldType::Decimal, - FieldType::Float, - FieldType::Int, - FieldType::I128, - FieldType::UInt, - FieldType::U128, - FieldType::Json, - ], - FieldType::Boolean, - ), - FieldType::String => ( - vec![ - FieldType::Binary, - FieldType::Boolean, - FieldType::Date, - FieldType::Decimal, - FieldType::Float, - FieldType::Int, - FieldType::I128, - FieldType::String, - FieldType::Text, - FieldType::Timestamp, - FieldType::UInt, - FieldType::U128, - FieldType::Json, - ], - FieldType::String, - ), - FieldType::Text => ( - vec![ - FieldType::Binary, - FieldType::Boolean, - FieldType::Date, - FieldType::Decimal, - FieldType::Float, - FieldType::Int, - FieldType::I128, - FieldType::String, - FieldType::Text, - FieldType::Timestamp, - FieldType::UInt, - FieldType::U128, - FieldType::Json, - ], - FieldType::Text, - ), - FieldType::Binary => (vec![FieldType::Binary], FieldType::Binary), - FieldType::Decimal => ( - vec![ - FieldType::Decimal, - FieldType::Float, - FieldType::Int, - FieldType::I128, - FieldType::String, - FieldType::UInt, - FieldType::U128, - ], - FieldType::Decimal, - ), - FieldType::Timestamp => ( - vec![FieldType::String, FieldType::Timestamp], - FieldType::Timestamp, - ), - FieldType::Date => ( - vec![FieldType::Date, FieldType::Timestamp, FieldType::String], - FieldType::Date, - ), - FieldType::Json => ( - vec![ - FieldType::Boolean, - FieldType::Float, - FieldType::Int, - FieldType::I128, - FieldType::String, - FieldType::Text, - FieldType::UInt, - FieldType::U128, - FieldType::Json, - ], - FieldType::Json, - ), - FieldType::Point => (vec![FieldType::Point], FieldType::Point), - FieldType::Duration => ( - vec![ - FieldType::UInt, - FieldType::U128, - FieldType::Int, - FieldType::I128, - FieldType::Duration, - FieldType::String, - FieldType::Text, - ], - FieldType::Duration, - ), - }; - - let expression_type = validate_arg_type(arg, expected_input_type, schema, self, 0)?; - Ok(ExpressionType { - return_type, - nullable: expression_type.nullable, - source: expression_type.source, - is_primary_key: expression_type.is_primary_key, - }) - } -} - -pub fn cast_field(input: &Field, output_type: FieldType) -> Result { - match output_type { - FieldType::UInt => { - if let Some(value) = input.to_uint() { - Ok(Field::UInt(value)) - } else { - Err(Error::InvalidCast { - from: input.clone(), - to: FieldType::UInt, - }) - } - } - FieldType::U128 => { - if let Some(value) = input.to_u128() { - Ok(Field::U128(value)) - } else { - Err(Error::InvalidCast { - from: input.clone(), - to: FieldType::U128, - }) - } - } - FieldType::Int => { - if let Some(value) = input.to_int() { - Ok(Field::Int(value)) - } else { - Err(Error::InvalidCast { - from: input.clone(), - to: FieldType::Int, - }) - } - } - FieldType::Int8 => { - if let Some(value) = input.to_int8() { - Ok(Field::Int8(value)) - } else { - Err(Error::InvalidCast { - from: input.clone(), - to: FieldType::Int, - }) - } - } - FieldType::I128 => { - if let Some(value) = input.to_i128() { - Ok(Field::I128(value)) - } else { - Err(Error::InvalidCast { - from: input.clone(), - to: FieldType::I128, - }) - } - } - FieldType::Float => { - if let Some(value) = input.to_float() { - Ok(Field::Float(OrderedFloat(value))) - } else { - Err(Error::InvalidCast { - from: input.clone(), - to: FieldType::Float, - }) - } - } - FieldType::Boolean => { - if let Some(value) = input.to_boolean() { - Ok(Field::Boolean(value)) - } else { - Err(Error::InvalidCast { - from: input.clone(), - to: FieldType::Boolean, - }) - } - } - FieldType::String => Ok(Field::String(input.to_string())), - FieldType::Text => Ok(Field::Text(input.to_text())), - FieldType::Binary => { - if let Some(value) = input.to_binary() { - Ok(Field::Binary(value.to_vec())) - } else { - Err(Error::InvalidCast { - from: input.clone(), - to: FieldType::Binary, - }) - } - } - FieldType::Decimal => { - if let Some(value) = input.to_decimal() { - Ok(Field::Decimal(value)) - } else { - Err(Error::InvalidCast { - from: input.clone(), - to: FieldType::Decimal, - }) - } - } - FieldType::Timestamp => { - if let Some(value) = input.to_timestamp() { - Ok(Field::Timestamp(value)) - } else { - Err(Error::InvalidCast { - from: input.clone(), - to: FieldType::Timestamp, - }) - } - } - FieldType::Date => { - if let Some(value) = input.to_date() { - Ok(Field::Date(value)) - } else { - Err(Error::InvalidCast { - from: input.clone(), - to: FieldType::Date, - }) - } - } - FieldType::Json => { - if let Some(value) = input.to_json() { - Ok(Field::Json(value)) - } else { - Err(Error::InvalidCast { - from: input.clone(), - to: FieldType::Json, - }) - } - } - FieldType::Point => { - if let Some(value) = input.to_point() { - Ok(Field::Point(value)) - } else { - Err(Error::InvalidCast { - from: input.clone(), - to: FieldType::Point, - }) - } - } - FieldType::Duration => { - if let Some(value) = input.to_duration() { - Ok(Field::Duration(value)) - } else { - Err(Error::InvalidCast { - from: input.clone(), - to: FieldType::Duration, - }) - } - } - } -} diff --git a/dozer-sql/expression/src/comparison/mod.rs b/dozer-sql/expression/src/comparison/mod.rs deleted file mode 100644 index 644815288d..0000000000 --- a/dozer-sql/expression/src/comparison/mod.rs +++ /dev/null @@ -1,1975 +0,0 @@ -use crate::error::Error as PipelineError; -use crate::execution::Expression; -use dozer_types::chrono::{DateTime, NaiveDate}; -use dozer_types::rust_decimal::Decimal; -use dozer_types::types::Record; -use dozer_types::types::DATE_FORMAT; -use dozer_types::types::{DozerDuration, DozerPoint, Field, Schema, TimeUnit}; -use num_traits::cast::*; -use std::str::FromStr; -use std::time::Duration; - -macro_rules! define_comparison { - ($id:ident, $op:expr, $function:expr) => { - pub fn $id( - schema: &Schema, - left: &mut Expression, - right: &mut Expression, - record: &Record, - ) -> Result { - let left_p = left.evaluate(&record, schema)?; - let right_p = right.evaluate(&record, schema)?; - - match left_p { - Field::Null => Ok(Field::Null), - Field::Boolean(left_v) => match right_p { - // left: Bool, right: Bool - Field::Boolean(right_v) => Ok(Field::Boolean($function(left_v, right_v))), - // left: Bool, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = bool::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Bool".to_string()) - })?; - Ok(Field::Boolean($function(left_v, right_v_b))) - } - // left: Bool, right: UInt - Field::UInt(right_v) => { - let right_v_b = - bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast( - format!("{}", right_v), - "Bool".to_string(), - ) - })?; - Ok(Field::Boolean($function(left_v, right_v_b))) - } - // left: Bool, right: U128 - Field::U128(right_v) => { - let right_v_b = - bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast( - format!("{}", right_v), - "Bool".to_string(), - ) - })?; - Ok(Field::Boolean($function(left_v, right_v_b))) - } - // left: Bool, right: Int - Field::Int(right_v) => { - let right_v_b = - bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast( - format!("{}", right_v), - "Bool".to_string(), - ) - })?; - Ok(Field::Boolean($function(left_v, right_v_b))) - } - Field::Int8(right_v) => { - let right_v_b = - bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast( - format!("{}", right_v), - "Bool".to_string(), - ) - })?; - Ok(Field::Boolean($function(left_v, right_v_b))) - } - // left: Bool, right: I128 - Field::I128(right_v) => { - let right_v_b = - bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast( - format!("{}", right_v), - "Bool".to_string(), - ) - })?; - Ok(Field::Boolean($function(left_v, right_v_b))) - } - // left: Bool, right: Float - Field::Float(right_v) => { - let right_v_b = - bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast( - format!("{}", right_v), - "Bool".to_string(), - ) - })?; - Ok(Field::Boolean($function(left_v, right_v_b))) - } - // left: Bool, right: Decimal - Field::Decimal(right_v) => { - let right_v_b = - bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast( - format!("{}", right_v), - "Bool".to_string(), - ) - })?; - Ok(Field::Boolean($function(left_v, right_v_b))) - } - Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) - | Field::Null => Ok(Field::Null), - }, - Field::Int(left_v) => match right_p { - // left: Int, right: Int - Field::Int(right_v) => Ok(Field::Boolean($function(left_v, right_v))), - Field::Int8(right_v) => Ok(Field::Boolean($function(left_v, right_v as i64))), - - // left: Int, right: I128 - Field::I128(right_v) => Ok(Field::Boolean($function(left_v as i128, right_v))), - // left: Int, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean($function(left_v, right_v as i64))), - // left: Int, right: U128 - Field::U128(right_v) => Ok(Field::Boolean($function(left_v, right_v as i64))), - // left: Int, right: Float - Field::Float(right_v) => Ok(Field::Boolean($function(left_v as f64, *right_v))), - // left: Int, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = - Decimal::from_i64(left_v).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean($function(left_v_d, right_v))) - } - // left: Int, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = i64::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Int".to_string()) - })?; - Ok(Field::Boolean($function(left_v, right_v_b))) - } - // left: Int, right: Duration - Field::Duration(right_v) => { - let left_v_b = DozerDuration( - Duration::from_nanos(left_v as u64), - TimeUnit::Nanoseconds, - ); - Ok(Field::Boolean($function(left_v_b, right_v))) - } - // left: Int, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::Int8(left_v) => match right_p { - // left: Int, right: Int - Field::Int(right_v) => Ok(Field::Boolean($function(left_v as i64, right_v))), - Field::Int8(right_v) => { - Ok(Field::Boolean($function(left_v as i64, right_v as i64))) - } - - // left: Int, right: I128 - Field::I128(right_v) => Ok(Field::Boolean($function(left_v as i128, right_v))), - // left: Int, right: UInt - Field::UInt(right_v) => { - Ok(Field::Boolean($function(left_v as i64, right_v as i64))) - } - // left: Int, right: U128 - Field::U128(right_v) => { - Ok(Field::Boolean($function(left_v as i64, right_v as i64))) - } - // left: Int, right: Float - Field::Float(right_v) => Ok(Field::Boolean($function(left_v as f64, *right_v))), - // left: Int, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = - Decimal::from_i64(left_v as i64).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean($function(left_v_d, right_v))) - } - // left: Int, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = i64::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Int".to_string()) - })?; - Ok(Field::Boolean($function(left_v as i64, right_v_b))) - } - // left: Int, right: Duration - Field::Duration(right_v) => { - let left_v_b = DozerDuration( - Duration::from_nanos(left_v as u64), - TimeUnit::Nanoseconds, - ); - Ok(Field::Boolean($function(left_v_b, right_v))) - } - // left: Int, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::I128(left_v) => match right_p { - // left: I128, right: Int - Field::Int(right_v) => Ok(Field::Boolean($function(left_v, right_v as i128))), - Field::Int8(right_v) => Ok(Field::Boolean($function(left_v, right_v as i128))), - // left: I128, right: I128 - Field::I128(right_v) => Ok(Field::Boolean($function(left_v, right_v))), - // left: I128, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean($function(left_v, right_v as i128))), - // left: I128, right: U128 - Field::U128(right_v) => Ok(Field::Boolean($function(left_v, right_v as i128))), - // left: I128, right: Float - Field::Float(right_v) => Ok(Field::Boolean($function(left_v as f64, *right_v))), - // left: I128, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = - Decimal::from_i128(left_v).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean($function(left_v_d, right_v))) - } - // left: I128, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = i128::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "I128".to_string()) - })?; - Ok(Field::Boolean($function(left_v, right_v_b))) - } - // left: I128, right: Duration - Field::Duration(right_v) => { - let left_v_b = DozerDuration( - Duration::from_nanos(left_v as u64), - TimeUnit::Nanoseconds, - ); - Ok(Field::Boolean($function(left_v_b, right_v))) - } - // left: I128, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::UInt(left_v) => match right_p { - // left: UInt, right: Int - Field::Int(right_v) => Ok(Field::Boolean($function(left_v as i64, right_v))), - Field::Int8(right_v) => { - Ok(Field::Boolean($function(left_v as i64, right_v as i64))) - } - // left: UInt, right: I128 - Field::I128(right_v) => Ok(Field::Boolean($function(left_v as i128, right_v))), - // left: UInt, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean($function(left_v, right_v))), - // left: UInt, right: U128 - Field::U128(right_v) => Ok(Field::Boolean($function(left_v as u128, right_v))), - // left: UInt, right: Float - Field::Float(right_v) => Ok(Field::Boolean($function(left_v as f64, *right_v))), - // left: UInt, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = - Decimal::from_f64(left_v as f64).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean($function(left_v_d, right_v))) - } - // left: UInt, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = u64::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "UInt".to_string()) - })?; - Ok(Field::Boolean($function(left_v, right_v_b))) - } - // left: UInt, right: Duration - Field::Duration(right_v) => { - let left_v_b = - DozerDuration(Duration::from_nanos(left_v), TimeUnit::Nanoseconds); - Ok(Field::Boolean($function(left_v_b, right_v))) - } - // left: UInt, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::U128(left_v) => match right_p { - // left: U128, right: Int - Field::Int(right_v) => { - Ok(Field::Boolean($function(left_v as i128, right_v as i128))) - } - Field::Int8(right_v) => { - Ok(Field::Boolean($function(left_v as i128, right_v as i128))) - } - // left: U128, right: I128 - Field::I128(right_v) => Ok(Field::Boolean($function(left_v as i128, right_v))), - // left: U128, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean($function(left_v, right_v as u128))), - // left: U128, right: U128 - Field::U128(right_v) => Ok(Field::Boolean($function(left_v, right_v))), - // left: U128, right: Float - Field::Float(right_v) => Ok(Field::Boolean($function(left_v as f64, *right_v))), - // left: U128, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = - Decimal::from_f64(left_v as f64).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean($function(left_v_d, right_v))) - } - // left: U128, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = u128::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "U128".to_string()) - })?; - Ok(Field::Boolean($function(left_v, right_v_b))) - } - // left: U128, right: Duration - Field::Duration(right_v) => { - let left_v_b = DozerDuration( - Duration::from_nanos(left_v as u64), - TimeUnit::Nanoseconds, - ); - Ok(Field::Boolean($function(left_v_b, right_v))) - } - // left: U128, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::Float(left_v) => match right_p { - // left: Float, right: Float - Field::Float(right_v) => Ok(Field::Boolean($function(left_v, right_v))), - // left: Float, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean($function(*left_v, right_v as f64))), - - // left: Float, right: U128 - Field::U128(right_v) => Ok(Field::Boolean($function(*left_v, right_v as f64))), - // left: Float, right: Int - Field::Int(right_v) => Ok(Field::Boolean($function(*left_v, right_v as f64))), - Field::Int8(right_v) => Ok(Field::Boolean($function(*left_v, right_v as f64))), - - // left: Float, right: I128 - Field::I128(right_v) => Ok(Field::Boolean($function(*left_v, right_v as f64))), - // left: Float, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = - Decimal::from_f64(*left_v).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean($function(left_v_d, right_v))) - } - // left: Float, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = f64::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Float".to_string()) - })?; - Ok(Field::Boolean($function(*left_v, right_v_b))) - } - // left: Float, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::Decimal(left_v) => match right_p { - // left: Decimal, right: Float - Field::Float(right_v) => { - let right_v_d = - Decimal::from_f64(*right_v).ok_or(PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean($function(left_v, right_v_d))) - } - // left: Decimal, right: Int - Field::Int(right_v) => { - let right_v_d = - Decimal::from_i64(right_v).ok_or(PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean($function(left_v, right_v_d))) - } - Field::Int8(right_v) => { - let right_v_d = Decimal::from_i64(right_v as i64).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ), - )?; - Ok(Field::Boolean($function(left_v, right_v_d))) - } - // left: Decimal, right: I128 - Field::I128(right_v) => { - let right_v_d = - Decimal::from_i128(right_v).ok_or(PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean($function(left_v, right_v_d))) - } - // left: Decimal, right: UInt - Field::UInt(right_v) => { - let right_v_d = - Decimal::from_u64(right_v).ok_or(PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean($function(left_v, right_v_d))) - } - // left: Decimal, right: U128 - Field::U128(right_v) => { - let right_v_d = - Decimal::from_u128(right_v).ok_or(PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean($function(left_v, right_v_d))) - } - // left: Decimal, right: Decimal - Field::Decimal(right_v) => Ok(Field::Boolean($function(left_v, right_v))), - // left: Decimal, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = Decimal::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ) - })?; - Ok(Field::Boolean($function(left_v, right_v_b))) - } - // left: Decimal, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::String(ref left_v) | Field::Text(ref left_v) => match right_p { - Field::String(ref right_v) => Ok(Field::Boolean($function(left_v, right_v))), - Field::Text(ref right_v) => Ok(Field::Boolean($function(left_v, right_v))), - Field::Null => Ok(Field::Null), - Field::UInt(right_v) => { - let left_val = u64::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(format!("{}", left_v), "UInt".to_string()) - })?; - Ok(Field::Boolean($function(left_val, right_v))) - } - Field::U128(right_v) => { - let left_val = u128::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(format!("{}", left_v), "U128".to_string()) - })?; - Ok(Field::Boolean($function(left_val, right_v))) - } - Field::Int(right_v) => { - let left_val = i64::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(format!("{}", left_v), "Int".to_string()) - })?; - Ok(Field::Boolean($function(left_val, right_v))) - } - Field::Int8(right_v) => { - let left_val = i64::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(format!("{}", left_v), "Int8".to_string()) - })?; - Ok(Field::Boolean($function(left_val, right_v as i64))) - } - Field::I128(right_v) => { - let left_val = i128::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(format!("{}", left_v), "I128".to_string()) - })?; - Ok(Field::Boolean($function(left_val, right_v))) - } - Field::Float(right_v) => { - let left_val = f64::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(format!("{}", left_v), "Float".to_string()) - })?; - Ok(Field::Boolean($function(left_val, *right_v))) - } - Field::Boolean(right_v) => { - let left_val = bool::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(format!("{}", left_v), "Bool".to_string()) - })?; - Ok(Field::Boolean($function(left_val, right_v))) - } - Field::Decimal(right_v) => { - let left_val = Decimal::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ) - })?; - Ok(Field::Boolean($function(left_val, right_v))) - } - Field::Timestamp(right_v) => { - let ts = DateTime::parse_from_rfc3339(left_v).map_err(|_| { - PipelineError::UnableToCast( - format!("{}", left_v), - "Timestamp".to_string(), - ) - })?; - Ok(Field::Boolean($function(ts, right_v))) - } - Field::Date(right_v) => { - let date = - NaiveDate::parse_from_str(left_v, DATE_FORMAT).map_err(|_| { - PipelineError::UnableToCast( - format!("{}", left_v), - "Date".to_string(), - ) - })?; - Ok(Field::Boolean($function(date, right_v))) - } - Field::Point(right_v) => { - let left_val = DozerPoint::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(format!("{}", left_v), "Point".to_string()) - })?; - Ok(Field::Boolean($function(left_val, right_v))) - } - Field::Duration(right_v) => { - let left_val = DozerDuration::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast( - format!("{}", left_v), - "Duration".to_string(), - ) - })?; - Ok(Field::Boolean($function(left_val, right_v))) - } - Field::Binary(_) | Field::Json(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::Timestamp(left_v) => match right_p { - Field::Timestamp(right_v) => Ok(Field::Boolean($function(left_v, right_v))), - Field::String(ref right_v) | Field::Text(ref right_v) => { - let ts = DateTime::parse_from_rfc3339(right_v).map_err(|_| { - PipelineError::UnableToCast( - format!("{}", right_v), - "Timestamp".to_string(), - ) - })?; - Ok(Field::Boolean($function(left_v, ts))) - } - Field::Null => Ok(Field::Null), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::Boolean(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::Date(left_v) => match right_p { - Field::Date(right_v) => Ok(Field::Boolean($function(left_v, right_v))), - Field::String(ref right_v) | Field::Text(ref right_v) => { - let date = - NaiveDate::parse_from_str(right_v, DATE_FORMAT).map_err(|_| { - PipelineError::UnableToCast( - format!("{}", right_v), - "Date".to_string(), - ) - })?; - Ok(Field::Boolean($function(left_v, date))) - } - Field::Null => Ok(Field::Null), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::Boolean(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Timestamp(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::Point(left_v) => match right_p { - Field::Point(right_v) => Ok(Field::Boolean($function(left_v, right_v))), - Field::String(right_v) | Field::Text(right_v) => { - let right_val = DozerPoint::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Point".to_string()) - })?; - Ok(Field::Boolean($function(left_v, right_val))) - } - Field::Null => Ok(Field::Null), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::Boolean(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Timestamp(_) - | Field::Json(_) - | Field::Date(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::Duration(left_v) => match right_p { - Field::Duration(right_v) => Ok(Field::Boolean($function(left_v, right_v))), - Field::String(right_v) | Field::Text(right_v) => { - let right_val = - DozerDuration::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast( - format!("{}", right_v), - "Duration".to_string(), - ) - })?; - Ok(Field::Boolean($function(left_v, right_val))) - } - Field::Null => Ok(Field::Null), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::Boolean(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Timestamp(_) - | Field::Json(_) - | Field::Date(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::Binary(_) | Field::Json(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - } - } - }; -} - -pub fn evaluate_lt( - schema: &Schema, - left: &mut Expression, - right: &mut Expression, - record: &Record, -) -> Result { - let left_p = left.evaluate(record, schema)?; - let right_p = right.evaluate(record, schema)?; - - match left_p { - Field::Null => Ok(Field::Null), - Field::Boolean(left_v) => match right_p { - // left: Bool, right: Bool - Field::Boolean(right_v) => Ok(Field::Boolean(!left_v & right_v)), - // left: Bool, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = bool::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Bool".to_string()) - })?; - Ok(Field::Boolean(!left_v & right_v_b)) - } - // left: Bool, right: UInt - Field::UInt(right_v) => { - let right_v_b = bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Bool".to_string()) - })?; - Ok(Field::Boolean(!left_v & right_v_b)) - } - // left: Bool, right: U128 - Field::U128(right_v) => { - let right_v_b = bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Bool".to_string()) - })?; - Ok(Field::Boolean(!left_v & right_v_b)) - } - // left: Bool, right: Int - Field::Int(right_v) => { - let right_v_b = bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Bool".to_string()) - })?; - Ok(Field::Boolean(!left_v & right_v_b)) - } - - Field::Int8(right_v) => { - let right_v_b = bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Bool".to_string()) - })?; - Ok(Field::Boolean(!left_v & right_v_b)) - } - // left: Bool, right: I128 - Field::I128(right_v) => { - let right_v_b = bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Bool".to_string()) - })?; - Ok(Field::Boolean(!left_v & right_v_b)) - } - // left: Bool, right: Float - Field::Float(right_v) => { - let right_v_b = bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Bool".to_string()) - })?; - Ok(Field::Boolean(!left_v & right_v_b)) - } - // left: Bool, right: Decimal - Field::Decimal(right_v) => { - let right_v_b = bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Bool".to_string()) - })?; - Ok(Field::Boolean(!left_v & right_v_b)) - } - Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) - | Field::Null => Ok(Field::Null), - }, - Field::Int(left_v) => match right_p { - // left: Int, right: Int - Field::Int(right_v) => Ok(Field::Boolean(left_v < right_v)), - Field::Int8(right_v) => Ok(Field::Boolean(left_v < (right_v as i64))), - - // left: Int, right: I128 - Field::I128(right_v) => Ok(Field::Boolean((left_v as i128) < right_v)), - // left: Int, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean(left_v < (right_v as i64))), - // left: Int, right: U128 - Field::U128(right_v) => Ok(Field::Boolean(left_v < (right_v as i64))), - // left: Int, right: Float - Field::Float(right_v) => Ok(Field::Boolean((left_v as f64) < *right_v)), - // left: Int, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = Decimal::from_i64(left_v).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean(left_v_d < right_v)) - } - // left: Int, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = i64::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Int".to_string()) - })?; - Ok(Field::Boolean(left_v < right_v_b)) - } - // left: Int, right: Duration - Field::Duration(right_v) => { - let left_v_b = - DozerDuration(Duration::from_nanos(left_v as u64), TimeUnit::Nanoseconds); - Ok(Field::Boolean(left_v_b < right_v)) - } - // left: Int, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - "<".to_string(), - )), - }, - Field::Int8(left_v) => match right_p { - // left: Int, right: Int - Field::Int(right_v) => Ok(Field::Boolean((left_v as i64) < right_v)), - Field::Int8(right_v) => Ok(Field::Boolean(left_v < right_v)), - - // left: Int, right: I128 - Field::I128(right_v) => Ok(Field::Boolean((left_v as i128) < right_v)), - // left: Int, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean((left_v as i64) < (right_v as i64))), - // left: Int, right: U128 - Field::U128(right_v) => Ok(Field::Boolean((left_v as i64) < (right_v as i64))), - // left: Int, right: Float - Field::Float(right_v) => Ok(Field::Boolean((left_v as f64) < *right_v)), - // left: Int, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = Decimal::from_i64(left_v as i64).ok_or( - PipelineError::UnableToCast(format!("{}", left_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v_d < right_v)) - } - // left: Int, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = i8::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Int".to_string()) - })?; - Ok(Field::Boolean(left_v < right_v_b)) - } - // left: Int, right: Duration - Field::Duration(right_v) => { - let left_v_b = - DozerDuration(Duration::from_nanos(left_v as u64), TimeUnit::Nanoseconds); - Ok(Field::Boolean(left_v_b < right_v)) - } - // left: Int, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - "<".to_string(), - )), - }, - Field::I128(left_v) => match right_p { - // left: I128, right: Int - Field::Int(right_v) => Ok(Field::Boolean(left_v < (right_v as i128))), - Field::Int8(right_v) => Ok(Field::Boolean(left_v < (right_v as i128))), - - // left: I128, right: I128 - Field::I128(right_v) => Ok(Field::Boolean(left_v < right_v)), - // left: I128, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean(left_v < (right_v as i128))), - - // left: I128, right: U128 - Field::U128(right_v) => Ok(Field::Boolean(left_v < (right_v as i128))), - // left: I128, right: Float - Field::Float(right_v) => Ok(Field::Boolean((left_v as f64) < *right_v)), - // left: I128, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = Decimal::from_i128(left_v).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean(left_v_d < right_v)) - } - // left: I128, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = i128::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "I128".to_string()) - })?; - Ok(Field::Boolean(left_v < right_v_b)) - } - // left: I128, right: Duration - Field::Duration(right_v) => { - let left_v_b = - DozerDuration(Duration::from_nanos(left_v as u64), TimeUnit::Nanoseconds); - Ok(Field::Boolean(left_v_b < right_v)) - } - // left: I128, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - "<".to_string(), - )), - }, - Field::UInt(left_v) => match right_p { - // left: UInt, right: Int - Field::Int(right_v) => Ok(Field::Boolean((left_v as i64) < right_v)), - Field::Int8(right_v) => Ok(Field::Boolean((left_v as i64) < (right_v as i64))), - - // left: UInt, right: I128 - Field::I128(right_v) => Ok(Field::Boolean((left_v as i128) < right_v)), - // left: UInt, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean(left_v < right_v)), - // left: UInt, right: U128 - Field::U128(right_v) => Ok(Field::Boolean((left_v as u128) < right_v)), - // left: UInt, right: Float - Field::Float(right_v) => Ok(Field::Boolean((left_v as f64) < *right_v)), - // left: UInt, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = Decimal::from_f64(left_v as f64).ok_or( - PipelineError::UnableToCast(format!("{}", left_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v_d < right_v)) - } - // left: UInt, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = u64::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "UInt".to_string()) - })?; - Ok(Field::Boolean(left_v < right_v_b)) - } - // left: UInt, right: Duration - Field::Duration(right_v) => { - let left_v_b = DozerDuration(Duration::from_nanos(left_v), TimeUnit::Nanoseconds); - Ok(Field::Boolean(left_v_b < right_v)) - } - // left: UInt, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - "<".to_string(), - )), - }, - Field::U128(left_v) => match right_p { - // left: U128, right: Int - Field::Int(right_v) => Ok(Field::Boolean((left_v as i128) < (right_v as i128))), - Field::Int8(right_v) => Ok(Field::Boolean((left_v as i128) < (right_v as i128))), - - // left: U128, right: I128 - Field::I128(right_v) => Ok(Field::Boolean((left_v as i128) < right_v)), - // left: U128, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean(left_v < (right_v as u128))), - // left: U128, right: U128 - Field::U128(right_v) => Ok(Field::Boolean(left_v < right_v)), - // left: U128, right: Float - Field::Float(right_v) => Ok(Field::Boolean((left_v as f64) < *right_v)), - // left: U128, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = Decimal::from_f64(left_v as f64).ok_or( - PipelineError::UnableToCast(format!("{}", left_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v_d < right_v)) - } - // left: U128, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = u128::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "U128".to_string()) - })?; - Ok(Field::Boolean(left_v < right_v_b)) - } - // left: U128, right: Duration - Field::Duration(right_v) => { - let left_v_b = - DozerDuration(Duration::from_nanos(left_v as u64), TimeUnit::Nanoseconds); - Ok(Field::Boolean(left_v_b < right_v)) - } - // left: U128, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - "<".to_string(), - )), - }, - Field::Float(left_v) => match right_p { - // left: Float, right: Float - Field::Float(right_v) => Ok(Field::Boolean(left_v < right_v)), - // left: Float, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean(*left_v < (right_v as f64))), - // left: Float, right: U128 - Field::U128(right_v) => Ok(Field::Boolean(*left_v < (right_v as f64))), - // left: Float, right: Int - Field::Int(right_v) => Ok(Field::Boolean(*left_v < (right_v as f64))), - Field::Int8(right_v) => Ok(Field::Boolean(*left_v < (right_v as f64))), - - // left: Float, right: I128 - Field::I128(right_v) => Ok(Field::Boolean(*left_v < (right_v as f64))), - // left: Float, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = Decimal::from_f64(*left_v).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean(left_v_d < right_v)) - } - // left: Float, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = f64::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Float".to_string()) - })?; - Ok(Field::Boolean(*left_v < right_v_b)) - } - // left: Float, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - "<".to_string(), - )), - }, - Field::Decimal(left_v) => { - match right_p { - // left: Decimal, right: Float - Field::Float(right_v) => { - let right_v_d = Decimal::from_f64(*right_v).ok_or( - PipelineError::UnableToCast(format!("{}", right_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v < right_v_d)) - } - // left: Decimal, right: Int - Field::Int(right_v) => { - let right_v_d = Decimal::from_i64(right_v).ok_or( - PipelineError::UnableToCast(format!("{}", right_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v < right_v_d)) - } - Field::Int8(right_v) => { - let right_v_d = Decimal::from_i64(right_v as i64).ok_or( - PipelineError::UnableToCast(format!("{}", right_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v < right_v_d)) - } - // left: Decimal, right: I128 - Field::I128(right_v) => { - let right_v_d = Decimal::from_i128(right_v).ok_or( - PipelineError::UnableToCast(format!("{}", right_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v < right_v_d)) - } - // left: Decimal, right: UInt - Field::UInt(right_v) => { - let right_v_d = Decimal::from_u64(right_v).ok_or( - PipelineError::UnableToCast(format!("{}", right_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v < right_v_d)) - } - // left: Decimal, right: U128 - Field::U128(right_v) => { - let right_v_d = Decimal::from_u128(right_v).ok_or( - PipelineError::UnableToCast(format!("{}", right_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v < right_v_d)) - } - // left: Decimal, right: Decimal - Field::Decimal(right_v) => Ok(Field::Boolean(left_v < right_v)), - // left: Decimal, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = Decimal::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Decimal".to_string()) - })?; - Ok(Field::Boolean(left_v < right_v_b)) - } - // left: Decimal, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - "<".to_string(), - )), - } - } - Field::String(ref left_v) | Field::Text(ref left_v) => match right_p { - Field::String(ref right_v) => Ok(Field::Boolean(left_v < right_v)), - Field::Text(ref right_v) => Ok(Field::Boolean(left_v < right_v)), - Field::Null => Ok(Field::Null), - Field::UInt(right_v) => { - let left_val = u64::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "UInt".to_string()) - })?; - Ok(Field::Boolean(left_val < right_v)) - } - Field::U128(right_v) => { - let left_val = u128::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "U128".to_string()) - })?; - Ok(Field::Boolean(left_val < right_v)) - } - Field::Int(right_v) => { - let left_val = i64::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Int".to_string()) - })?; - Ok(Field::Boolean(left_val < right_v)) - } - Field::Int8(right_v) => { - let left_val = i64::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Int".to_string()) - })?; - Ok(Field::Boolean(left_val < (right_v as i64))) - } - Field::I128(right_v) => { - let left_val = i128::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "I128".to_string()) - })?; - Ok(Field::Boolean(left_val < right_v)) - } - Field::Float(right_v) => { - let left_val = f64::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Float".to_string()) - })?; - Ok(Field::Boolean(left_val < *right_v)) - } - Field::Boolean(right_v) => { - let left_val = bool::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Bool".to_string()) - })?; - Ok(Field::Boolean(!left_val & right_v)) - } - Field::Decimal(right_v) => { - let left_val = Decimal::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Decimal".to_string()) - })?; - Ok(Field::Boolean(left_val < right_v)) - } - Field::Timestamp(right_v) => { - let ts = DateTime::parse_from_rfc3339(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Timestamp".to_string()) - })?; - Ok(Field::Boolean(ts < right_v)) - } - Field::Date(right_v) => { - let date = NaiveDate::parse_from_str(left_v, DATE_FORMAT).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Date".to_string()) - })?; - Ok(Field::Boolean(date < right_v)) - } - Field::Point(right_v) => { - let left_val = DozerPoint::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Point".to_string()) - })?; - Ok(Field::Boolean(left_val < right_v)) - } - Field::Duration(right_v) => { - let left_val = DozerDuration::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Duration".to_string()) - })?; - Ok(Field::Boolean(left_val < right_v)) - } - Field::Binary(_) | Field::Json(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - "<".to_string(), - )), - }, - Field::Timestamp(left_v) => match right_p { - Field::Timestamp(right_v) => Ok(Field::Boolean(left_v < right_v)), - Field::String(ref right_v) | Field::Text(ref right_v) => { - let ts = DateTime::parse_from_rfc3339(right_v).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Timestamp".to_string()) - })?; - Ok(Field::Boolean(left_v < ts)) - } - Field::Null => Ok(Field::Null), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::Boolean(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - "<".to_string(), - )), - }, - Field::Date(left_v) => match right_p { - Field::Date(right_v) => Ok(Field::Boolean(left_v < right_v)), - Field::String(ref right_v) | Field::Text(ref right_v) => { - let date = NaiveDate::parse_from_str(right_v, DATE_FORMAT).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Date".to_string()) - })?; - Ok(Field::Boolean(left_v < date)) - } - Field::Null => Ok(Field::Null), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::Boolean(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Timestamp(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - "<".to_string(), - )), - }, - Field::Point(left_v) => match right_p { - Field::Point(right_v) => Ok(Field::Boolean(left_v < right_v)), - Field::String(right_v) | Field::Text(right_v) => { - let right_val = DozerPoint::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Point".to_string()) - })?; - Ok(Field::Boolean(left_v < right_val)) - } - Field::Null => Ok(Field::Null), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::Boolean(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Timestamp(_) - | Field::Json(_) - | Field::Date(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - "<".to_string(), - )), - }, - Field::Duration(left_v) => match right_p { - Field::Duration(right_v) => Ok(Field::Boolean(left_v < right_v)), - Field::String(right_v) | Field::Text(right_v) => { - let right_val = DozerDuration::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Duration".to_string()) - })?; - Ok(Field::Boolean(left_v < right_val)) - } - Field::Null => Ok(Field::Null), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::Boolean(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Timestamp(_) - | Field::Json(_) - | Field::Date(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - "<".to_string(), - )), - }, - Field::Binary(_) | Field::Json(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - "<".to_string(), - )), - } -} - -pub fn evaluate_gt( - schema: &Schema, - left: &mut Expression, - right: &mut Expression, - record: &Record, -) -> Result { - let left_p = left.evaluate(record, schema)?; - let right_p = right.evaluate(record, schema)?; - - match left_p { - Field::Null => Ok(Field::Null), - Field::Boolean(left_v) => match right_p { - // left: Bool, right: Bool - Field::Boolean(right_v) => Ok(Field::Boolean(left_v & !right_v)), - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = bool::from_str(right_v.as_str()).unwrap(); - Ok(Field::Boolean(left_v & !right_v_b)) - } - // left: Bool, right: UInt - Field::UInt(right_v) => { - let right_v_b = bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Bool".to_string()) - })?; - Ok(Field::Boolean(left_v & !right_v_b)) - } - // left: Bool, right: U128 - Field::U128(right_v) => { - let right_v_b = bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Bool".to_string()) - })?; - Ok(Field::Boolean(left_v & !right_v_b)) - } - // left: Bool, right: Int - Field::Int(right_v) => { - let right_v_b = bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Bool".to_string()) - })?; - Ok(Field::Boolean(left_v & !right_v_b)) - } - Field::Int8(right_v) => { - let right_v_b = bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Bool".to_string()) - })?; - Ok(Field::Boolean(left_v & !right_v_b)) - } - // left: Bool, right: I128 - Field::I128(right_v) => { - let right_v_b = bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Bool".to_string()) - })?; - Ok(Field::Boolean(left_v & !right_v_b)) - } - // left: Bool, right: Float - Field::Float(right_v) => { - let right_v_b = bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Bool".to_string()) - })?; - Ok(Field::Boolean(left_v & !right_v_b)) - } - // left: Bool, right: Decimal - Field::Decimal(right_v) => { - let right_v_b = bool::from_str(right_v.to_string().as_str()).map_err(|_| { - PipelineError::UnableToCast(format!("{}", right_v), "Bool".to_string()) - })?; - Ok(Field::Boolean(left_v & !right_v_b)) - } - Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) - | Field::Null => Ok(Field::Null), - }, - Field::Int(left_v) => match right_p { - // left: Int, right: Int - Field::Int(right_v) => Ok(Field::Boolean(left_v > right_v)), - Field::Int8(right_v) => Ok(Field::Boolean(left_v > (right_v as i64))), - - // left: Int, right: I128 - Field::I128(right_v) => Ok(Field::Boolean((left_v as i128) > right_v)), - // left: Int, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean(left_v > (right_v as i64))), - // left: Int, right: U128 - Field::U128(right_v) => Ok(Field::Boolean(left_v > (right_v as i64))), - // left: Int, right: Float - Field::Float(right_v) => Ok(Field::Boolean(left_v as f64 > *right_v)), - // left: Int, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = Decimal::from_i64(left_v).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean(left_v_d > right_v)) - } - // left: Int, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = i64::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Int".to_string()) - })?; - Ok(Field::Boolean(left_v > right_v_b)) - } - // left: Int, right: Duration - Field::Duration(right_v) => { - let left_v_b = - DozerDuration(Duration::from_nanos(left_v as u64), TimeUnit::Nanoseconds); - Ok(Field::Boolean(left_v_b > right_v)) - } - // left: Int, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - ">".to_string(), - )), - }, - Field::Int8(left_v) => match right_p { - // left: Int, right: Int - Field::Int(right_v) => Ok(Field::Boolean((left_v as i64) > right_v)), - Field::Int8(right_v) => Ok(Field::Boolean(left_v > right_v)), - - // left: Int, right: I128 - Field::I128(right_v) => Ok(Field::Boolean((left_v as i128) > right_v)), - // left: Int, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean((left_v as i64) > (right_v as i64))), - // left: Int, right: U128 - Field::U128(right_v) => Ok(Field::Boolean((left_v as i64) > (right_v as i64))), - // left: Int, right: Float - Field::Float(right_v) => Ok(Field::Boolean(left_v as f64 > *right_v)), - // left: Int, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = Decimal::from_i64(left_v as i64).ok_or( - PipelineError::UnableToCast(format!("{}", left_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v_d > right_v)) - } - // left: Int, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = i64::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Int".to_string()) - })?; - Ok(Field::Boolean((left_v as i64) > right_v_b)) - } - // left: Int, right: Duration - Field::Duration(right_v) => { - let left_v_b = - DozerDuration(Duration::from_nanos(left_v as u64), TimeUnit::Nanoseconds); - Ok(Field::Boolean(left_v_b > right_v)) - } - // left: Int, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - ">".to_string(), - )), - }, - Field::I128(left_v) => match right_p { - // left: I128, right: Int - Field::Int(right_v) => Ok(Field::Boolean(left_v > (right_v as i128))), - Field::Int8(right_v) => Ok(Field::Boolean(left_v > (right_v as i128))), - - // left: I128, right: I128 - Field::I128(right_v) => Ok(Field::Boolean(left_v > right_v)), - // left: I128, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean(left_v > (right_v as i128))), - // left: I128, right: U128 - Field::U128(right_v) => Ok(Field::Boolean(left_v > (right_v as i128))), - // left: I128, right: Float - Field::Float(right_v) => Ok(Field::Boolean((left_v as f64) > *right_v)), - // left: I128, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = Decimal::from_i128(left_v).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean(left_v_d > right_v)) - } - // left: I128, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = i128::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "I128".to_string()) - })?; - Ok(Field::Boolean(left_v > right_v_b)) - } - // left: I128, right: Duration - Field::Duration(right_v) => { - let left_v_b = - DozerDuration(Duration::from_nanos(left_v as u64), TimeUnit::Nanoseconds); - Ok(Field::Boolean(left_v_b > right_v)) - } - // left: I128, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - ">".to_string(), - )), - }, - Field::UInt(left_v) => match right_p { - // left: UInt, right: Int - Field::Int(right_v) => Ok(Field::Boolean((left_v as i64) > right_v)), - Field::Int8(right_v) => Ok(Field::Boolean((left_v as i64) > (right_v as i64))), - - // left: UInt, right: I128 - Field::I128(right_v) => Ok(Field::Boolean((left_v as i128) > right_v)), - // left: UInt, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean(left_v > right_v)), - // left: UInt, right: U128 - Field::U128(right_v) => Ok(Field::Boolean((left_v as u128) > right_v)), - // left: UInt, right: Float - Field::Float(right_v) => Ok(Field::Boolean((left_v as f64) > *right_v)), - // left: UInt, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = Decimal::from_f64(left_v as f64).ok_or( - PipelineError::UnableToCast(format!("{}", left_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v_d > right_v)) - } - // left: UInt, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = u64::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "UInt".to_string()) - })?; - Ok(Field::Boolean(left_v > right_v_b)) - } - // left: UInt, right: Duration - Field::Duration(right_v) => { - let left_v_b = DozerDuration(Duration::from_nanos(left_v), TimeUnit::Nanoseconds); - Ok(Field::Boolean(left_v_b > right_v)) - } - // left: UInt, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - ">".to_string(), - )), - }, - Field::U128(left_v) => match right_p { - // left: U128, right: Int - Field::Int(right_v) => Ok(Field::Boolean((left_v as i128) > (right_v as i128))), - Field::Int8(right_v) => Ok(Field::Boolean((left_v as i128) > (right_v as i128))), - - // left: U128, right: I128 - Field::I128(right_v) => Ok(Field::Boolean((left_v as i128) > right_v)), - // left: U128, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean(left_v > (right_v as u128))), - // left: U128, right: U128 - Field::U128(right_v) => Ok(Field::Boolean(left_v > right_v)), - // left: U128, right: Float - Field::Float(right_v) => Ok(Field::Boolean((left_v as f64) > *right_v)), - // left: U128, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = Decimal::from_f64(left_v as f64).ok_or( - PipelineError::UnableToCast(format!("{}", left_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v_d > right_v)) - } - // left: U128, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = u128::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "U128".to_string()) - })?; - Ok(Field::Boolean(left_v > right_v_b)) - } - // left: U128, right: Duration - Field::Duration(right_v) => { - let left_v_b = - DozerDuration(Duration::from_nanos(left_v as u64), TimeUnit::Nanoseconds); - Ok(Field::Boolean(left_v_b > right_v)) - } - // left: U128, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - ">".to_string(), - )), - }, - Field::Float(left_v) => match right_p { - // left: Float, right: Float - Field::Float(right_v) => Ok(Field::Boolean(left_v > right_v)), - // left: Float, right: UInt - Field::UInt(right_v) => Ok(Field::Boolean(*left_v > right_v as f64)), - // left: Float, right: U128 - Field::U128(right_v) => Ok(Field::Boolean(*left_v > right_v as f64)), - // left: Float, right: Int - Field::Int(right_v) => Ok(Field::Boolean(*left_v > right_v as f64)), - Field::Int8(right_v) => Ok(Field::Boolean(*left_v > right_v as f64)), - // left: Float, right: I128 - Field::I128(right_v) => Ok(Field::Boolean(*left_v > right_v as f64)), - // left: Float, right: Decimal - Field::Decimal(right_v) => { - let left_v_d = Decimal::from_f64(*left_v).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?; - Ok(Field::Boolean(left_v_d > right_v)) - } - // left: Float, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = f64::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Float".to_string()) - })?; - Ok(Field::Boolean(*left_v > right_v_b)) - } - // left: Float, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - ">".to_string(), - )), - }, - Field::Decimal(left_v) => { - match right_p { - // left: Decimal, right: Float - Field::Float(right_v) => { - let right_v_d = Decimal::from_f64(*right_v).ok_or( - PipelineError::UnableToCast(format!("{}", right_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v > right_v_d)) - } - // left: Decimal, right: Int - Field::Int(right_v) => { - let right_v_d = Decimal::from_i64(right_v).ok_or( - PipelineError::UnableToCast(format!("{}", right_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v > right_v_d)) - } - Field::Int8(right_v) => { - let right_v_d = Decimal::from_i64(right_v as i64).ok_or( - PipelineError::UnableToCast(format!("{}", right_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v > right_v_d)) - } - // left: Decimal, right: I128 - Field::I128(right_v) => { - let right_v_d = Decimal::from_i128(right_v).ok_or( - PipelineError::UnableToCast(format!("{}", right_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v > right_v_d)) - } - // left: Decimal, right: UInt - Field::UInt(right_v) => { - let right_v_d = Decimal::from_u64(right_v).ok_or( - PipelineError::UnableToCast(format!("{}", right_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v > right_v_d)) - } - // left: Decimal, right: U128 - Field::U128(right_v) => { - let right_v_d = Decimal::from_u128(right_v).ok_or( - PipelineError::UnableToCast(format!("{}", right_v), "Decimal".to_string()), - )?; - Ok(Field::Boolean(left_v > right_v_d)) - } - // left: Decimal, right: Decimal - Field::Decimal(right_v) => Ok(Field::Boolean(left_v > right_v)), - // left: Decimal, right: String or Text - Field::String(right_v) | Field::Text(right_v) => { - let right_v_b = Decimal::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Decimal".to_string()) - })?; - Ok(Field::Boolean(left_v > right_v_b)) - } - // left: Decimal, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - ">".to_string(), - )), - } - } - Field::String(ref left_v) | Field::Text(ref left_v) => match right_p { - Field::String(ref right_v) => Ok(Field::Boolean(left_v > right_v)), - Field::Text(ref right_v) => Ok(Field::Boolean(left_v > right_v)), - Field::Null => Ok(Field::Null), - Field::UInt(right_v) => { - let left_val = u64::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "UInt".to_string()) - })?; - Ok(Field::Boolean(left_val > right_v)) - } - Field::U128(right_v) => { - let left_val = u128::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "U128".to_string()) - })?; - Ok(Field::Boolean(left_val > right_v)) - } - Field::Int(right_v) => { - let left_val = i64::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Int".to_string()) - })?; - Ok(Field::Boolean(left_val > right_v)) - } - Field::Int8(right_v) => { - let left_val = i64::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Int".to_string()) - })?; - Ok(Field::Boolean(left_val > (right_v as i64))) - } - Field::I128(right_v) => { - let left_val = i128::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "I128".to_string()) - })?; - Ok(Field::Boolean(left_val > right_v)) - } - Field::Float(right_v) => { - let left_val = f64::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Float".to_string()) - })?; - Ok(Field::Boolean(left_val > *right_v)) - } - Field::Boolean(right_v) => { - let left_val = bool::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Bool".to_string()) - })?; - Ok(Field::Boolean(left_val & !right_v)) - } - Field::Decimal(right_v) => { - let left_val = Decimal::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Decimal".to_string()) - })?; - Ok(Field::Boolean(left_val > right_v)) - } - Field::Timestamp(right_v) => { - let ts = DateTime::parse_from_rfc3339(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Timestamp".to_string()) - })?; - Ok(Field::Boolean(ts > right_v)) - } - Field::Date(right_v) => { - let date = NaiveDate::parse_from_str(left_v, DATE_FORMAT).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Date".to_string()) - })?; - Ok(Field::Boolean(date > right_v)) - } - Field::Point(right_v) => { - let left_val = DozerPoint::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Point".to_string()) - })?; - Ok(Field::Boolean(left_val > right_v)) - } - Field::Duration(right_v) => { - let left_val = DozerDuration::from_str(left_v).map_err(|_| { - PipelineError::UnableToCast(left_v.to_string(), "Duration".to_string()) - })?; - Ok(Field::Boolean(left_val > right_v)) - } - Field::Binary(_) | Field::Json(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - ">".to_string(), - )), - }, - Field::Timestamp(left_v) => match right_p { - Field::Timestamp(right_v) => Ok(Field::Boolean(left_v > right_v)), - Field::String(ref right_v) | Field::Text(ref right_v) => { - let ts = DateTime::parse_from_rfc3339(right_v).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Timestamp".to_string()) - })?; - Ok(Field::Boolean(left_v > ts)) - } - Field::Null => Ok(Field::Null), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::Boolean(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - ">".to_string(), - )), - }, - Field::Date(left_v) => match right_p { - Field::Date(right_v) => Ok(Field::Boolean(left_v > right_v)), - Field::String(ref right_v) | Field::Text(ref right_v) => { - let date = NaiveDate::parse_from_str(right_v, DATE_FORMAT).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Date".to_string()) - })?; - Ok(Field::Boolean(left_v > date)) - } - Field::Null => Ok(Field::Null), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::Boolean(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Timestamp(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - ">".to_string(), - )), - }, - Field::Point(left_v) => match right_p { - Field::Point(right_v) => Ok(Field::Boolean(left_v > right_v)), - Field::String(right_v) | Field::Text(right_v) => { - let right_val = DozerPoint::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Point".to_string()) - })?; - Ok(Field::Boolean(left_v > right_val)) - } - Field::Null => Ok(Field::Null), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::Boolean(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Timestamp(_) - | Field::Json(_) - | Field::Date(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - ">".to_string(), - )), - }, - Field::Duration(left_v) => match right_p { - Field::Duration(right_v) => Ok(Field::Boolean(left_v > right_v)), - Field::String(right_v) | Field::Text(right_v) => { - let right_val = DozerDuration::from_str(right_v.as_str()).map_err(|_| { - PipelineError::UnableToCast(right_v.to_string(), "Duration".to_string()) - })?; - Ok(Field::Boolean(left_v > right_val)) - } - Field::Null => Ok(Field::Null), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::Boolean(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Timestamp(_) - | Field::Json(_) - | Field::Date(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - ">".to_string(), - )), - }, - Field::Binary(_) | Field::Json(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - ">".to_string(), - )), - } -} - -fn eq(left: T, right: T) -> bool { - left == right -} - -fn ne(left: T, right: T) -> bool { - left != right -} - -fn le(left: T, right: T) -> bool { - left <= right -} - -fn ge(left: T, right: T) -> bool { - left >= right -} - -define_comparison!(evaluate_eq, "=", eq); -define_comparison!(evaluate_ne, "!=", ne); -define_comparison!(evaluate_lte, "<=", le); -define_comparison!(evaluate_gte, ">=", ge); - -#[cfg(test)] -mod tests; diff --git a/dozer-sql/expression/src/comparison/tests.rs b/dozer-sql/expression/src/comparison/tests.rs deleted file mode 100644 index 17823e133c..0000000000 --- a/dozer-sql/expression/src/comparison/tests.rs +++ /dev/null @@ -1,523 +0,0 @@ -use crate::tests::ArbitraryDecimal; - -use super::*; - -use dozer_types::{ordered_float::OrderedFloat, rust_decimal::Decimal}; -use num_traits::FromPrimitive; -use proptest::prelude::*; -use Expression::Literal; - -#[test] -fn test_comparison() { - proptest!(ProptestConfig::with_cases(1000), move |( - u_num1: u64, u_num2: u64, i_num1: i64, i_num2: i64, f_num1: f64, f_num2: f64, d_num1: ArbitraryDecimal, d_num2: ArbitraryDecimal)| { - let row = Record::new(vec![]); - - let mut uint1 = Literal(Field::UInt(u_num1)); - let mut uint2 = Literal(Field::UInt(u_num2)); - let mut int1 = Literal(Field::Int(i_num1)); - let mut int2 = Literal(Field::Int(i_num2)); - let mut float1 = Literal(Field::Float(OrderedFloat(f_num1))); - let mut float2 = Literal(Field::Float(OrderedFloat(f_num2))); - let mut dec1 = Literal(Field::Decimal(d_num1.0)); - let mut dec2 = Literal(Field::Decimal(d_num2.0)); - let mut null = Literal(Field::Null); - - // eq: UInt - let mut uint1_clone = uint1.clone(); - test_eq(&mut uint1, &mut uint1_clone, &row, None); - if u_num1 == u_num2 && u_num1 as i64 == i_num1 && u_num1 as f64 == f_num1 && Decimal::from(u_num1) == d_num1.0 { - test_eq(&mut uint1, &mut uint2, &row, None); - test_eq(&mut uint1, &mut int1, &row, None); - test_eq(&mut uint1, &mut float1, &row, None); - test_eq(&mut uint1, &mut dec1, &row, None); - test_eq(&mut uint1, &mut null, &row, Some(Field::Null)); - - test_gte(&mut uint1, &mut uint2, &row, None); - test_gte(&mut uint1, &mut int1, &row, None); - test_gte(&mut uint1, &mut float1, &row, None); - test_gte(&mut uint1, &mut dec1, &row, None); - test_gte(&mut uint1, &mut null, &row, Some(Field::Null)); - - test_lte(&mut uint1, &mut uint2, &row, None); - test_lte(&mut uint1, &mut int1, &row, None); - test_lte(&mut uint1, &mut float1, &row, None); - test_lte(&mut uint1, &mut dec1, &row, None); - test_lte(&mut uint1, &mut null, &row, Some(Field::Null)); - } - - // eq: Int - let mut int1_clone = int1.clone(); - test_eq(&mut int1, &mut int1_clone, &row, None); - if i_num1 == u_num1 as i64 && i_num1 == i_num2 && i_num1 as f64 == f_num1 && Decimal::from(i_num1) == d_num1.0 { - test_eq(&mut int1, &mut uint1, &row, None); - test_eq(&mut int1, &mut int2, &row, None); - test_eq(&mut int1, &mut float1, &row, None); - test_eq(&mut int1, &mut dec1, &row, None); - test_eq(&mut int1, &mut null, &row, Some(Field::Null)); - - test_gte(&mut int1, &mut uint1, &row, None); - test_gte(&mut int1, &mut int2, &row, None); - test_gte(&mut int1, &mut float1, &row, None); - test_gte(&mut int1, &mut dec1, &row, None); - test_gte(&mut int1, &mut null, &row, Some(Field::Null)); - - test_lte(&mut int1, &mut uint1, &row, None); - test_lte(&mut int1, &mut int2, &row, None); - test_lte(&mut int1, &mut float1, &row, None); - test_lte(&mut int1, &mut dec1, &row, None); - test_lte(&mut int1, &mut null, &row, Some(Field::Null)); - } - - // eq: Float - let mut float1_clone = float1.clone(); - test_eq(&mut float1, &mut float1_clone, &row, None); - let d_val = Decimal::from_f64(f_num1); - if d_val.is_some() && f_num1 == u_num1 as f64 && f_num1 == i_num1 as f64 && f_num1 == f_num2 && d_val.unwrap() == d_num1.0 { - test_eq(&mut float1, &mut uint1, &row, None); - test_eq(&mut float1, &mut int1, &row, None); - test_eq(&mut float1, &mut float2, &row, None); - test_eq(&mut float1, &mut dec1, &row, None); - test_eq(&mut float1, &mut null, &row, Some(Field::Null)); - - test_gte(&mut float1, &mut uint1, &row, None); - test_gte(&mut float1, &mut int1, &row, None); - test_gte(&mut float1, &mut float2, &row, None); - test_gte(&mut float1, &mut dec1, &row, None); - test_gte(&mut float1, &mut null, &row, Some(Field::Null)); - - test_lte(&mut float1, &mut uint1, &row, None); - test_lte(&mut float1, &mut int1, &row, None); - test_lte(&mut float1, &mut float2, &row, None); - test_lte(&mut float1, &mut dec1, &row, None); - test_lte(&mut float1, &mut null, &row, Some(Field::Null)); - } - - // eq: Decimal - let mut dec1_clone = dec1.clone(); - test_eq(&mut dec1, &mut dec1_clone, &row, None); - let d_val = Decimal::from_f64(f_num1); - if d_val.is_some() && d_num1.0 == Decimal::from(u_num1) && d_num1.0 == Decimal::from(i_num1) && d_num1.0 == d_val.unwrap() && d_num1.0 == d_num2.0 { - test_eq(&mut dec1, &mut uint1, &row, None); - test_eq(&mut dec1, &mut int1, &row, None); - test_eq(&mut dec1, &mut float1, &row, None); - test_eq(&mut dec1, &mut dec2, &row, None); - test_eq(&mut dec1, &mut null, &row, Some(Field::Null)); - - test_gte(&mut dec1, &mut uint1, &row, None); - test_gte(&mut dec1, &mut int1, &row, None); - test_gte(&mut dec1, &mut float1, &row, None); - test_gte(&mut dec1, &mut dec2, &row, None); - test_gte(&mut dec1, &mut null, &row, Some(Field::Null)); - - test_lte(&mut dec1, &mut uint1, &row, None); - test_lte(&mut dec1, &mut int1, &row, None); - test_lte(&mut dec1, &mut float1, &row, None); - test_lte(&mut dec1, &mut dec2, &row, None); - test_lte(&mut dec1, &mut null, &row, Some(Field::Null)); - } - - // eq: Null - test_eq(&mut null, &mut uint2, &row, Some(Field::Null)); - test_eq(&mut null, &mut int2, &row, Some(Field::Null)); - test_eq(&mut null, &mut float2, &row, Some(Field::Null)); - test_eq(&mut null, &mut dec2, &row, Some(Field::Null)); - let mut null_clone = null.clone(); - test_eq(&mut null, &mut null_clone, &row, Some(Field::Null)); - - // not eq: UInt - if u_num1 != u_num2 && u_num1 as i64 != i_num1 && u_num1 as f64 != f_num1 && Decimal::from(u_num1) != d_num1.0 { - test_eq(&mut uint1, &mut uint2, &row, Some(Field::Boolean(false))); - test_eq(&mut uint1, &mut int1, &row, Some(Field::Boolean(false))); - test_eq(&mut uint1, &mut float1, &row, Some(Field::Boolean(false))); - test_eq(&mut uint1, &mut dec1, &row, Some(Field::Boolean(false))); - test_eq(&mut uint1, &mut null, &row, Some(Field::Null)); - - test_ne(&mut uint1, &mut uint2, &row, None); - test_ne(&mut uint1, &mut int1, &row, None); - test_ne(&mut uint1, &mut float1, &row, None); - test_ne(&mut uint1, &mut dec1, &row, None); - test_ne(&mut uint1, &mut null, &row, Some(Field::Null)); - } - - // not eq: Int - if i_num1 != u_num1 as i64 && i_num1 != i_num2 && i_num1 as f64 != f_num1 && Decimal::from(i_num1) != d_num1.0 { - test_eq(&mut int1, &mut uint1, &row, Some(Field::Boolean(false))); - test_eq(&mut int1, &mut int2, &row, Some(Field::Boolean(false))); - test_eq(&mut int1, &mut float1, &row, Some(Field::Boolean(false))); - test_eq(&mut int1, &mut dec1, &row, Some(Field::Boolean(false))); - test_eq(&mut int1, &mut null, &row, Some(Field::Null)); - - test_ne(&mut int1, &mut uint1, &row, None); - test_ne(&mut int1, &mut int2, &row, None); - test_ne(&mut int1, &mut float1, &row, None); - test_ne(&mut int1, &mut dec1, &row, None); - test_ne(&mut int1, &mut null, &row, Some(Field::Null)); - } - - // not eq: Float - let d_val = Decimal::from_f64(f_num1); - if d_val.is_some() && f_num1 != u_num1 as f64 && f_num1 != i_num1 as f64 && f_num1 != f_num2 && d_val.unwrap() != d_num1.0 { - test_eq(&mut float1, &mut uint1, &row, Some(Field::Boolean(false))); - test_eq(&mut float1, &mut int1, &row, Some(Field::Boolean(false))); - test_eq(&mut float1, &mut float2, &row, Some(Field::Boolean(false))); - test_eq(&mut float1, &mut dec1, &row, Some(Field::Boolean(false))); - test_eq(&mut float1, &mut null, &row, Some(Field::Null)); - - test_ne(&mut float1, &mut uint1, &row, None); - test_ne(&mut float1, &mut int1, &row, None); - test_ne(&mut float1, &mut float2, &row, None); - test_ne(&mut float1, &mut dec1, &row, None); - test_ne(&mut float1, &mut null, &row, Some(Field::Null)); - } - - // not eq: Decimal - let d_val = Decimal::from_f64(f_num1); - if d_val.is_some() && d_num1.0 != Decimal::from(u_num1) && d_num1.0 != Decimal::from(i_num1) && d_num1.0 != d_val.unwrap() && d_num1.0 != d_num2.0 { - test_eq(&mut dec1, &mut uint1, &row, Some(Field::Boolean(false))); - test_eq(&mut dec1, &mut int1, &row, Some(Field::Boolean(false))); - test_eq(&mut dec1, &mut float1, &row, Some(Field::Boolean(false))); - test_eq(&mut dec1, &mut dec2, &row, Some(Field::Boolean(false))); - test_eq(&mut dec1, &mut null, &row, Some(Field::Null)); - - test_ne(&mut dec1, &mut uint1, &row, None); - test_ne(&mut dec1, &mut int1, &row, None); - test_ne(&mut dec1, &mut float1, &row, None); - test_ne(&mut dec1, &mut dec2, &row, None); - test_ne(&mut dec1, &mut null, &row, Some(Field::Null)); - } - - // not eq: Null - test_eq(&mut null, &mut uint2, &row, Some(Field::Null)); - test_eq(&mut null, &mut int2, &row, Some(Field::Null)); - test_eq(&mut null, &mut float2, &row, Some(Field::Null)); - test_eq(&mut null, &mut dec2, &row, Some(Field::Null)); - - test_ne(&mut null, &mut uint2, &row, Some(Field::Null)); - test_ne(&mut null, &mut int2, &row, Some(Field::Null)); - test_ne(&mut null, &mut float2, &row, Some(Field::Null)); - test_ne(&mut null, &mut dec2, &row, Some(Field::Null)); - - // gt: UInt - if u_num1 > u_num2 && u_num1 as i64 > i_num1 && u_num1 as f64 > f_num1 && Decimal::from(u_num1) > d_num1.0 { - test_gt(&mut uint1, &mut uint2, &row, None); - test_gt(&mut uint1, &mut int1, &row, None); - test_gt(&mut uint1, &mut float1, &row, None); - test_gt(&mut uint1, &mut dec1, &row, None); - test_gt(&mut uint1, &mut null, &row, Some(Field::Null)); - - test_gte(&mut uint1, &mut uint2, &row, None); - test_gte(&mut uint1, &mut int1, &row, None); - test_gte(&mut uint1, &mut float1, &row, None); - test_gte(&mut uint1, &mut dec1, &row, None); - test_gte(&mut uint1, &mut null, &row, Some(Field::Null)); - - test_lt(&mut uint1, &mut uint2, &row, Some(Field::Boolean(false))); - test_lt(&mut uint1, &mut int1, &row, Some(Field::Boolean(false))); - test_lt(&mut uint1, &mut float1, &row, Some(Field::Boolean(false))); - test_lt(&mut uint1, &mut dec1, &row, Some(Field::Boolean(false))); - test_lt(&mut uint1, &mut null, &row, Some(Field::Null)); - - test_lte(&mut uint1, &mut uint2, &row, Some(Field::Boolean(false))); - test_lte(&mut uint1, &mut int1, &row, Some(Field::Boolean(false))); - test_lte(&mut uint1, &mut float1, &row, Some(Field::Boolean(false))); - test_lte(&mut uint1, &mut dec1, &row, Some(Field::Boolean(false))); - test_lte(&mut uint1, &mut null, &row, Some(Field::Null)); - } - - // gt: Int - if i_num1 > u_num1 as i64 && i_num1 > i_num2 && i_num1 as f64 > f_num1 && Decimal::from(i_num1) > d_num1.0 { - test_gt(&mut int1, &mut uint1, &row, None); - test_gt(&mut int1, &mut int2, &row, None); - test_gt(&mut int1, &mut float1, &row, None); - test_gt(&mut int1, &mut dec1, &row, None); - test_gt(&mut int1, &mut null, &row, Some(Field::Null)); - - test_gte(&mut int1, &mut uint1, &row, None); - test_gte(&mut int1, &mut int2, &row, None); - test_gte(&mut int1, &mut float1, &row, None); - test_gte(&mut int1, &mut dec1, &row, None); - test_gte(&mut int1, &mut null, &row, Some(Field::Null)); - - test_lt(&mut int1, &mut uint1, &row, Some(Field::Boolean(false))); - test_lt(&mut int1, &mut int2, &row, Some(Field::Boolean(false))); - test_lt(&mut int1, &mut float1, &row, Some(Field::Boolean(false))); - test_lt(&mut int1, &mut dec1, &row, Some(Field::Boolean(false))); - test_lt(&mut int1, &mut null, &row, Some(Field::Null)); - - test_lte(&mut int1, &mut uint1, &row, Some(Field::Boolean(false))); - test_lte(&mut int1, &mut int2, &row, Some(Field::Boolean(false))); - test_lte(&mut int1, &mut float1, &row, Some(Field::Boolean(false))); - test_lte(&mut int1, &mut dec1, &row, Some(Field::Boolean(false))); - test_lte(&mut int1, &mut null, &row, Some(Field::Null)); - } - - // gt: Float - let d_val = Decimal::from_f64(f_num1); - if d_val.is_some() && f_num1 > u_num1 as f64 && f_num1 > i_num1 as f64 && f_num1 > f_num2 && d_val.unwrap() > d_num1.0 { - test_gt(&mut float1, &mut uint1, &row, None); - test_gt(&mut float1, &mut int1, &row, None); - test_gt(&mut float1, &mut float2, &row, None); - test_gt(&mut float1, &mut dec1, &row, None); - test_gt(&mut float1, &mut null, &row, Some(Field::Null)); - - test_gte(&mut float1, &mut uint1, &row, None); - test_gte(&mut float1, &mut int1, &row, None); - test_gte(&mut float1, &mut float2, &row, None); - test_gte(&mut float1, &mut dec1, &row, None); - test_gte(&mut float1, &mut null, &row, Some(Field::Null)); - - test_lt(&mut float1, &mut uint1, &row, Some(Field::Boolean(false))); - test_lt(&mut float1, &mut int1, &row, Some(Field::Boolean(false))); - test_lt(&mut float1, &mut float2, &row, Some(Field::Boolean(false))); - test_lt(&mut float1, &mut dec1, &row, Some(Field::Boolean(false))); - test_lt(&mut float1, &mut null, &row, Some(Field::Null)); - - test_lte(&mut float1, &mut uint1, &row, Some(Field::Boolean(false))); - test_lte(&mut float1, &mut int1, &row, Some(Field::Boolean(false))); - test_lte(&mut float1, &mut float2, &row, Some(Field::Boolean(false))); - test_lte(&mut float1, &mut dec1, &row, Some(Field::Boolean(false))); - test_lte(&mut float1, &mut null, &row, Some(Field::Null)); - } - - // gt: Decimal - let d_val = Decimal::from_f64(f_num1); - if d_val.is_some() && d_num1.0 > Decimal::from(u_num1) && d_num1.0 > Decimal::from(i_num1) && d_num1.0 > d_val.unwrap() && d_num1.0 > d_num2.0 { - test_gt(&mut dec1, &mut uint1, &row, None); - test_gt(&mut dec1, &mut int1, &row, None); - test_gt(&mut dec1, &mut float1, &row, None); - test_gt(&mut dec1, &mut dec2, &row, None); - test_gt(&mut dec1, &mut null, &row, Some(Field::Null)); - - test_gte(&mut dec1, &mut uint1, &row, None); - test_gte(&mut dec1, &mut int1, &row, None); - test_gte(&mut dec1, &mut float1, &row, None); - test_gte(&mut dec1, &mut dec2, &row, None); - test_gte(&mut dec1, &mut null, &row, Some(Field::Null)); - - test_lt(&mut dec1, &mut uint1, &row, Some(Field::Boolean(false))); - test_lt(&mut dec1, &mut int1, &row, Some(Field::Boolean(false))); - test_lt(&mut dec1, &mut float1, &row, Some(Field::Boolean(false))); - test_lt(&mut dec1, &mut dec2, &row, Some(Field::Boolean(false))); - test_lt(&mut dec1, &mut null, &row, Some(Field::Null)); - - test_lte(&mut dec1, &mut uint1, &row, Some(Field::Boolean(false))); - test_lte(&mut dec1, &mut int1, &row, Some(Field::Boolean(false))); - test_lt(&mut dec1, &mut float1, &row, Some(Field::Boolean(false))); - test_lte(&mut dec1, &mut dec2, &row, Some(Field::Boolean(false))); - test_lte(&mut dec1, &mut null, &row, Some(Field::Null)); - } - - // lt: UInt - if u_num1 < u_num2 && (u_num1 as i64) < i_num1 && (u_num1 as f64) < f_num1 && Decimal::from(u_num1) < d_num1.0 { - test_lt(&mut uint1, &mut uint2, &row, None); - test_lt(&mut uint1, &mut int1, &row, None); - test_lt(&mut uint1, &mut float1, &row, None); - test_lt(&mut uint1, &mut dec1, &row, None); - test_lt(&mut uint1, &mut null, &row, Some(Field::Null)); - - test_lte(&mut uint1, &mut uint2, &row, None); - test_lte(&mut uint1, &mut int1, &row, None); - test_lte(&mut uint1, &mut float1, &row, None); - test_lte(&mut uint1, &mut dec1, &row, None); - test_lte(&mut uint1, &mut null, &row, Some(Field::Null)); - - test_gt(&mut uint1, &mut uint2, &row, Some(Field::Boolean(false))); - test_gt(&mut uint1, &mut int1, &row, Some(Field::Boolean(false))); - test_gt(&mut uint1, &mut float1, &row, Some(Field::Boolean(false))); - test_gt(&mut uint1, &mut dec1, &row, Some(Field::Boolean(false))); - test_gt(&mut uint1, &mut null, &row, Some(Field::Null)); - - test_gte(&mut uint1, &mut uint2, &row, Some(Field::Boolean(false))); - test_gte(&mut uint1, &mut int1, &row, Some(Field::Boolean(false))); - test_gte(&mut uint1, &mut float1, &row, Some(Field::Boolean(false))); - test_gte(&mut uint1, &mut dec1, &row, Some(Field::Boolean(false))); - test_gte(&mut uint1, &mut null, &row, Some(Field::Null)); - } - - // gt: Int - if i_num1 < (u_num1 as i64) && i_num1 < i_num2 && (i_num1 as f64) < f_num1 && Decimal::from(i_num1) < d_num1.0 { - test_lt(&mut int1, &mut uint1, &row, None); - test_lt(&mut int1, &mut int2, &row, None); - test_lt(&mut int1, &mut float1, &row, None); - test_lt(&mut int1, &mut dec1, &row, None); - test_lt(&mut int1, &mut null, &row, Some(Field::Null)); - - test_lte(&mut int1, &mut uint1, &row, None); - test_lte(&mut int1, &mut int2, &row, None); - test_lte(&mut int1, &mut float1, &row, None); - test_lte(&mut int1, &mut dec1, &row, None); - test_lte(&mut int1, &mut null, &row, Some(Field::Null)); - - test_gt(&mut int1, &mut uint1, &row, Some(Field::Boolean(false))); - test_gt(&mut int1, &mut int2, &row, Some(Field::Boolean(false))); - test_gt(&mut int1, &mut float1, &row, Some(Field::Boolean(false))); - test_gt(&mut int1, &mut dec1, &row, Some(Field::Boolean(false))); - test_gt(&mut int1, &mut null, &row, Some(Field::Null)); - - test_gte(&mut int1, &mut uint1, &row, Some(Field::Boolean(false))); - test_gte(&mut int1, &mut int2, &row, Some(Field::Boolean(false))); - test_gte(&mut int1, &mut float1, &row, Some(Field::Boolean(false))); - test_gte(&mut int1, &mut dec1, &row, Some(Field::Boolean(false))); - test_gte(&mut int1, &mut null, &row, Some(Field::Null)); - } - - // gt: Float - let d_val = Decimal::from_f64(f_num1); - if d_val.is_some() && f_num1 < u_num1 as f64 && f_num1 < i_num1 as f64 && f_num1 < f_num2 && d_val.unwrap() < d_num1.0 { - test_lt(&mut float1, &mut uint1, &row, None); - test_lt(&mut float1, &mut int1, &row, None); - test_lt(&mut float1, &mut float2, &row, None); - test_lt(&mut float1, &mut dec1, &row, None); - test_lt(&mut float1, &mut null, &row, Some(Field::Null)); - - test_lte(&mut float1, &mut uint1, &row, None); - test_lte(&mut float1, &mut int1, &row, None); - test_lte(&mut float1, &mut float2, &row, None); - test_lte(&mut float1, &mut dec1, &row, None); - test_lte(&mut float1, &mut null, &row, Some(Field::Null)); - - test_gt(&mut float1, &mut uint1, &row, Some(Field::Boolean(false))); - test_gt(&mut float1, &mut int1, &row, Some(Field::Boolean(false))); - test_gt(&mut float1, &mut float2, &row, Some(Field::Boolean(false))); - test_gt(&mut float1, &mut dec1, &row, Some(Field::Boolean(false))); - test_gt(&mut float1, &mut null, &row, Some(Field::Null)); - - test_gte(&mut float1, &mut uint1, &row, Some(Field::Boolean(false))); - test_gte(&mut float1, &mut int1, &row, Some(Field::Boolean(false))); - test_gte(&mut float1, &mut float2, &row, Some(Field::Boolean(false))); - test_gte(&mut float1, &mut dec1, &row, Some(Field::Boolean(false))); - test_gte(&mut float1, &mut null, &row, Some(Field::Null)); - } - - // gt: Decimal - let d_val = Decimal::from_f64(f_num1); - if d_val.is_some() && d_num1.0 < Decimal::from(u_num1) && d_num1.0 < Decimal::from(i_num1) && d_num1.0 < d_val.unwrap() && d_num1.0 < d_num2.0 { - test_lt(&mut dec1, &mut uint1, &row, None); - test_lt(&mut dec1, &mut int1, &row, None); - test_lt(&mut dec1, &mut float1, &row, None); - test_lt(&mut dec1, &mut dec2, &row, None); - test_lt(&mut dec1, &mut null, &row, Some(Field::Null)); - - test_lte(&mut dec1, &mut uint1, &row, None); - test_lte(&mut dec1, &mut int1, &row, None); - test_lte(&mut dec1, &mut float1, &row, None); - test_lte(&mut dec1, &mut dec2, &row, None); - test_lte(&mut dec1, &mut null, &row, Some(Field::Null)); - - test_gt(&mut dec1, &mut uint1, &row, Some(Field::Boolean(false))); - test_gt(&mut dec1, &mut int1, &row, Some(Field::Boolean(false))); - test_gt(&mut dec1, &mut float1, &row, Some(Field::Boolean(false))); - test_gt(&mut dec1, &mut dec2, &row, Some(Field::Boolean(false))); - test_gt(&mut dec1, &mut null, &row, Some(Field::Null)); - - test_gte(&mut dec1, &mut uint1, &row, Some(Field::Boolean(false))); - test_gte(&mut dec1, &mut int1, &row, Some(Field::Boolean(false))); - test_gte(&mut dec1, &mut float1, &row, Some(Field::Boolean(false))); - test_gte(&mut dec1, &mut dec2, &row, Some(Field::Boolean(false))); - test_gte(&mut dec1, &mut null, &row, Some(Field::Null)); - } - }); -} - -fn test_eq(exp1: &mut Expression, exp2: &mut Expression, row: &Record, result: Option) { - match result { - None => { - assert!(matches!( - evaluate_eq(&Schema::default(), exp1, exp2, row), - Ok(Field::Boolean(true)) - )); - } - Some(_val) => { - assert!(matches!( - evaluate_eq(&Schema::default(), exp1, exp2, row), - Ok(_val) - )); - } - } -} - -fn test_ne(exp1: &mut Expression, exp2: &mut Expression, row: &Record, result: Option) { - match result { - None => { - assert!(matches!( - evaluate_ne(&Schema::default(), exp1, exp2, row), - Ok(Field::Boolean(true)) - )); - } - Some(_val) => { - assert!(matches!( - evaluate_ne(&Schema::default(), exp1, exp2, row), - Ok(_val) - )); - } - } -} - -fn test_gt(exp1: &mut Expression, exp2: &mut Expression, row: &Record, result: Option) { - match result { - None => { - assert!(matches!( - evaluate_gt(&Schema::default(), exp1, exp2, row), - Ok(Field::Boolean(true)) - )); - } - Some(_val) => { - assert!(matches!( - evaluate_gt(&Schema::default(), exp1, exp2, row), - Ok(_val) - )); - } - } -} - -fn test_lt(exp1: &mut Expression, exp2: &mut Expression, row: &Record, result: Option) { - match result { - None => { - assert!(matches!( - evaluate_lt(&Schema::default(), exp1, exp2, row), - Ok(Field::Boolean(true)) - )); - } - Some(_val) => { - assert!(matches!( - evaluate_lt(&Schema::default(), exp1, exp2, row), - Ok(_val) - )); - } - } -} - -fn test_gte(exp1: &mut Expression, exp2: &mut Expression, row: &Record, result: Option) { - match result { - None => { - assert!(matches!( - evaluate_gte(&Schema::default(), exp1, exp2, row), - Ok(Field::Boolean(true)) - )); - } - Some(_val) => { - assert!(matches!( - evaluate_gte(&Schema::default(), exp1, exp2, row), - Ok(_val) - )); - } - } -} - -fn test_lte(exp1: &mut Expression, exp2: &mut Expression, row: &Record, result: Option) { - match result { - None => { - assert!(matches!( - evaluate_lte(&Schema::default(), exp1, exp2, row), - Ok(Field::Boolean(true)) - )); - } - Some(_val) => { - assert!(matches!( - evaluate_lte(&Schema::default(), exp1, exp2, row), - Ok(_val) - )); - } - } -} diff --git a/dozer-sql/expression/src/conditional.rs b/dozer-sql/expression/src/conditional.rs deleted file mode 100644 index 4c36755f15..0000000000 --- a/dozer-sql/expression/src/conditional.rs +++ /dev/null @@ -1,416 +0,0 @@ -use crate::cast::CastOperatorType; -use crate::error::Error; -use crate::execution::{Expression, ExpressionType}; -use dozer_types::types::Record; -use dozer_types::types::{Field, FieldType, Schema}; -use std::fmt::{Display, Formatter}; - -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Hash)] -pub enum ConditionalExpressionType { - Coalesce, - NullIf, - Least, -} - -pub(crate) fn get_conditional_expr_type( - function: &ConditionalExpressionType, - args: &[Expression], - schema: &Schema, -) -> Result { - match function { - ConditionalExpressionType::Coalesce => validate_coalesce(args, schema), - ConditionalExpressionType::Least => validate_least(args, schema), - ConditionalExpressionType::NullIf => todo!(), - } -} - -impl ConditionalExpressionType { - pub(crate) fn new(name: &str) -> Option { - match name { - "coalesce" => Some(ConditionalExpressionType::Coalesce), - "nullif" => Some(ConditionalExpressionType::NullIf), - _ => None, - } - } - - pub(crate) fn evaluate( - &self, - schema: &Schema, - args: &mut [Expression], - record: &Record, - ) -> Result { - match self { - ConditionalExpressionType::Coalesce => evaluate_coalesce(schema, args, record), - ConditionalExpressionType::Least => evaluate_least(schema, args, record), - ConditionalExpressionType::NullIf => todo!(), - } - } -} - -pub(crate) fn validate_coalesce( - args: &[Expression], - schema: &Schema, -) -> Result { - if args.is_empty() { - return Err(Error::EmptyCoalesceArguments); - } - - let return_types = args - .iter() - .map(|expr| expr.get_type(schema).unwrap().return_type) - .collect::>(); - let return_type = return_types[0]; - - Ok(ExpressionType::new( - return_type, - false, - dozer_types::types::SourceDefinition::Dynamic, - false, - )) -} - -pub(crate) fn evaluate_coalesce( - schema: &Schema, - args: &mut [Expression], - record: &Record, -) -> Result { - // The COALESCE function returns the first of its arguments that is not null. - for expr in args { - let field = expr.evaluate(record, schema)?; - if field != Field::Null { - return Ok(field); - } - } - // Null is returned only if all arguments are null. - Ok(Field::Null) -} - -pub(crate) fn validate_least( - args: &[Expression], - schema: &Schema, -) -> Result { - if args.is_empty() { - return Err(Error::EmptyLeastArguments); - } - - let return_types = args - .iter() - .map(|expr| Ok(expr.get_type(schema)?.return_type)) - .collect::, Error>>()?; - let return_type = return_types[0]; - - Ok(ExpressionType::new( - return_type, - false, - dozer_types::types::SourceDefinition::Dynamic, - false, - )) -} - -pub(crate) fn evaluate_least( - schema: &Schema, - args: &mut [Expression], - record: &Record, -) -> Result { - let typ = args[0].get_type(schema)?; - let cast = CastOperatorType(typ.return_type); - args.iter_mut() - .map(|arg| cast.evaluate(schema, arg, record)) - .reduce(|left, right| Ok(left?.min(right?))) - .unwrap() -} - -impl Display for ConditionalExpressionType { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - match self { - ConditionalExpressionType::Coalesce => f.write_str("COALESCE"), - ConditionalExpressionType::Least => f.write_str("LEAST"), - ConditionalExpressionType::NullIf => f.write_str("NULLIF"), - } - } -} - -#[cfg(test)] -mod tests { - use crate::tests::{ArbitraryDateTime, ArbitraryDecimal}; - - use super::*; - - use dozer_types::{ - ordered_float::OrderedFloat, - types::{FieldDefinition, SourceDefinition}, - }; - use proptest::prelude::*; - - #[test] - fn test_coalesce() { - proptest!(ProptestConfig::with_cases(1000), move |( - u_num1: u64, u_num2: u64, i_num1: i64, i_num2: i64, f_num1: f64, f_num2: f64, - d_num1: ArbitraryDecimal, d_num2: ArbitraryDecimal, - s_val1: String, s_val2: String, - dt_val1: ArbitraryDateTime, dt_val2: ArbitraryDateTime)| { - let uint1 = Expression::Literal(Field::UInt(u_num1)); - let uint2 = Expression::Literal(Field::UInt(u_num2)); - let int1 = Expression::Literal(Field::Int(i_num1)); - let int2 = Expression::Literal(Field::Int(i_num2)); - let float1 = Expression::Literal(Field::Float(OrderedFloat(f_num1))); - let float2 = Expression::Literal(Field::Float(OrderedFloat(f_num2))); - let dec1 = Expression::Literal(Field::Decimal(d_num1.0)); - let dec2 = Expression::Literal(Field::Decimal(d_num2.0)); - let str1 = Expression::Literal(Field::String(s_val1.clone())); - let str2 = Expression::Literal(Field::String(s_val2)); - let t1 = Expression::Literal(Field::Timestamp(dt_val1.0)); - let t2 = Expression::Literal(Field::Timestamp(dt_val1.0)); - let dt1 = Expression::Literal(Field::Date(dt_val1.0.date_naive())); - let dt2 = Expression::Literal(Field::Date(dt_val2.0.date_naive())); - let null = Expression::Column{ index: 0usize }; - - // UInt - let typ = FieldType::UInt; - let f = Field::UInt(u_num1); - let row = Record::new(vec![f.clone()]); - - let mut args = vec![null.clone(), uint1.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), uint1.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), uint1.clone(), null.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), uint1, uint2]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f); - - // Int - let typ = FieldType::Int; - let f = Field::Int(i_num1); - let row = Record::new(vec![f.clone()]); - - let mut args = vec![null.clone(), int1.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), int1.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), int1.clone(), null.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), int1, int2]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f); - - // Float - let typ = FieldType::Float; - let f = Field::Float(OrderedFloat(f_num1)); - let row = Record::new(vec![f.clone()]); - - let mut args = vec![null.clone(), float1.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), float1.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), float1.clone(), null.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), float1, float2]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f); - - // Decimal - let typ = FieldType::Decimal; - let f = Field::Decimal(d_num1.0); - let row = Record::new(vec![f.clone()]); - - let mut args = vec![null.clone(), dec1.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), dec1.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), dec1.clone(), null.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), dec1, dec2]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f); - - // String - let typ = FieldType::String; - let f = Field::String(s_val1.clone()); - let row = Record::new(vec![f.clone()]); - - let mut args = vec![null.clone(), str1.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), str1.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), str1.clone(), null.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), str1.clone(), str2.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f); - - // String - let typ = FieldType::String; - let f = Field::String(s_val1); - let row = Record::new(vec![f.clone()]); - - let mut args = vec![null.clone(), str1.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), str1.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), str1.clone(), null.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), str1, str2]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f); - - // Timestamp - let typ = FieldType::Timestamp; - let f = Field::Timestamp(dt_val1.0); - let row = Record::new(vec![f.clone()]); - - let mut args = vec![null.clone(), t1.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), t1.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), t1.clone(), null.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), t1, t2]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f); - - // Date - let typ = FieldType::Date; - let f = Field::Date(dt_val1.0.date_naive()); - let row = Record::new(vec![f.clone()]); - - let mut args = vec![null.clone(), dt1.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), dt1.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), dt1.clone(), null.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null.clone(), dt1, dt2]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f); - - // Null - let typ = FieldType::Date; - let f = Field::Null; - let row = Record::new(vec![f.clone()]); - - let mut args = vec![null.clone()]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f.clone()); - - let mut args = vec![null.clone(), null]; - test_validate_coalesce(&args, typ); - test_evaluate_coalesce(&mut args, &row, typ, f); - }); - } - - fn test_validate_coalesce(args: &[Expression], typ: FieldType) { - let schema = Schema::default() - .field( - FieldDefinition::new(String::from("field"), typ, false, SourceDefinition::Dynamic), - false, - ) - .clone(); - - let result = validate_coalesce(args, &schema).unwrap().return_type; - assert_eq!(result, typ); - } - - fn test_evaluate_coalesce( - args: &mut [Expression], - row: &Record, - typ: FieldType, - _result: Field, - ) { - let schema = Schema::default() - .field( - FieldDefinition::new(String::from("field"), typ, false, SourceDefinition::Dynamic), - false, - ) - .clone(); - - let res = evaluate_coalesce(&schema, args, row).unwrap(); - assert_eq!(res, _result); - } - - fn check_least(fields: Vec, expected: Field) { - let schema = Schema::default(); - let row = Record::new(vec![]); - let mut args: Vec<_> = fields.into_iter().map(Expression::Literal).collect(); - assert_eq!(evaluate_least(&schema, &mut args, &row).unwrap(), expected); - } - - #[test] - fn test_least() { - check_least(vec![Field::Int(41), Field::Int(42)], Field::Int(41)); - check_least( - vec![ - Field::Float(OrderedFloat(4.1)), - Field::Float(OrderedFloat(4.2)), - ], - Field::Float(OrderedFloat(4.1)), - ); - - check_least( - vec![Field::String("BCD".into()), Field::String("ABC".into())], - Field::String("ABC".into()), - ); - } - - #[test] - fn test_least_mixed_types() { - check_least( - vec![ - Field::Float(OrderedFloat(1.2)), - Field::UInt(1), - Field::Int(-2), - Field::String("-3.24".to_owned()), - ], - Field::Float(OrderedFloat(-3.24)), - ) - } -} diff --git a/dozer-sql/expression/src/datetime.rs b/dozer-sql/expression/src/datetime.rs deleted file mode 100644 index ae864da9a9..0000000000 --- a/dozer-sql/expression/src/datetime.rs +++ /dev/null @@ -1,256 +0,0 @@ -use crate::arg_utils::{extract_timestamp, extract_uint, validate_arg_type}; -use crate::error::Error; -use crate::execution::{Expression, ExpressionType}; - -use dozer_types::chrono::{DateTime, Datelike, FixedOffset, Offset, Timelike, Utc}; -use dozer_types::types::Record; -use dozer_types::types::{DozerDuration, Field, FieldType, Schema, TimeUnit}; -use num_traits::ToPrimitive; -use sqlparser::ast::DateTimeField; -use std::fmt::{Display, Formatter}; - -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Hash)] -pub enum DateTimeFunctionType { - Extract { - field: sqlparser::ast::DateTimeField, - }, - Interval { - field: sqlparser::ast::DateTimeField, - }, - Now, -} - -impl Display for DateTimeFunctionType { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - match self { - DateTimeFunctionType::Extract { field } => { - f.write_str(format!("EXTRACT {field}").as_str()) - } - DateTimeFunctionType::Interval { field } => { - f.write_str(format!("INTERVAL {field}").as_str()) - } - DateTimeFunctionType::Now => f.write_str("NOW".to_string().as_str()), - } - } -} - -pub(crate) fn get_datetime_function_type( - function: &DateTimeFunctionType, - arg: &Expression, - schema: &Schema, -) -> Result { - validate_arg_type( - arg, - vec![ - FieldType::Date, - FieldType::Timestamp, - FieldType::Duration, - FieldType::String, - FieldType::Text, - ], - schema, - function, - 0, - )?; - match function { - DateTimeFunctionType::Extract { field: _ } => Ok(ExpressionType::new( - FieldType::Int, - false, - dozer_types::types::SourceDefinition::Dynamic, - false, - )), - DateTimeFunctionType::Interval { field: _ } => Ok(ExpressionType::new( - FieldType::Duration, - false, - dozer_types::types::SourceDefinition::Dynamic, - false, - )), - DateTimeFunctionType::Now => Ok(ExpressionType::new( - FieldType::Timestamp, - false, - dozer_types::types::SourceDefinition::Dynamic, - false, - )), - } -} - -impl DateTimeFunctionType { - pub(crate) fn new(name: &str) -> Option { - match name { - "now" => Some(DateTimeFunctionType::Now), - _ => None, - } - } - - pub(crate) fn evaluate( - &self, - schema: &Schema, - arg: &mut Expression, - record: &Record, - ) -> Result { - match self { - DateTimeFunctionType::Extract { field } => { - evaluate_date_part(schema, field, arg, record) - } - DateTimeFunctionType::Interval { field } => { - evaluate_interval(schema, field, arg, record) - } - DateTimeFunctionType::Now => self.evaluate_now(), - } - } - - pub(crate) fn evaluate_now(&self) -> Result { - Ok(Field::Timestamp(DateTime::::from(Utc::now()))) - } -} - -pub(crate) fn evaluate_date_part( - schema: &Schema, - field: &sqlparser::ast::DateTimeField, - arg: &mut Expression, - record: &Record, -) -> Result { - let value = arg.evaluate(record, schema)?; - - let ts = extract_timestamp(value, DateTimeFunctionType::Extract { field: *field }, 0)?; - - match field { - DateTimeField::Dow => ts.weekday().num_days_from_monday().to_i64(), - DateTimeField::Day => ts.day().to_i64(), - DateTimeField::Month => ts.month().to_i64(), - DateTimeField::Year => ts.year().to_i64(), - DateTimeField::Hour => ts.hour().to_i64(), - DateTimeField::Minute => ts.minute().to_i64(), - DateTimeField::Second => ts.second().to_i64(), - DateTimeField::Millisecond | DateTimeField::Milliseconds => ts.timestamp_millis().to_i64(), - DateTimeField::Microsecond | DateTimeField::Microseconds => ts.timestamp_micros().to_i64(), - DateTimeField::Nanoseconds | DateTimeField::Nanosecond => ts - .timestamp_nanos_opt() - .expect("value can not be represented in a timestamp with nanosecond precision.") - .to_i64(), - DateTimeField::Quarter => ts.month0().to_i64().map(|m| m / 3 + 1), - DateTimeField::Epoch => ts.timestamp().to_i64(), - DateTimeField::Week => ts.iso_week().week().to_i64(), - DateTimeField::Century => ts.year().to_i64().map(|y| (y as f64 / 100.0).ceil() as i64), - DateTimeField::Decade => ts.year().to_i64().map(|y| (y as f64 / 10.0).ceil() as i64), - DateTimeField::Doy => ts.ordinal().to_i64(), - DateTimeField::Timezone => ts.offset().fix().local_minus_utc().to_i64(), - DateTimeField::Isodow - | DateTimeField::Isoyear - | DateTimeField::Julian - | DateTimeField::Millenium - | DateTimeField::Millennium - | DateTimeField::TimezoneHour - | DateTimeField::TimezoneMinute - | DateTimeField::Date - | DateTimeField::NoDateTime => None, - } - .ok_or(Error::UnsupportedExtract(*field)) - .map(Field::Int) -} - -pub(crate) fn evaluate_interval( - schema: &Schema, - field: &sqlparser::ast::DateTimeField, - arg: &mut Expression, - record: &Record, -) -> Result { - let value = arg.evaluate(record, schema)?; - let dur = extract_uint(value, DateTimeFunctionType::Interval { field: *field }, 0)?; - - match field { - DateTimeField::Second => Ok(Field::Duration(DozerDuration( - std::time::Duration::from_secs(dur), - TimeUnit::Seconds, - ))), - DateTimeField::Millisecond | DateTimeField::Milliseconds => { - Ok(Field::Duration(DozerDuration( - std::time::Duration::from_millis(dur), - TimeUnit::Milliseconds, - ))) - } - DateTimeField::Microsecond | DateTimeField::Microseconds => { - Ok(Field::Duration(DozerDuration( - std::time::Duration::from_micros(dur), - TimeUnit::Microseconds, - ))) - } - DateTimeField::Nanoseconds | DateTimeField::Nanosecond => Ok(Field::Duration( - DozerDuration(std::time::Duration::from_nanos(dur), TimeUnit::Nanoseconds), - )), - DateTimeField::Day => Ok(Field::Duration(DozerDuration( - std::time::Duration::from_secs(dur * 24 * 60 * 60), - TimeUnit::Nanoseconds, - ))), - DateTimeField::Isodow - | DateTimeField::Timezone - | DateTimeField::Dow - | DateTimeField::Isoyear - | DateTimeField::Julian - | DateTimeField::Millenium - | DateTimeField::Millennium - | DateTimeField::TimezoneHour - | DateTimeField::TimezoneMinute - | DateTimeField::Date - | DateTimeField::NoDateTime - | DateTimeField::Month - | DateTimeField::Year - | DateTimeField::Hour - | DateTimeField::Minute - | DateTimeField::Quarter - | DateTimeField::Epoch - | DateTimeField::Week - | DateTimeField::Century - | DateTimeField::Decade - | DateTimeField::Doy => Err(Error::UnsupportedInterval(*field)), - } -} - -#[cfg(test)] -mod tests { - use crate::tests::ArbitraryDateTime; - - use super::*; - - use proptest::prelude::*; - - #[test] - fn test_time() { - proptest!( - ProptestConfig::with_cases(1000), - move |(datetime: ArbitraryDateTime)| { - test_date_parts(datetime) - }); - } - - fn test_date_parts(datetime: ArbitraryDateTime) { - let row = Record::new(vec![]); - - let date_parts = vec![ - ( - DateTimeField::Dow, - datetime - .0 - .weekday() - .num_days_from_monday() - .to_i64() - .unwrap(), - ), - (DateTimeField::Year, datetime.0.year().to_i64().unwrap()), - (DateTimeField::Month, datetime.0.month().to_i64().unwrap()), - (DateTimeField::Hour, 0), - (DateTimeField::Second, 0), - ( - DateTimeField::Quarter, - datetime.0.month0().to_i64().map(|m| m / 3 + 1).unwrap(), - ), - ]; - - let mut v = Expression::Literal(Field::Date(datetime.0.date_naive())); - - for (part, value) in date_parts { - let result = evaluate_date_part(&Schema::default(), &part, &mut v, &row).unwrap(); - assert_eq!(result, Field::Int(value)); - } - } -} diff --git a/dozer-sql/expression/src/error.rs b/dozer-sql/expression/src/error.rs deleted file mode 100644 index 7078229713..0000000000 --- a/dozer-sql/expression/src/error.rs +++ /dev/null @@ -1,135 +0,0 @@ -use std::ops::Range; - -use dozer_types::{ - thiserror::{self, Error}, - types::{Field, FieldType}, -}; -use sqlparser::ast::{ - BinaryOperator, DataType, DateTimeField, Expr, FunctionArg, Ident, UnaryOperator, -}; - -use crate::{aggregate::AggregateFunctionType, operator::BinaryOperatorType}; - -#[derive(Debug, Error)] -pub enum Error { - #[error("Unsupported SQL expression: {0:?}")] - UnsupportedExpression(Expr), - #[error("Unsupported SQL function arg: {0:?}")] - UnsupportedFunctionArg(FunctionArg), - #[error("Invalid ident: {}", .0.iter().map(|ident| ident.value.as_str()).collect::>().join("."))] - InvalidIdent(Vec), - #[error("Unknown function: {0}")] - UnknownFunction(String), - #[error("Missing leading field in interval")] - MissingLeadingFieldInInterval, - #[error("Unsupported SQL unary operator: {0:?}")] - UnsupportedUnaryOperator(UnaryOperator), - #[error("Unsupported SQL binary operator: {0:?}")] - UnsupportedBinaryOperator(BinaryOperator), - #[error("Not a number: {0}")] - NotANumber(String), - #[error("Unsupported data type: {0}")] - UnsupportedDataType(DataType), - - #[error("Aggregate Function {0:?} should not be executed at this point")] - UnexpectedAggregationExecution(AggregateFunctionType), - #[error("literal expression cannot be null")] - LiteralExpressionIsNull, - #[error("cannot apply NOT to {0:?}")] - CannotApplyNotTo(FieldType), - #[error("cannot apply {operator:?} to {left_field_type:?} and {right_field_type:?}")] - CannotApplyBinaryOperator { - operator: BinaryOperatorType, - left_field_type: FieldType, - right_field_type: FieldType, - }, - #[error("expected {expected:?} arguments for function {function_name}, got {actual}")] - InvalidNumberOfArguments { - function_name: String, - expected: Range, - actual: usize, - }, - #[error("Empty coalesce arguments")] - EmptyCoalesceArguments, - #[error("Empty LEAST arguments")] - EmptyLeastArguments, - #[error( - "Invalid argument type for function {function_name}: type: {actual}, expected types: {expected:?}, index: {argument_index}" - )] - InvalidFunctionArgumentType { - function_name: String, - argument_index: usize, - expected: Vec, - actual: FieldType, - }, - #[error("Invalid cast: from: {from}, to: {to}")] - InvalidCast { from: Field, to: FieldType }, - #[error("Invalid argument for function {function_name}(): argument: {argument}, index: {argument_index}")] - InvalidFunctionArgument { - function_name: String, - argument_index: usize, - argument: Field, - }, - - #[error("Invalid distance algorithm: {0}")] - InvalidDistanceAlgorithm(String), - #[error("Failed to calculate vincenty distance: {0}")] - FailedToCalculateVincentyDistance( - #[from] dozer_types::geo::vincenty_distance::FailedToConvergeError, - ), - - #[error("Invalid like escape: {0}")] - InvalidLikeEscape(#[from] like::InvalidEscapeError), - #[error("Invalid like pattern: {0}")] - InvalidLikePattern(#[from] like::InvalidPatternError), - - #[error("Unsupported extract: {0}")] - UnsupportedExtract(DateTimeField), - #[error("Unsupported interval: {0}")] - UnsupportedInterval(DateTimeField), - - #[error("Invalid json path: {0}")] - InvalidJsonPath(String), - - #[cfg(feature = "python")] - #[error("Python UDF error: {0}")] - PythonUdf(#[from] crate::python_udf::Error), - - #[cfg(feature = "onnx")] - #[error("ONNX UDF error: {0}")] - Onnx(#[from] crate::onnx::error::Error), - #[cfg(not(feature = "onnx"))] - #[error("ONNX UDF is not enabled")] - OnnxNotEnabled, - - #[error("Javascript is not enabled")] - JavaScriptNotEnabled, - - #[cfg(feature = "javascript")] - #[error("JavaScript UDF error: {0}")] - JavaScript(#[from] crate::javascript::Error), - - // Legacy error types. - #[error("Sql error: {0}")] - SqlError(#[source] OperationError), - #[error("Invalid types on {0} and {1} for {2} operand")] - InvalidTypeComparison(Field, Field, String), - #[error("Unable to cast {0} to {1}")] - UnableToCast(String, String), - #[error("Invalid types on {0} for {1} operand")] - InvalidType(Field, String), -} - -#[derive(Error, Debug)] -pub enum OperationError { - #[error("SQL Error: Addition operation cannot be done due to overflow.")] - AdditionOverflow, - #[error("SQL Error: Subtraction operation cannot be done due to overflow.")] - SubtractionOverflow, - #[error("SQL Error: Multiplication operation cannot be done due to overflow.")] - MultiplicationOverflow, - #[error("SQL Error: Division operation cannot be done.")] - DivisionByZeroOrOverflow, - #[error("SQL Error: Modulo operation cannot be done.")] - ModuloByZeroOrOverflow, -} diff --git a/dozer-sql/expression/src/execution.rs b/dozer-sql/expression/src/execution.rs deleted file mode 100644 index 1c2891b234..0000000000 --- a/dozer-sql/expression/src/execution.rs +++ /dev/null @@ -1,1139 +0,0 @@ -use crate::arg_utils::{validate_one_argument, validate_two_arguments}; -use crate::case::evaluate_case; -use crate::conditional::{get_conditional_expr_type, ConditionalExpressionType}; -use crate::datetime::{get_datetime_function_type, DateTimeFunctionType}; -use crate::error::Error; -use crate::geo::common::{get_geo_function_type, GeoFunctionType}; -use crate::is_null::{evaluate_is_not_null, evaluate_is_null}; -use crate::json_functions::JsonFunctionType; -use crate::operator::{BinaryOperatorType, UnaryOperatorType}; -use crate::scalar::common::{get_scalar_function_type, ScalarFunctionType}; -use crate::scalar::string::{evaluate_trim, validate_trim, TrimType}; -use std::iter::zip; - -use super::aggregate::AggregateFunctionType; -use super::cast::CastOperatorType; -use super::in_list::evaluate_in_list; -use super::scalar::string::{evaluate_like, get_like_operator_type}; -use dozer_types::types::Record; -use dozer_types::types::{Field, FieldType, Schema, SourceDefinition}; - -#[derive(Clone, Debug, PartialEq)] -pub enum Expression { - Column { - index: usize, - }, - Literal(Field), - UnaryOperator { - operator: UnaryOperatorType, - arg: Box, - }, - BinaryOperator { - left: Box, - operator: BinaryOperatorType, - right: Box, - }, - ScalarFunction { - fun: ScalarFunctionType, - args: Vec, - }, - GeoFunction { - fun: GeoFunctionType, - args: Vec, - }, - ConditionalExpression { - fun: ConditionalExpressionType, - args: Vec, - }, - DateTimeFunction { - fun: DateTimeFunctionType, - arg: Box, - }, - AggregateFunction { - fun: AggregateFunctionType, - args: Vec, - }, - Cast { - arg: Box, - typ: CastOperatorType, - }, - Trim { - arg: Box, - what: Option>, - typ: Option, - }, - Like { - arg: Box, - pattern: Box, - escape: Option, - }, - InList { - expr: Box, - list: Vec, - negated: bool, - }, - Now { - fun: DateTimeFunctionType, - }, - Json { - fun: JsonFunctionType, - args: Vec, - }, - Case { - operand: Option>, - conditions: Vec, - results: Vec, - else_result: Option>, - }, - IsNull { - arg: Box, - }, - IsNotNull { - arg: Box, - }, - #[cfg(feature = "python")] - PythonUDF { - name: String, - args: Vec, - return_type: FieldType, - }, - #[cfg(feature = "onnx")] - OnnxUDF { - name: String, - session: crate::onnx::DozerSession, - args: Vec, - }, - #[cfg(feature = "javascript")] - JavaScriptUdf(crate::javascript::Udf), -} - -impl Expression { - pub fn to_string(&self, schema: &Schema) -> String { - match &self { - Expression::Column { index } => schema.fields[*index].name.clone(), - Expression::Literal(value) => format!("{}", value), - Expression::UnaryOperator { operator, arg } => { - operator.to_string() + arg.to_string(schema).as_str() - } - Expression::BinaryOperator { - left, - operator, - right, - } => { - left.to_string(schema) - + operator.to_string().as_str() - + right.to_string(schema).as_str() - } - Expression::ScalarFunction { fun, args } => { - fun.to_string() - + "(" - + args - .iter() - .map(|e| e.to_string(schema)) - .collect::>() - .join(",") - .as_str() - + ")" - } - Expression::ConditionalExpression { fun, args } => { - fun.to_string() - + "(" - + args - .iter() - .map(|e| e.to_string(schema)) - .collect::>() - .join(",") - .as_str() - + ")" - } - Expression::AggregateFunction { fun, args } => { - fun.to_string() - + "(" - + args - .iter() - .map(|e| e.to_string(schema)) - .collect::>() - .join(",") - .as_str() - + ")" - } - #[cfg(feature = "python")] - Expression::PythonUDF { name, args, .. } => { - name.to_string() - + "(" - + args - .iter() - .map(|expr| expr.to_string(schema)) - .collect::>() - .join(",") - .as_str() - + ")" - } - #[cfg(feature = "onnx")] - Expression::OnnxUDF { name, args, .. } => { - name.to_string() - + "(" - + args - .iter() - .map(|expr| expr.to_string(schema)) - .collect::>() - .join(",") - .as_str() - + ")" - } - Expression::Cast { arg, typ } => { - "CAST(".to_string() - + arg.to_string(schema).as_str() - + " AS " - + typ.to_string().as_str() - + ")" - } - Expression::Case { - operand, - conditions, - results, - else_result, - } => { - let mut op_str = String::new(); - if let Some(op) = operand { - op_str += " "; - op_str += op.to_string(schema).as_str(); - } - let mut when_then_str = String::new(); - let iter = zip(conditions, results); - for (cond, res) in iter { - when_then_str += " WHEN "; - when_then_str += cond.to_string(schema).as_str(); - when_then_str += " THEN "; - when_then_str += res.to_string(schema).as_str(); - } - let mut else_str = String::new(); - if let Some(else_res) = else_result { - else_str += " ELSE "; - else_str += else_res.to_string(schema).as_str(); - } - - "CASE".to_string() - + op_str.as_str() - + when_then_str.as_str() - + else_str.as_str() - + " END" - } - Expression::Trim { typ, what, arg } => { - "TRIM(".to_string() - + if let Some(t) = typ { - t.to_string() - } else { - "".to_string() - } - .as_str() - + if let Some(w) = what { - w.to_string(schema) + " FROM " - } else { - "".to_string() - } - .as_str() - + arg.to_string(schema).as_str() - + ")" - } - Expression::Like { - arg, - pattern, - escape: _, - } => arg.to_string(schema) + " LIKE " + pattern.to_string(schema).as_str(), - Expression::InList { - expr, - list, - negated, - } => { - expr.to_string(schema) - + if *negated { " NOT" } else { "" } - + " IN (" - + list - .iter() - .map(|e| e.to_string(schema)) - .collect::>() - .join(",") - .as_str() - + ")" - } - Expression::GeoFunction { fun, args } => { - fun.to_string() - + "(" - + args - .iter() - .map(|e| e.to_string(schema)) - .collect::>() - .join(",") - .as_str() - + ")" - } - Expression::DateTimeFunction { fun, arg } => { - fun.to_string() + "(" + arg.to_string(schema).as_str() + ")" - } - Expression::Now { fun } => fun.to_string() + "()", - Expression::Json { fun, args } => { - fun.to_string() - + "(" - + args - .iter() - .map(|e| e.to_string(schema)) - .collect::>() - .join(",") - .as_str() - + ")" - } - #[cfg(feature = "javascript")] - Expression::JavaScriptUdf(udf) => udf.to_string(schema), - Expression::IsNull { arg } => arg.to_string(schema) + " IS NULL ", - Expression::IsNotNull { arg } => arg.to_string(schema) + " IS NOT NULL ", - } - } -} - -pub struct ExpressionType { - pub return_type: FieldType, - pub nullable: bool, - pub source: SourceDefinition, - pub is_primary_key: bool, -} - -impl ExpressionType { - pub fn new( - return_type: FieldType, - nullable: bool, - source: SourceDefinition, - is_primary_key: bool, - ) -> Self { - Self { - return_type, - nullable, - source, - is_primary_key, - } - } -} - -impl Expression { - pub fn evaluate(&mut self, record: &Record, schema: &Schema) -> Result { - match self { - Expression::Literal(field) => Ok(field.clone()), - Expression::Column { index } => Ok(record.values[*index].clone()), - Expression::BinaryOperator { - left, - operator, - right, - } => operator.evaluate(schema, left, right, record), - Expression::ScalarFunction { fun, args } => fun.evaluate(schema, args, record), - - #[cfg(feature = "python")] - Expression::PythonUDF { - name, - args, - return_type, - .. - } => { - use crate::python_udf::evaluate_py_udf; - evaluate_py_udf(schema, name, args, return_type, record) - } - #[cfg(feature = "onnx")] - Expression::OnnxUDF { - name: _name, - session, - args, - .. - } => { - use std::borrow::Borrow; - crate::onnx::udf::evaluate_onnx_udf(schema, session.0.borrow(), args, record) - } - - Expression::UnaryOperator { operator, arg } => operator.evaluate(schema, arg, record), - Expression::AggregateFunction { fun, args: _ } => { - Err(Error::UnexpectedAggregationExecution(fun.clone())) - } - Expression::Trim { typ, what, arg } => evaluate_trim(schema, arg, what, typ, record), - Expression::Like { - arg, - pattern, - escape, - } => evaluate_like(schema, arg, pattern, *escape, record), - Expression::InList { - expr, - list, - negated, - } => evaluate_in_list(schema, expr, list, *negated, record), - Expression::Cast { arg, typ } => typ.evaluate(schema, arg, record), - Expression::GeoFunction { fun, args } => fun.evaluate(schema, args, record), - Expression::ConditionalExpression { fun, args } => fun.evaluate(schema, args, record), - Expression::DateTimeFunction { fun, arg } => fun.evaluate(schema, arg, record), - Expression::Now { fun } => fun.evaluate_now(), - Expression::Json { fun, args } => fun.evaluate(schema, args, record), - Expression::Case { - operand, - conditions, - results, - else_result, - } => evaluate_case(schema, operand, conditions, results, else_result, record), - Expression::IsNull { arg } => evaluate_is_null(schema, arg, record), - Expression::IsNotNull { arg } => evaluate_is_not_null(schema, arg, record), - #[cfg(feature = "javascript")] - Expression::JavaScriptUdf(udf) => udf.evaluate(record, schema), - } - } - - pub fn get_type(&self, schema: &Schema) -> Result { - match self { - Expression::Literal(field) => { - let field_type = field.ty(); - match field_type { - Some(f) => Ok(ExpressionType::new( - f, - false, - SourceDefinition::Dynamic, - false, - )), - None => Err(Error::LiteralExpressionIsNull), - } - } - Expression::Column { index } => { - let t = schema.fields.get(*index).unwrap(); - - Ok(ExpressionType::new( - t.typ, - t.nullable, - t.source.clone(), - schema.primary_index.contains(index), - )) - } - Expression::UnaryOperator { operator, arg } => { - get_unary_operator_type(operator, arg, schema) - } - Expression::BinaryOperator { - left, - operator, - right, - } => get_binary_operator_type(left, operator, right, schema), - Expression::ScalarFunction { fun, args } => get_scalar_function_type(fun, args, schema), - Expression::ConditionalExpression { fun, args } => { - get_conditional_expr_type(fun, args, schema) - } - Expression::AggregateFunction { fun, args } => { - get_aggregate_function_type(fun, args, schema) - } - Expression::Trim { - what: _, - typ: _, - arg, - } => validate_trim(arg, schema), - Expression::Like { - arg, - pattern, - escape: _, - } => get_like_operator_type(arg, pattern, schema), - Expression::InList { - expr: _, - list: _, - negated: _, - } => Ok(ExpressionType::new( - FieldType::Boolean, - false, - SourceDefinition::Dynamic, - false, - )), - Expression::Cast { arg, typ } => typ.get_return_type(schema, arg), - Expression::GeoFunction { fun, args } => get_geo_function_type(fun, args, schema), - Expression::DateTimeFunction { fun, arg } => { - get_datetime_function_type(fun, arg, schema) - } - Expression::Now { fun: _ } => Ok(ExpressionType::new( - FieldType::Timestamp, - false, - dozer_types::types::SourceDefinition::Dynamic, - false, - )), - Expression::Json { fun: _, args: _ } => Ok(ExpressionType::new( - FieldType::Json, - false, - dozer_types::types::SourceDefinition::Dynamic, - false, - )), - Expression::Case { - operand: _, - conditions: _, - results, - else_result: _, - } => { - let typ = results.first().unwrap().get_type(schema)?; - Ok(ExpressionType::new( - typ.return_type, - true, - dozer_types::types::SourceDefinition::Dynamic, - false, - )) - } - #[cfg(feature = "python")] - Expression::PythonUDF { return_type, .. } => Ok(ExpressionType::new( - *return_type, - false, - SourceDefinition::Dynamic, - false, - )), - #[cfg(feature = "onnx")] - Expression::OnnxUDF { .. } => Ok(ExpressionType::new( - FieldType::Float, - false, - SourceDefinition::Dynamic, - false, - )), - #[cfg(feature = "javascript")] - Expression::JavaScriptUdf(udf) => Ok(udf.get_type()), - Expression::IsNull { arg: _ } => Ok(ExpressionType::new( - FieldType::Boolean, - false, - SourceDefinition::Dynamic, - false, - )), - Expression::IsNotNull { arg: _ } => Ok(ExpressionType::new( - FieldType::Boolean, - false, - SourceDefinition::Dynamic, - false, - )), - } - } -} - -fn get_unary_operator_type( - operator: &UnaryOperatorType, - expression: &Expression, - schema: &Schema, -) -> Result { - let field_type = expression.get_type(schema)?; - match operator { - UnaryOperatorType::Not => match field_type.return_type { - FieldType::Boolean => Ok(field_type), - field_type => Err(Error::CannotApplyNotTo(field_type)), - }, - UnaryOperatorType::Plus => Ok(field_type), - UnaryOperatorType::Minus => Ok(field_type), - } -} - -fn get_binary_operator_type( - left: &Expression, - operator: &BinaryOperatorType, - right: &Expression, - schema: &Schema, -) -> Result { - let left_field_type = left.get_type(schema)?; - let right_field_type = right.get_type(schema)?; - match operator { - BinaryOperatorType::Eq - | BinaryOperatorType::Ne - | BinaryOperatorType::Gt - | BinaryOperatorType::Gte - | BinaryOperatorType::Lt - | BinaryOperatorType::Lte => Ok(ExpressionType::new( - FieldType::Boolean, - false, - SourceDefinition::Dynamic, - false, - )), - - BinaryOperatorType::And | BinaryOperatorType::Or => { - match (left_field_type.return_type, right_field_type.return_type) { - (FieldType::Boolean, FieldType::Boolean) => Ok(ExpressionType::new( - FieldType::Boolean, - false, - SourceDefinition::Dynamic, - false, - )), - ( - FieldType::Boolean, - FieldType::UInt - | FieldType::U128 - | FieldType::Int - | FieldType::I128 - | FieldType::String - | FieldType::Text, - ) => Ok(ExpressionType::new( - FieldType::Boolean, - false, - SourceDefinition::Dynamic, - false, - )), - ( - FieldType::UInt - | FieldType::U128 - | FieldType::Int - | FieldType::I128 - | FieldType::String - | FieldType::Text, - FieldType::Boolean, - ) => Ok(ExpressionType::new( - FieldType::Boolean, - false, - SourceDefinition::Dynamic, - false, - )), - (left_field_type, right_field_type) => Err(Error::CannotApplyBinaryOperator { - operator: operator.clone(), - left_field_type, - right_field_type, - }), - } - } - - BinaryOperatorType::Add - | BinaryOperatorType::Sub - | BinaryOperatorType::Mul - | BinaryOperatorType::Mod => { - match (left_field_type.return_type, right_field_type.return_type) { - (FieldType::UInt, FieldType::UInt) => Ok(ExpressionType::new( - FieldType::UInt, - false, - SourceDefinition::Dynamic, - false, - )), - (FieldType::U128, FieldType::U128) - | (FieldType::U128, FieldType::UInt) - | (FieldType::UInt, FieldType::U128) => Ok(ExpressionType::new( - FieldType::U128, - false, - SourceDefinition::Dynamic, - false, - )), - (FieldType::Timestamp, FieldType::Timestamp) => Ok(ExpressionType::new( - FieldType::Duration, - false, - SourceDefinition::Dynamic, - false, - )), - (FieldType::Timestamp, FieldType::Duration) => Ok(ExpressionType::new( - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - false, - )), - (FieldType::Duration, FieldType::Timestamp) => Ok(ExpressionType::new( - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - false, - )), - (FieldType::Duration, FieldType::Duration) => Ok(ExpressionType::new( - FieldType::Duration, - false, - SourceDefinition::Dynamic, - false, - )), - (FieldType::Int, FieldType::Int) - | (FieldType::Int, FieldType::UInt) - | (FieldType::UInt, FieldType::Int) => Ok(ExpressionType::new( - FieldType::Int, - false, - SourceDefinition::Dynamic, - false, - )), - (FieldType::I128, FieldType::I128) - | (FieldType::I128, FieldType::UInt) - | (FieldType::I128, FieldType::U128) - | (FieldType::I128, FieldType::Int) - | (FieldType::UInt, FieldType::I128) - | (FieldType::U128, FieldType::I128) - | (FieldType::Int, FieldType::I128) => Ok(ExpressionType::new( - FieldType::I128, - false, - SourceDefinition::Dynamic, - false, - )), - (FieldType::Float, FieldType::Float) - | (FieldType::Float, FieldType::UInt) - | (FieldType::Float, FieldType::U128) - | (FieldType::Float, FieldType::Int) - | (FieldType::Float, FieldType::I128) - | (FieldType::UInt, FieldType::Float) - | (FieldType::U128, FieldType::Float) - | (FieldType::Int, FieldType::Float) - | (FieldType::I128, FieldType::Float) => Ok(ExpressionType::new( - FieldType::Float, - false, - SourceDefinition::Dynamic, - false, - )), - (FieldType::Decimal, FieldType::Decimal) - | (FieldType::UInt, FieldType::Decimal) - | (FieldType::U128, FieldType::Decimal) - | (FieldType::Int, FieldType::Decimal) - | (FieldType::I128, FieldType::Decimal) - | (FieldType::Float, FieldType::Decimal) - | (FieldType::Decimal, FieldType::UInt) - | (FieldType::Decimal, FieldType::U128) - | (FieldType::Decimal, FieldType::Int) - | (FieldType::Decimal, FieldType::I128) - | (FieldType::Decimal, FieldType::Float) => Ok(ExpressionType::new( - FieldType::Decimal, - false, - SourceDefinition::Dynamic, - false, - )), - (left_field_type, right_field_type) => Err(Error::CannotApplyBinaryOperator { - operator: operator.clone(), - left_field_type, - right_field_type, - }), - } - } - - BinaryOperatorType::Div => { - match (left_field_type.return_type, right_field_type.return_type) { - (FieldType::Int, FieldType::UInt) - | (FieldType::Int, FieldType::Int) - | (FieldType::Int, FieldType::U128) - | (FieldType::Int, FieldType::I128) - | (FieldType::Int, FieldType::Float) - | (FieldType::I128, FieldType::UInt) - | (FieldType::I128, FieldType::Int) - | (FieldType::I128, FieldType::U128) - | (FieldType::I128, FieldType::I128) - | (FieldType::I128, FieldType::Float) - | (FieldType::UInt, FieldType::UInt) - | (FieldType::UInt, FieldType::U128) - | (FieldType::UInt, FieldType::Int) - | (FieldType::UInt, FieldType::I128) - | (FieldType::UInt, FieldType::Float) - | (FieldType::U128, FieldType::UInt) - | (FieldType::U128, FieldType::U128) - | (FieldType::U128, FieldType::Int) - | (FieldType::U128, FieldType::I128) - | (FieldType::U128, FieldType::Float) - | (FieldType::Float, FieldType::UInt) - | (FieldType::Float, FieldType::U128) - | (FieldType::Float, FieldType::Int) - | (FieldType::Float, FieldType::I128) - | (FieldType::Float, FieldType::Float) => Ok(ExpressionType::new( - FieldType::Float, - false, - SourceDefinition::Dynamic, - false, - )), - (FieldType::Decimal, FieldType::Decimal) - | (FieldType::Decimal, FieldType::UInt) - | (FieldType::Decimal, FieldType::U128) - | (FieldType::Decimal, FieldType::Int) - | (FieldType::Decimal, FieldType::I128) - | (FieldType::Decimal, FieldType::Float) - | (FieldType::UInt, FieldType::Decimal) - | (FieldType::U128, FieldType::Decimal) - | (FieldType::Int, FieldType::Decimal) - | (FieldType::I128, FieldType::Decimal) - | (FieldType::Float, FieldType::Decimal) => Ok(ExpressionType::new( - FieldType::Decimal, - false, - SourceDefinition::Dynamic, - false, - )), - (left_field_type, right_field_type) => Err(Error::CannotApplyBinaryOperator { - operator: operator.clone(), - left_field_type, - right_field_type, - }), - } - } - } -} - -fn get_aggregate_function_type( - function: &AggregateFunctionType, - args: &[Expression], - schema: &Schema, -) -> Result { - match function { - AggregateFunctionType::Avg => validate_avg(args, schema), - AggregateFunctionType::Count => validate_count(args, schema), - AggregateFunctionType::Max => validate_max(args, schema), - AggregateFunctionType::MaxAppendOnly => validate_max_append_only(args, schema), - AggregateFunctionType::MaxValue => validate_max_value(args, schema), - AggregateFunctionType::Min => validate_min(args, schema), - AggregateFunctionType::MinAppendOnly => validate_min_append_only(args, schema), - AggregateFunctionType::MinValue => validate_min_value(args, schema), - AggregateFunctionType::Sum => validate_sum(args, schema), - } -} - -fn validate_avg(args: &[Expression], schema: &Schema) -> Result { - let arg = validate_one_argument(args, schema, AggregateFunctionType::Avg)?; - - let ret_type = match arg.return_type { - FieldType::UInt => FieldType::Decimal, - FieldType::U128 => FieldType::Decimal, - FieldType::Int => FieldType::Decimal, - FieldType::Int8 => FieldType::Decimal, - FieldType::I128 => FieldType::Decimal, - FieldType::Float => FieldType::Float, - FieldType::Decimal => FieldType::Decimal, - FieldType::Duration => FieldType::Duration, - FieldType::Boolean - | FieldType::String - | FieldType::Text - | FieldType::Date - | FieldType::Timestamp - | FieldType::Binary - | FieldType::Json - | FieldType::Point => { - return Err(Error::InvalidFunctionArgumentType { - function_name: AggregateFunctionType::Avg.to_string(), - argument_index: 0, - actual: arg.return_type, - expected: vec![ - FieldType::UInt, - FieldType::U128, - FieldType::Int, - FieldType::I128, - FieldType::Float, - FieldType::Decimal, - FieldType::Duration, - ], - }); - } - }; - - Ok(ExpressionType::new( - ret_type, - true, - SourceDefinition::Dynamic, - false, - )) -} - -fn validate_count(_args: &[Expression], _schema: &Schema) -> Result { - Ok(ExpressionType::new( - FieldType::Int, - false, - SourceDefinition::Dynamic, - false, - )) -} - -fn validate_max(args: &[Expression], schema: &Schema) -> Result { - let arg = validate_one_argument(args, schema, AggregateFunctionType::Max)?; - - let ret_type = match arg.return_type { - FieldType::UInt => FieldType::UInt, - FieldType::U128 => FieldType::U128, - FieldType::Int => FieldType::Int, - FieldType::Int8 => FieldType::Int8, - FieldType::I128 => FieldType::I128, - FieldType::Float => FieldType::Float, - FieldType::Decimal => FieldType::Decimal, - FieldType::Timestamp => FieldType::Timestamp, - FieldType::Date => FieldType::Date, - FieldType::Duration => FieldType::Duration, - FieldType::Boolean - | FieldType::String - | FieldType::Text - | FieldType::Binary - | FieldType::Json - | FieldType::Point => { - return Err(Error::InvalidFunctionArgumentType { - function_name: AggregateFunctionType::Max.to_string(), - argument_index: 0, - actual: arg.return_type, - expected: vec![ - FieldType::Decimal, - FieldType::UInt, - FieldType::U128, - FieldType::Int, - FieldType::I128, - FieldType::Float, - FieldType::Timestamp, - FieldType::Date, - FieldType::Duration, - ], - }); - } - }; - Ok(ExpressionType::new( - ret_type, - true, - SourceDefinition::Dynamic, - false, - )) -} - -fn validate_min(args: &[Expression], schema: &Schema) -> Result { - let arg = validate_one_argument(args, schema, AggregateFunctionType::Min)?; - - let ret_type = match arg.return_type { - FieldType::UInt => FieldType::UInt, - FieldType::U128 => FieldType::U128, - FieldType::Int => FieldType::Int, - FieldType::Int8 => FieldType::Int8, - FieldType::I128 => FieldType::I128, - FieldType::Float => FieldType::Float, - FieldType::Decimal => FieldType::Decimal, - FieldType::Timestamp => FieldType::Timestamp, - FieldType::Date => FieldType::Date, - FieldType::Duration => FieldType::Duration, - FieldType::Boolean - | FieldType::String - | FieldType::Text - | FieldType::Binary - | FieldType::Json - | FieldType::Point => { - return Err(Error::InvalidFunctionArgumentType { - function_name: AggregateFunctionType::Min.to_string(), - argument_index: 0, - actual: arg.return_type, - expected: vec![ - FieldType::Decimal, - FieldType::UInt, - FieldType::U128, - FieldType::Int, - FieldType::I128, - FieldType::Float, - FieldType::Timestamp, - FieldType::Date, - FieldType::Duration, - ], - }); - } - }; - - Ok(ExpressionType::new( - ret_type, - true, - SourceDefinition::Dynamic, - false, - )) -} - -fn validate_max_append_only(args: &[Expression], schema: &Schema) -> Result { - let arg = validate_one_argument(args, schema, AggregateFunctionType::MaxAppendOnly)?; - - let ret_type = match arg.return_type { - FieldType::UInt => FieldType::UInt, - FieldType::U128 => FieldType::U128, - FieldType::Int => FieldType::Int, - FieldType::Int8 => FieldType::Int8, - FieldType::I128 => FieldType::I128, - FieldType::Float => FieldType::Float, - FieldType::Decimal => FieldType::Decimal, - FieldType::Timestamp => FieldType::Timestamp, - FieldType::Date => FieldType::Date, - FieldType::Duration => FieldType::Duration, - FieldType::Boolean - | FieldType::String - | FieldType::Text - | FieldType::Binary - | FieldType::Json - | FieldType::Point => { - return Err(Error::InvalidFunctionArgumentType { - function_name: AggregateFunctionType::MaxAppendOnly.to_string(), - argument_index: 0, - actual: arg.return_type, - expected: vec![ - FieldType::Decimal, - FieldType::UInt, - FieldType::U128, - FieldType::Int, - FieldType::I128, - FieldType::Float, - FieldType::Timestamp, - FieldType::Date, - FieldType::Duration, - ], - }); - } - }; - Ok(ExpressionType::new( - ret_type, - true, - SourceDefinition::Dynamic, - false, - )) -} - -fn validate_min_append_only(args: &[Expression], schema: &Schema) -> Result { - let arg = validate_one_argument(args, schema, AggregateFunctionType::MinAppendOnly)?; - - let ret_type = match arg.return_type { - FieldType::UInt => FieldType::UInt, - FieldType::U128 => FieldType::U128, - FieldType::Int => FieldType::Int, - FieldType::Int8 => FieldType::Int8, - FieldType::I128 => FieldType::I128, - FieldType::Float => FieldType::Float, - FieldType::Decimal => FieldType::Decimal, - FieldType::Timestamp => FieldType::Timestamp, - FieldType::Date => FieldType::Date, - FieldType::Duration => FieldType::Duration, - FieldType::Boolean - | FieldType::String - | FieldType::Text - | FieldType::Binary - | FieldType::Json - | FieldType::Point => { - return Err(Error::InvalidFunctionArgumentType { - function_name: AggregateFunctionType::MinAppendOnly.to_string(), - argument_index: 0, - actual: arg.return_type, - expected: vec![ - FieldType::Decimal, - FieldType::UInt, - FieldType::U128, - FieldType::Int, - FieldType::I128, - FieldType::Float, - FieldType::Timestamp, - FieldType::Date, - FieldType::Duration, - ], - }); - } - }; - Ok(ExpressionType::new( - ret_type, - true, - SourceDefinition::Dynamic, - false, - )) -} - -fn validate_sum(args: &[Expression], schema: &Schema) -> Result { - let arg = validate_one_argument(args, schema, AggregateFunctionType::Sum)?; - - let ret_type = match arg.return_type { - FieldType::UInt => FieldType::UInt, - FieldType::U128 => FieldType::U128, - FieldType::Int => FieldType::Int, - FieldType::Int8 => FieldType::Int8, - FieldType::I128 => FieldType::I128, - FieldType::Float => FieldType::Float, - FieldType::Decimal => FieldType::Decimal, - FieldType::Duration => FieldType::Duration, - FieldType::Boolean - | FieldType::String - | FieldType::Text - | FieldType::Date - | FieldType::Timestamp - | FieldType::Binary - | FieldType::Json - | FieldType::Point => { - return Err(Error::InvalidFunctionArgumentType { - function_name: AggregateFunctionType::Sum.to_string(), - argument_index: 0, - actual: arg.return_type, - expected: vec![ - FieldType::UInt, - FieldType::U128, - FieldType::Int, - FieldType::I128, - FieldType::Float, - FieldType::Decimal, - FieldType::Duration, - ], - }); - } - }; - Ok(ExpressionType::new( - ret_type, - true, - SourceDefinition::Dynamic, - false, - )) -} - -fn validate_max_value(args: &[Expression], schema: &Schema) -> Result { - let (base_arg, arg) = validate_two_arguments(args, schema, AggregateFunctionType::MaxValue)?; - - match base_arg.return_type { - FieldType::UInt => FieldType::UInt, - FieldType::U128 => FieldType::U128, - FieldType::Int => FieldType::Int, - FieldType::Int8 => FieldType::Int8, - FieldType::I128 => FieldType::I128, - FieldType::Float => FieldType::Float, - FieldType::Decimal => FieldType::Decimal, - FieldType::Timestamp => FieldType::Timestamp, - FieldType::Date => FieldType::Date, - FieldType::Duration => FieldType::Duration, - FieldType::Boolean - | FieldType::String - | FieldType::Text - | FieldType::Binary - | FieldType::Json - | FieldType::Point => { - return Err(Error::InvalidFunctionArgumentType { - function_name: AggregateFunctionType::MaxValue.to_string(), - argument_index: 0, - actual: base_arg.return_type, - expected: vec![ - FieldType::Decimal, - FieldType::UInt, - FieldType::U128, - FieldType::Int, - FieldType::I128, - FieldType::Float, - FieldType::Timestamp, - FieldType::Date, - FieldType::Duration, - ], - }); - } - }; - - Ok(ExpressionType::new( - arg.return_type, - true, - SourceDefinition::Dynamic, - false, - )) -} - -fn validate_min_value(args: &[Expression], schema: &Schema) -> Result { - let (base_arg, arg) = validate_two_arguments(args, schema, AggregateFunctionType::MinValue)?; - - match base_arg.return_type { - FieldType::UInt => FieldType::UInt, - FieldType::U128 => FieldType::U128, - FieldType::Int => FieldType::Int, - FieldType::Int8 => FieldType::Int8, - FieldType::I128 => FieldType::I128, - FieldType::Float => FieldType::Float, - FieldType::Decimal => FieldType::Decimal, - FieldType::Timestamp => FieldType::Timestamp, - FieldType::Date => FieldType::Date, - FieldType::Duration => FieldType::Duration, - FieldType::Boolean - | FieldType::String - | FieldType::Text - | FieldType::Binary - | FieldType::Json - | FieldType::Point => { - return Err(Error::InvalidFunctionArgumentType { - function_name: AggregateFunctionType::MinValue.to_string(), - argument_index: 0, - actual: base_arg.return_type, - expected: vec![ - FieldType::Decimal, - FieldType::UInt, - FieldType::U128, - FieldType::Int, - FieldType::I128, - FieldType::Float, - FieldType::Timestamp, - FieldType::Date, - FieldType::Duration, - ], - }); - } - }; - - Ok(ExpressionType::new( - arg.return_type, - true, - SourceDefinition::Dynamic, - false, - )) -} diff --git a/dozer-sql/expression/src/geo/common.rs b/dozer-sql/expression/src/geo/common.rs deleted file mode 100644 index 9022f28c69..0000000000 --- a/dozer-sql/expression/src/geo/common.rs +++ /dev/null @@ -1,56 +0,0 @@ -use crate::error::Error; -use crate::execution::{Expression, ExpressionType}; - -use crate::geo::distance::{evaluate_distance, validate_distance}; -use crate::geo::point::{evaluate_point, validate_point}; -use dozer_types::types::Record; -use dozer_types::types::{Field, Schema}; -use std::fmt::{Display, Formatter}; - -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Hash)] -pub enum GeoFunctionType { - Point, - Distance, -} - -impl Display for GeoFunctionType { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - match self { - GeoFunctionType::Point => f.write_str("POINT"), - GeoFunctionType::Distance => f.write_str("DISTANCE"), - } - } -} - -pub(crate) fn get_geo_function_type( - function: &GeoFunctionType, - args: &[Expression], - schema: &Schema, -) -> Result { - match function { - GeoFunctionType::Point => validate_point(args, schema), - GeoFunctionType::Distance => validate_distance(args, schema), - } -} - -impl GeoFunctionType { - pub fn new(name: &str) -> Option { - match name { - "point" => Some(GeoFunctionType::Point), - "distance" => Some(GeoFunctionType::Distance), - _ => None, - } - } - - pub(crate) fn evaluate( - &self, - schema: &Schema, - args: &mut [Expression], - record: &Record, - ) -> Result { - match self { - GeoFunctionType::Point => evaluate_point(schema, args, record), - GeoFunctionType::Distance => evaluate_distance(schema, args, record), - } - } -} diff --git a/dozer-sql/expression/src/geo/distance.rs b/dozer-sql/expression/src/geo/distance.rs deleted file mode 100644 index 082a47e87a..0000000000 --- a/dozer-sql/expression/src/geo/distance.rs +++ /dev/null @@ -1,268 +0,0 @@ -use std::str::FromStr; - -use crate::arg_utils::{extract_point, validate_num_arguments}; -use crate::error::Error; -use dozer_types::types::Record; -use dozer_types::types::{Field, FieldType, Schema}; - -use crate::execution::{Expression, ExpressionType}; -use crate::geo::common::GeoFunctionType; -use dozer_types::geo::GeodesicDistance; -use dozer_types::geo::HaversineDistance; -use dozer_types::geo::VincentyDistance; - -use dozer_types::ordered_float::OrderedFloat; - -const EXPECTED_ARGS_TYPES: &[FieldType] = &[FieldType::Point, FieldType::Point, FieldType::String]; - -pub enum Algorithm { - Geodesic, - Haversine, - Vincenty, -} - -impl FromStr for Algorithm { - type Err = Error; - - fn from_str(s: &str) -> Result { - match s { - "GEODESIC" => Ok(Algorithm::Geodesic), - "HAVERSINE" => Ok(Algorithm::Haversine), - "VINCENTY" => Ok(Algorithm::Vincenty), - &_ => Err(Error::InvalidDistanceAlgorithm(s.to_string())), - } - } -} - -const DEFAULT_ALGORITHM: Algorithm = Algorithm::Geodesic; - -pub(crate) fn validate_distance( - args: &[Expression], - schema: &Schema, -) -> Result { - let ret_type = FieldType::Float; - validate_num_arguments(2..4, args.len(), GeoFunctionType::Distance)?; - - for (argument_index, exp) in args.iter().enumerate() { - let return_type = exp.get_type(schema)?.return_type; - let expected_arg_type_option = EXPECTED_ARGS_TYPES.get(argument_index); - if let Some(expected_arg_type) = expected_arg_type_option { - if &return_type != expected_arg_type { - return Err(Error::InvalidFunctionArgumentType { - function_name: GeoFunctionType::Distance.to_string(), - argument_index, - actual: return_type, - expected: vec![*expected_arg_type], - }); - } - } - } - - Ok(ExpressionType::new( - ret_type, - false, - dozer_types::types::SourceDefinition::Dynamic, - false, - )) -} - -pub(crate) fn evaluate_distance( - schema: &Schema, - args: &mut [Expression], - record: &Record, -) -> Result { - validate_num_arguments(2..4, args.len(), GeoFunctionType::Distance)?; - let f_from = args[0].evaluate(record, schema)?; - - let f_to = args[1].evaluate(record, schema)?; - - if f_from == Field::Null || f_to == Field::Null { - Ok(Field::Null) - } else { - let from = extract_point(f_from, GeoFunctionType::Distance, 0)?; - let to = extract_point(f_to, GeoFunctionType::Distance, 1)?; - let calculation_type = args.get_mut(2).map_or_else( - || Ok(DEFAULT_ALGORITHM), - |arg| { - let f = arg.evaluate(record, schema)?; - let t = f.to_string(); - Algorithm::from_str(&t) - }, - )?; - - let distance: OrderedFloat = match calculation_type { - Algorithm::Geodesic => Ok(from.geodesic_distance(&to)), - Algorithm::Haversine => Ok(from.0.haversine_distance(&to.0)), - Algorithm::Vincenty => from - .0 - .vincenty_distance(&to.0) - .map_err(Error::FailedToCalculateVincentyDistance), - }?; - - Ok(Field::Float(distance)) - } -} - -#[cfg(test)] -mod tests { - use super::*; - - use dozer_types::types::{DozerPoint, FieldDefinition, SourceDefinition}; - use proptest::prelude::*; - use Expression::Literal; - - #[test] - fn test_geo() { - proptest!(ProptestConfig::with_cases(1000), move |(x1: f64, x2: f64, y1: f64, y2: f64)| { - let row = Record::new(vec![]); - let from = Field::Point(DozerPoint::from((x1, y1))); - let to = Field::Point(DozerPoint::from((x2, y2))); - let null = Field::Null; - - test_distance(&from, &to, None, &row, None); - test_distance(&from, &null, None, &row, Some(Ok(Field::Null))); - test_distance(&null, &to, None, &row, Some(Ok(Field::Null))); - - test_distance(&from, &to, Some(Algorithm::Geodesic), &row, None); - test_distance(&from, &null, Some(Algorithm::Geodesic), &row, Some(Ok(Field::Null))); - test_distance(&null, &to, Some(Algorithm::Geodesic), &row, Some(Ok(Field::Null))); - - test_distance(&from, &to, Some(Algorithm::Haversine), &row, None); - test_distance(&from, &null, Some(Algorithm::Haversine), &row, Some(Ok(Field::Null))); - test_distance(&null, &to, Some(Algorithm::Haversine), &row, Some(Ok(Field::Null))); - - // test_distance(&from, &to, Some(Algorithm::Vincenty), &row, None); - // test_distance(&from, &null, Some(Algorithm::Vincenty), &row, Some(Ok(Field::Null))); - // test_distance(&null, &to, Some(Algorithm::Vincenty), &row, Some(Ok(Field::Null))); - }); - } - - fn test_distance( - from: &Field, - to: &Field, - typ: Option, - row: &Record, - result: Option>, - ) { - let args = &mut [Literal(from.clone()), Literal(to.clone())]; - if validate_distance(args, &Schema::default()).is_ok() { - match result { - None => { - let from_f = from.to_owned(); - let to_f = to.to_owned(); - let f = extract_point(from_f, GeoFunctionType::Distance, 0).unwrap(); - let t = extract_point(to_f, GeoFunctionType::Distance, 0).unwrap(); - let _dist = match typ { - None => f.geodesic_distance(&t), - Some(Algorithm::Geodesic) => f.geodesic_distance(&t), - Some(Algorithm::Haversine) => f.0.haversine_distance(&t.0), - Some(Algorithm::Vincenty) => OrderedFloat(0.0), - // Some(Algorithm::Vincenty) => f.0.vincenty_distance(&t.0).unwrap(), - }; - assert!(matches!( - evaluate_distance(&Schema::default(), args, row), - Ok(Field::Float(_dist)), - )) - } - Some(_val) => { - assert!(matches!( - evaluate_distance(&Schema::default(), args, row), - _val, - )) - } - } - } - } - - #[test] - fn test_validate_distance() { - let schema = Schema::default() - .field( - FieldDefinition::new( - String::from("from"), - FieldType::Point, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("to"), - FieldType::Point, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(); - - let result = validate_distance(&[], &schema); - assert!(result.is_err()); - assert!(matches!( - result, - Err(Error::InvalidNumberOfArguments { .. }) - )); - - let result = validate_distance(&[Expression::Column { index: 0 }], &schema); - - assert!(result.is_err()); - assert!(matches!( - result, - Err(Error::InvalidNumberOfArguments { .. }) - )); - - let result = validate_distance( - &[ - Expression::Column { index: 0 }, - Expression::Column { index: 1 }, - ], - &schema, - ); - - assert!(result.is_ok()); - - let result = validate_distance( - &[ - Expression::Column { index: 0 }, - Expression::Column { index: 1 }, - Expression::Literal(Field::String("GEODESIC".to_string())), - ], - &schema, - ); - - assert!(result.is_ok()); - - let result = validate_distance( - &[ - Expression::Column { index: 0 }, - Expression::Column { index: 1 }, - Expression::Literal(Field::String("GEODESIC".to_string())), - Expression::Column { index: 2 }, - ], - &schema, - ); - - assert!(result.is_err()); - assert!(matches!( - result, - Err(Error::InvalidNumberOfArguments { .. }) - )); - - let result = validate_distance( - &[ - Expression::Column { index: 0 }, - Expression::Literal(Field::String("GEODESIC".to_string())), - Expression::Column { index: 2 }, - ], - &schema, - ); - - let _expected_types = [FieldType::Point]; - assert!(result.is_err()); - assert!(matches!( - result, - Err(Error::InvalidFunctionArgumentType { .. }) - )); - } -} diff --git a/dozer-sql/expression/src/geo/mod.rs b/dozer-sql/expression/src/geo/mod.rs deleted file mode 100644 index c718300627..0000000000 --- a/dozer-sql/expression/src/geo/mod.rs +++ /dev/null @@ -1,3 +0,0 @@ -pub mod common; -pub mod distance; -pub mod point; diff --git a/dozer-sql/expression/src/geo/point.rs b/dozer-sql/expression/src/geo/point.rs deleted file mode 100644 index 378749a426..0000000000 --- a/dozer-sql/expression/src/geo/point.rs +++ /dev/null @@ -1,234 +0,0 @@ -use crate::arg_utils::{extract_float, validate_num_arguments}; -use crate::error::Error; -use dozer_types::types::Record; -use dozer_types::types::{DozerPoint, Field, FieldType, Schema}; - -use crate::execution::{Expression, ExpressionType}; -use crate::geo::common::GeoFunctionType; - -pub fn validate_point(args: &[Expression], schema: &Schema) -> Result { - let ret_type = FieldType::Point; - let expected_arg_type = FieldType::Float; - - validate_num_arguments(2..3, args.len(), GeoFunctionType::Point)?; - - for (argument_index, exp) in args.iter().enumerate() { - let return_type = exp.get_type(schema)?.return_type; - if return_type != expected_arg_type { - return Err(Error::InvalidFunctionArgumentType { - function_name: GeoFunctionType::Point.to_string(), - argument_index, - actual: return_type, - expected: vec![expected_arg_type], - }); - } - } - - Ok(ExpressionType::new( - ret_type, - false, - dozer_types::types::SourceDefinition::Dynamic, - false, - )) -} - -pub fn evaluate_point( - schema: &Schema, - args: &mut [Expression], - record: &Record, -) -> Result { - validate_num_arguments(2..3, args.len(), GeoFunctionType::Point)?; - let f_x = args[0].evaluate(record, schema)?; - let f_y = args[1].evaluate(record, schema)?; - - if f_x == Field::Null || f_y == Field::Null { - Ok(Field::Null) - } else { - let x = extract_float(f_x, GeoFunctionType::Point, 0)?; - let y = extract_float(f_y, GeoFunctionType::Point, 1)?; - - Ok(Field::Point(DozerPoint::from((x, y)))) - } -} - -#[cfg(test)] -mod tests { - use super::*; - - use dozer_types::types::{FieldDefinition, SourceDefinition}; - use proptest::prelude::*; - - #[test] - fn test_point() { - proptest!( - ProptestConfig::with_cases(1000), move |(x: i64, y: i64)| { - test_validate_point(x, y); - test_evaluate_point(x, y); - }); - } - - fn test_validate_point(x: i64, y: i64) { - let schema = Schema::default() - .field( - FieldDefinition::new( - String::from("x"), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("y"), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(); - - let result = validate_point(&[], &schema); - assert!(result.is_err()); - assert!(matches!( - result, - Err(Error::InvalidNumberOfArguments { .. }) - )); - - let result = validate_point(&[Expression::Column { index: 0 }], &schema); - - assert!(result.is_err()); - assert!(matches!( - result, - Err(Error::InvalidNumberOfArguments { .. }) - )); - - let result = validate_point( - &[ - Expression::Column { index: 0 }, - Expression::Column { index: 1 }, - ], - &schema, - ); - - assert!(result.is_ok()); - - let result = validate_point( - &[ - Expression::Column { index: 0 }, - Expression::Column { index: 1 }, - Expression::Column { index: 2 }, - ], - &schema, - ); - - assert!(result.is_err()); - assert!(matches!( - result, - Err(Error::InvalidNumberOfArguments { .. }) - )); - - let result = validate_point( - &[ - Expression::Column { index: 0 }, - Expression::Literal(Field::Int(y)), - ], - &schema, - ); - - assert!(result.is_err()); - assert!(matches!( - result, - Err(Error::InvalidFunctionArgumentType { .. }) - )); - - let result = validate_point( - &[ - Expression::Literal(Field::Int(x)), - Expression::Column { index: 0 }, - ], - &schema, - ); - - assert!(result.is_err()); - assert!(matches!( - result, - Err(Error::InvalidFunctionArgumentType { .. }) - )); - } - - fn test_evaluate_point(x: i64, y: i64) { - let row = Record::new(vec![]); - - let schema = Schema::default() - .field( - FieldDefinition::new( - String::from("x"), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("y"), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(); - - let result = evaluate_point(&schema, &mut [], &row); - assert!(result.is_err()); - assert!(matches!( - result, - Err(Error::InvalidNumberOfArguments { .. }) - )); - - let result = evaluate_point(&schema, &mut [Expression::Literal(Field::Int(x))], &row); - assert!(result.is_err()); - assert!(matches!( - result, - Err(Error::InvalidNumberOfArguments { .. }) - )); - - let result = evaluate_point( - &schema, - &mut [ - Expression::Literal(Field::Int(x)), - Expression::Literal(Field::Int(y)), - ], - &row, - ); - - assert!(result.is_ok()); - - let result = evaluate_point( - &schema, - &mut [ - Expression::Literal(Field::Int(x)), - Expression::Literal(Field::Null), - ], - &row, - ); - - assert!(result.is_ok()); - assert!(matches!(result, Ok(Field::Null))); - - let result = evaluate_point( - &schema, - &mut [ - Expression::Literal(Field::Null), - Expression::Literal(Field::Int(y)), - ], - &row, - ); - - assert!(result.is_ok()); - assert!(matches!(result, Ok(Field::Null))); - } -} diff --git a/dozer-sql/expression/src/in_list.rs b/dozer-sql/expression/src/in_list.rs deleted file mode 100644 index d538214d65..0000000000 --- a/dozer-sql/expression/src/in_list.rs +++ /dev/null @@ -1,28 +0,0 @@ -use dozer_types::types::Record; -use dozer_types::types::{Field, Schema}; - -use crate::error::Error; -use crate::execution::Expression; - -pub(crate) fn evaluate_in_list( - schema: &Schema, - expr: &mut Expression, - list: &mut [Expression], - negated: bool, - record: &Record, -) -> Result { - let field = expr.evaluate(record, schema)?; - let mut result = false; - for item in list { - let item = item.evaluate(record, schema)?; - if field == item { - result = true; - break; - } - } - // Negate the result if the IN list was negated. - if negated { - result = !result; - } - Ok(Field::Boolean(result)) -} diff --git a/dozer-sql/expression/src/is_null.rs b/dozer-sql/expression/src/is_null.rs deleted file mode 100644 index 5d682991f7..0000000000 --- a/dozer-sql/expression/src/is_null.rs +++ /dev/null @@ -1,50 +0,0 @@ -use dozer_types::types::Record; -use dozer_types::types::{Field, Schema}; - -use crate::error::Error; -use crate::execution::Expression; - -pub(crate) fn evaluate_is_null( - schema: &Schema, - expr: &mut Expression, - record: &Record, -) -> Result { - let field = expr.evaluate(record, schema)?; - Ok(Field::Boolean(field == Field::Null)) -} - -pub(crate) fn evaluate_is_not_null( - schema: &Schema, - expr: &mut Expression, - record: &Record, -) -> Result { - let field = expr.evaluate(record, schema)?; - Ok(Field::Boolean(field != Field::Null)) -} - -#[test] -fn test_is_null() { - let mut value = Box::new(Expression::Literal(Field::Int(65))); - assert_eq!( - evaluate_is_null(&Schema::default(), &mut value, &Record::new(vec![])).unwrap(), - Field::Boolean(false) - ); - - let mut value = Box::new(Expression::Literal(Field::Null)); - assert_eq!( - evaluate_is_null(&Schema::default(), &mut value, &Record::new(vec![])).unwrap(), - Field::Boolean(true) - ); - - let mut value = Box::new(Expression::Literal(Field::Int(65))); - assert_eq!( - evaluate_is_not_null(&Schema::default(), &mut value, &Record::new(vec![])).unwrap(), - Field::Boolean(true) - ); - - let mut value = Box::new(Expression::Literal(Field::Null)); - assert_eq!( - evaluate_is_not_null(&Schema::default(), &mut value, &Record::new(vec![])).unwrap(), - Field::Boolean(false) - ); -} diff --git a/dozer-sql/expression/src/javascript/evaluate.rs b/dozer-sql/expression/src/javascript/evaluate.rs deleted file mode 100644 index c2c4b7023d..0000000000 --- a/dozer-sql/expression/src/javascript/evaluate.rs +++ /dev/null @@ -1,127 +0,0 @@ -use std::{num::NonZeroI32, sync::Arc}; - -use deno_core::{error::AnyError, *}; -use dozer_types::{ - errors::types::{DeserializationError, SerializationError}, - json_types::JsonValue, - parking_lot, serde_json, thiserror, - types::{Field, FieldType, Record, Schema, SourceDefinition}, -}; -use tokio::{runtime::Runtime, sync::Mutex}; - -use crate::execution::{Expression, ExpressionType}; - -#[derive(Debug, Clone)] -pub struct Udf { - function_name: String, - arg: Box, - tokio_runtime: Arc, - /// `Arc` to enable `Clone`. Not sure why `Expression` should be `Clone`. - deno_runtime: Arc>, - function: NonZeroI32, -} - -impl PartialEq for Udf { - fn eq(&self, other: &Self) -> bool { - // This is obviously wrong. We have to lift the `PartialEq` constraint. - self.function_name == other.function_name && self.arg == other.arg - } -} - -#[derive(Debug, thiserror::Error)] -pub enum Error { - #[error("failed to create deno runtime: {0}")] - CreateRuntime(#[from] dozer_deno::RuntimeError), - #[error("failed to evaluate udf: {0}")] - Evaluate(#[source] AnyError), - #[error("serialization: {0}")] - Serialization(#[from] SerializationError), - #[error("deserialization: {0}")] - Deserialization(#[from] DeserializationError), - #[error("serde json: {0}")] - SerdeJson(#[from] serde_json::Error), -} - -#[op2] -fn set_state(#[state] state: &Arc>, #[serde] new_state: JsonValue) { - *state.lock() = new_state; -} - -#[op2] -#[serde] -fn get_state(#[state] state: &Arc>) -> JsonValue { - state.lock().clone() -} - -impl Udf { - pub async fn new( - tokio_runtime: Arc, - function_name: String, - module: String, - arg: Expression, - ) -> Result { - let (deno_runtime, functions) = - dozer_deno::Runtime::new(vec![module], Vec:: Extension>::new()).await?; - let function = functions[0]; - Ok(Self { - function_name, - arg: Box::new(arg), - tokio_runtime, - deno_runtime: Arc::new(Mutex::new(deno_runtime)), - function, - }) - } - - pub fn get_type(&self) -> ExpressionType { - ExpressionType { - return_type: FieldType::Json, - nullable: false, - source: SourceDefinition::Dynamic, - is_primary_key: false, - } - } - - pub fn evaluate( - &mut self, - record: &Record, - schema: &Schema, - ) -> Result { - self.tokio_runtime.block_on(evaluate_impl( - self.function_name.clone(), - &mut self.arg, - &self.deno_runtime, - self.function, - record, - schema, - )) - } - - pub fn to_string(&self, schema: &Schema) -> String { - format!("{}({})", self.function_name, self.arg.to_string(schema)) - } -} - -async fn evaluate_impl( - function_name: String, - arg: &mut Expression, - runtime: &Arc>, - function: NonZeroI32, - record: &Record, - schema: &Schema, -) -> Result { - let arg = arg.evaluate(record, schema)?; - let Field::Json(arg) = arg else { - return Err(crate::error::Error::InvalidFunctionArgument { - function_name, - argument_index: 0, - argument: arg, - }); - }; - - let mut runtime = runtime.lock().await; - let result = runtime - .call_function(function, vec![arg]) - .await - .map_err(Error::Evaluate)?; - Ok(Field::Json(result)) -} diff --git a/dozer-sql/expression/src/javascript/mod.rs b/dozer-sql/expression/src/javascript/mod.rs deleted file mode 100644 index 10a48e1756..0000000000 --- a/dozer-sql/expression/src/javascript/mod.rs +++ /dev/null @@ -1,5 +0,0 @@ -mod evaluate; -mod validate; - -pub use evaluate::{Error, Udf}; -pub use validate::validate_args; diff --git a/dozer-sql/expression/src/javascript/validate.rs b/dozer-sql/expression/src/javascript/validate.rs deleted file mode 100644 index b6359ae78a..0000000000 --- a/dozer-sql/expression/src/javascript/validate.rs +++ /dev/null @@ -1,27 +0,0 @@ -use dozer_types::types::{FieldType, Schema}; - -use crate::{error::Error, execution::Expression}; - -pub fn validate_args( - function_name: String, - args: &[Expression], - schema: &Schema, -) -> Result<(), Error> { - if args.len() != 1 { - return Err(Error::InvalidNumberOfArguments { - function_name, - expected: 1..2, - actual: args.len(), - }); - } - let typ = args[0].get_type(schema)?; - if typ.return_type != FieldType::Json { - return Err(Error::InvalidFunctionArgumentType { - function_name, - argument_index: 0, - expected: vec![FieldType::Json], - actual: typ.return_type, - }); - } - Ok(()) -} diff --git a/dozer-sql/expression/src/json_functions.rs b/dozer-sql/expression/src/json_functions.rs deleted file mode 100644 index d7528bbfa1..0000000000 --- a/dozer-sql/expression/src/json_functions.rs +++ /dev/null @@ -1,118 +0,0 @@ -use crate::arg_utils::validate_num_arguments; -use crate::error::Error; -use crate::execution::Expression; - -use dozer_types::json_types::JsonValue; -use dozer_types::types::Record; -use dozer_types::types::{Field, Schema}; -use jsonpath::{JsonPathFinder, JsonPathInst}; -use std::fmt::{Display, Formatter}; -use std::str::FromStr; - -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Hash)] -pub enum JsonFunctionType { - JsonValue, - JsonQuery, -} - -impl Display for JsonFunctionType { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - match self { - JsonFunctionType::JsonValue => f.write_str("JSON_VALUE".to_string().as_str()), - JsonFunctionType::JsonQuery => f.write_str("JSON_QUERY".to_string().as_str()), - } - } -} - -impl JsonFunctionType { - pub(crate) fn new(name: &str) -> Option { - match name { - "json_value" => Some(JsonFunctionType::JsonValue), - "json_query" => Some(JsonFunctionType::JsonQuery), - _ => None, - } - } - - pub(crate) fn evaluate( - &self, - schema: &Schema, - args: &mut [Expression], - record: &Record, - ) -> Result { - match self { - JsonFunctionType::JsonValue => self.evaluate_json_value(schema, args, record), - JsonFunctionType::JsonQuery => self.evaluate_json_query(schema, args, record), - } - } - - pub(crate) fn evaluate_json_value( - &self, - schema: &Schema, - args: &mut [Expression], - record: &Record, - ) -> Result { - validate_num_arguments(2..3, args.len(), self)?; - let json_input = args[0].evaluate(record, schema)?; - let path = args[1].evaluate(record, schema)?.to_string(); - - if let Ok(json_value) = self.evaluate_json(json_input, path) { - if json_value.is_string() || json_value.is_number() || json_value.is_bool() { - return Ok(Field::Json(json_value)); - } - Ok(Field::Json(JsonValue::NULL)) - } else { - Ok(Field::Null) - } - } - - pub(crate) fn evaluate_json_query( - &self, - schema: &Schema, - args: &mut [Expression], - record: &Record, - ) -> Result { - validate_num_arguments(1..3, args.len(), self)?; - if args.len() == 1 { - Ok(Field::Json(self.evaluate_json( - args[0].evaluate(record, schema)?, - String::from("$"), - )?)) - } else { - let json_input = args[0].evaluate(record, schema)?; - let path = args[1].evaluate(record, schema)?.to_string(); - - if let Ok(json_value) = self.evaluate_json(json_input, path) { - if json_value.is_object() || json_value.is_array() { - return Ok(Field::Json(json_value)); - } - Ok(Field::Json(JsonValue::NULL)) - } else { - Ok(Field::Null) - } - } - } - - pub(crate) fn evaluate_json( - &self, - json_input: Field, - path: String, - ) -> Result { - let json_val = match json_input.to_json() { - Some(json) => json, - None => JsonValue::NULL, - }; - - let finder = JsonPathFinder::new( - Box::from(json_val), - Box::from(JsonPathInst::from_str(path.as_str()).map_err(Error::InvalidJsonPath)?), - ); - - let found = finder.find(); - if let Some(a) = found.as_array() { - if a.len() == 1 { - return Ok(a.first().unwrap().clone()); - } - } - Ok(found) - } -} diff --git a/dozer-sql/expression/src/lib.rs b/dozer-sql/expression/src/lib.rs deleted file mode 100644 index c94463f9d9..0000000000 --- a/dozer-sql/expression/src/lib.rs +++ /dev/null @@ -1,87 +0,0 @@ -pub mod aggregate; -mod arg_utils; -pub mod builder; -mod case; -mod cast; -mod comparison; -mod conditional; -mod datetime; -pub mod error; -pub mod execution; -mod geo; -mod in_list; -mod is_null; -mod json_functions; -mod logical; -mod mathematical; -pub mod operator; -pub mod scalar; - -#[cfg(feature = "javascript")] -mod javascript; -#[cfg(feature = "onnx")] -mod onnx; -#[cfg(feature = "python")] -mod python_udf; - -pub use num_traits; -pub use sqlparser; - -#[cfg(test)] -mod tests { - use dozer_types::{ - chrono::{DateTime, Datelike, FixedOffset, NaiveDate, NaiveDateTime, NaiveTime, Timelike}, - rust_decimal::Decimal, - }; - use proptest::{ - prelude::Arbitrary, - strategy::{BoxedStrategy, Strategy}, - }; - - #[derive(Debug)] - pub struct ArbitraryDecimal(pub Decimal); - - impl Arbitrary for ArbitraryDecimal { - type Parameters = (); - type Strategy = BoxedStrategy; - - fn arbitrary_with(_args: Self::Parameters) -> Self::Strategy { - (i64::MIN..i64::MAX, u32::MIN..29u32) - .prop_map(|(num, scale)| ArbitraryDecimal(Decimal::new(num, scale))) - .boxed() - } - } - - #[derive(Debug)] - pub struct ArbitraryDateTime(pub DateTime); - - impl Arbitrary for ArbitraryDateTime { - type Parameters = (); - type Strategy = BoxedStrategy; - - fn arbitrary_with(_args: Self::Parameters) -> Self::Strategy { - ( - NaiveDateTime::MIN.year()..NaiveDateTime::MAX.year(), - 1..13u32, - 1..32u32, - 0..NaiveDateTime::MAX.second(), - 0..NaiveDateTime::MAX.nanosecond(), - ) - .prop_map(|(year, month, day, secs, nano)| { - let timezone_east = FixedOffset::east_opt(8 * 60 * 60).unwrap(); - let date = NaiveDate::from_ymd_opt(year, month, day); - // Some dates are not able to created caused by leap in February with day larger than 28 or 29 - if date.is_none() { - return ArbitraryDateTime(DateTime::default()); - } - let time = NaiveTime::from_num_seconds_from_midnight_opt(secs, nano).unwrap(); - let datetime = DateTime::::from_naive_utc_and_offset( - NaiveDateTime::new(date.unwrap(), time), - timezone_east, - ); - ArbitraryDateTime(datetime) - }) - .boxed() - } - } -} diff --git a/dozer-sql/expression/src/logical.rs b/dozer-sql/expression/src/logical.rs deleted file mode 100644 index 63a9017840..0000000000 --- a/dozer-sql/expression/src/logical.rs +++ /dev/null @@ -1,303 +0,0 @@ -use dozer_types::types::Record; -use dozer_types::types::{Field, Schema}; - -use crate::error::Error; -use crate::execution::Expression; - -pub fn evaluate_and( - schema: &Schema, - left: &mut Expression, - right: &mut Expression, - record: &Record, -) -> Result { - let l_field = left.evaluate(record, schema)?; - let r_field = right.evaluate(record, schema)?; - match l_field { - Field::Boolean(true) => match r_field { - Field::Boolean(true) => Ok(Field::Boolean(true)), - Field::Boolean(false) => Ok(Field::Boolean(false)), - Field::Null => Ok(Field::Boolean(false)), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(Error::InvalidType(r_field, "AND".to_string())), - }, - Field::Boolean(false) => match r_field { - Field::Boolean(true) => Ok(Field::Boolean(false)), - Field::Boolean(false) => Ok(Field::Boolean(false)), - Field::Null => Ok(Field::Boolean(false)), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(Error::InvalidType(r_field, "AND".to_string())), - }, - Field::Null => Ok(Field::Boolean(false)), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(Error::InvalidType(l_field, "AND".to_string())), - } -} - -pub fn evaluate_or( - schema: &Schema, - left: &mut Expression, - right: &mut Expression, - record: &Record, -) -> Result { - let l_field = left.evaluate(record, schema)?; - let r_field = right.evaluate(record, schema)?; - match l_field { - Field::Boolean(true) => match r_field { - Field::Boolean(false) => Ok(Field::Boolean(true)), - Field::Boolean(true) => Ok(Field::Boolean(true)), - Field::Null => Ok(Field::Boolean(true)), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(Error::InvalidType(r_field, "OR".to_string())), - }, - Field::Boolean(false) | Field::Null => match right.evaluate(record, schema)? { - Field::Boolean(false) => Ok(Field::Boolean(false)), - Field::Boolean(true) => Ok(Field::Boolean(true)), - Field::Null => Ok(Field::Boolean(false)), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(Error::InvalidType(r_field, "OR".to_string())), - }, - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(Error::InvalidType(l_field, "OR".to_string())), - } -} - -pub fn evaluate_not( - schema: &Schema, - value: &mut Expression, - record: &Record, -) -> Result { - let value_p = value.evaluate(record, schema)?; - - match value_p { - Field::Boolean(value_v) => Ok(Field::Boolean(!value_v)), - Field::Null => Ok(Field::Null), - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(Error::InvalidType(value_p, "NOT".to_string())), - } -} - -#[cfg(test)] -mod tests { - use super::*; - - use dozer_types::types::Record; - use dozer_types::types::{Field, Schema}; - use dozer_types::{ordered_float::OrderedFloat, rust_decimal::Decimal}; - use proptest::prelude::*; - use Expression::Literal; - - #[test] - fn test_logical() { - proptest!( - ProptestConfig::with_cases(1000), - move |(bool1: bool, bool2: bool, u_num: u64, i_num: i64, f_num: f64, str in ".*")| { - _test_bool_bool_and(bool1, bool2); - _test_bool_null_and(Field::Boolean(bool1), Field::Null); - _test_bool_null_and(Field::Null, Field::Boolean(bool1)); - - _test_bool_bool_or(bool1, bool2); - _test_bool_null_or(bool1); - _test_null_bool_or(bool2); - - _test_bool_not(bool2); - - _test_bool_non_bool_and(Field::UInt(u_num), Field::Boolean(bool1)); - _test_bool_non_bool_and(Field::Int(i_num), Field::Boolean(bool1)); - _test_bool_non_bool_and(Field::Float(OrderedFloat(f_num)), Field::Boolean(bool1)); - _test_bool_non_bool_and(Field::Decimal(Decimal::from(u_num)), Field::Boolean(bool1)); - _test_bool_non_bool_and(Field::String(str.clone()), Field::Boolean(bool1)); - _test_bool_non_bool_and(Field::Text(str.clone()), Field::Boolean(bool1)); - - _test_bool_non_bool_and(Field::Boolean(bool2), Field::UInt(u_num)); - _test_bool_non_bool_and(Field::Boolean(bool2), Field::Int(i_num)); - _test_bool_non_bool_and(Field::Boolean(bool2), Field::Float(OrderedFloat(f_num))); - _test_bool_non_bool_and(Field::Boolean(bool2), Field::Decimal(Decimal::from(u_num))); - _test_bool_non_bool_and(Field::Boolean(bool2), Field::String(str.clone())); - _test_bool_non_bool_and(Field::Boolean(bool2), Field::Text(str.clone())); - - _test_bool_non_bool_or(Field::UInt(u_num), Field::Boolean(bool1)); - _test_bool_non_bool_or(Field::Int(i_num), Field::Boolean(bool1)); - _test_bool_non_bool_or(Field::Float(OrderedFloat(f_num)), Field::Boolean(bool1)); - _test_bool_non_bool_or(Field::Decimal(Decimal::from(u_num)), Field::Boolean(bool1)); - _test_bool_non_bool_or(Field::String(str.clone()), Field::Boolean(bool1)); - _test_bool_non_bool_or(Field::Text(str.clone()), Field::Boolean(bool1)); - - _test_bool_non_bool_or(Field::Boolean(bool2), Field::UInt(u_num)); - _test_bool_non_bool_or(Field::Boolean(bool2), Field::Int(i_num)); - _test_bool_non_bool_or(Field::Boolean(bool2), Field::Float(OrderedFloat(f_num))); - _test_bool_non_bool_or(Field::Boolean(bool2), Field::Decimal(Decimal::from(u_num))); - _test_bool_non_bool_or(Field::Boolean(bool2), Field::String(str.clone())); - _test_bool_non_bool_or(Field::Boolean(bool2), Field::Text(str)); - }); - } - - fn _test_bool_bool_and(bool1: bool, bool2: bool) { - let row = Record::new(vec![]); - let mut l = Box::new(Literal(Field::Boolean(bool1))); - let mut r = Box::new(Literal(Field::Boolean(bool2))); - assert!(matches!( - evaluate_and(&Schema::default(), &mut l, &mut r, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Boolean(_ans) - )); - } - - fn _test_bool_null_and(f1: Field, f2: Field) { - let row = Record::new(vec![]); - let mut l = Box::new(Literal(f1)); - let mut r = Box::new(Literal(f2)); - assert!(matches!( - evaluate_and(&Schema::default(), &mut l, &mut r, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Boolean(false) - )); - } - - fn _test_bool_bool_or(bool1: bool, bool2: bool) { - let row = Record::new(vec![]); - let mut l = Box::new(Literal(Field::Boolean(bool1))); - let mut r = Box::new(Literal(Field::Boolean(bool2))); - assert!(matches!( - evaluate_or(&Schema::default(), &mut l, &mut r, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Boolean(_ans) - )); - } - - fn _test_bool_null_or(_bool: bool) { - let row = Record::new(vec![]); - let mut l = Box::new(Literal(Field::Boolean(_bool))); - let mut r = Box::new(Literal(Field::Null)); - assert!(matches!( - evaluate_or(&Schema::default(), &mut l, &mut r, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Boolean(_bool) - )); - } - - fn _test_null_bool_or(_bool: bool) { - let row = Record::new(vec![]); - let mut l = Box::new(Literal(Field::Null)); - let mut r = Box::new(Literal(Field::Boolean(_bool))); - assert!(matches!( - evaluate_or(&Schema::default(), &mut l, &mut r, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Boolean(_bool) - )); - } - - fn _test_bool_not(bool: bool) { - let row = Record::new(vec![]); - let mut v = Box::new(Literal(Field::Boolean(bool))); - assert!(matches!( - evaluate_not(&Schema::default(), &mut v, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Boolean(_ans) - )); - } - - fn _test_bool_non_bool_and(f1: Field, f2: Field) { - let row = Record::new(vec![]); - let mut l = Box::new(Literal(f1)); - let mut r = Box::new(Literal(f2)); - assert!(evaluate_and(&Schema::default(), &mut l, &mut r, &row).is_err()); - } - - fn _test_bool_non_bool_or(f1: Field, f2: Field) { - let row = Record::new(vec![]); - let mut l = Box::new(Literal(f1)); - let mut r = Box::new(Literal(f2)); - assert!(evaluate_or(&Schema::default(), &mut l, &mut r, &row).is_err()); - } -} diff --git a/dozer-sql/expression/src/mathematical/mod.rs b/dozer-sql/expression/src/mathematical/mod.rs deleted file mode 100644 index fa084be15e..0000000000 --- a/dozer-sql/expression/src/mathematical/mod.rs +++ /dev/null @@ -1,2372 +0,0 @@ -use dozer_types::rust_decimal::Decimal; -use dozer_types::types::Record; -use dozer_types::types::Schema; -use dozer_types::types::{DozerDuration, TimeUnit}; -use dozer_types::{chrono, ordered_float::OrderedFloat, types::Field}; -use num_traits::{FromPrimitive, Zero}; -use std::num::Wrapping; -use std::ops::Neg; - -use crate::execution::Expression; - -use crate::error::{Error as PipelineError, OperationError}; - -macro_rules! define_math_operator { - ($id:ident, $op:expr, $fct:expr, $t: expr) => { - pub fn $id( - schema: &Schema, - left: &mut Expression, - right: &mut Expression, - record: &Record, - ) -> Result { - let left_p = left.evaluate(&record, schema)?; - let right_p = right.evaluate(&record, schema)?; - - match left_p { - Field::Duration(left_v) => match right_p { - Field::Duration(right_v) => { - match $op { - "-" => { - let duration = left_v.0.checked_sub(right_v.0).ok_or( - PipelineError::SqlError(OperationError::AdditionOverflow), - )?; - Ok(Field::from(DozerDuration(duration, TimeUnit::Nanoseconds))) - } - "+" => { - let duration = left_v.0.checked_add(right_v.0).ok_or( - PipelineError::SqlError(OperationError::SubtractionOverflow), - )?; - Ok(Field::from(DozerDuration(duration, TimeUnit::Nanoseconds))) - } - "*" | "/" | "%" => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - } - } - Field::Timestamp(right_v) => match $op { - "+" => { - let duration = right_v - .checked_add_signed(chrono::Duration::nanoseconds( - left_v.0.as_nanos() as i64, - )) - .ok_or(PipelineError::SqlError(OperationError::AdditionOverflow))?; - Ok(Field::Timestamp(duration)) - } - "-" | "*" | "/" | "%" => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Null => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::Timestamp(left_v) => match right_p { - Field::Duration(right_v) => match $op { - "-" => { - let duration = left_v - .checked_sub_signed(chrono::Duration::nanoseconds( - right_v.0.as_nanos() as i64, - )) - .ok_or(PipelineError::SqlError(OperationError::AdditionOverflow))?; - Ok(Field::Timestamp(duration)) - } - "+" => { - let duration = left_v - .checked_add_signed(chrono::Duration::nanoseconds( - right_v.0.as_nanos() as i64, - )) - .ok_or(PipelineError::SqlError( - OperationError::SubtractionOverflow, - ))?; - Ok(Field::Timestamp(duration)) - } - "*" | "/" | "%" => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::Timestamp(right_v) => match $op { - "-" => { - if left_v > right_v { - let duration: i64 = (left_v - right_v).num_nanoseconds().ok_or( - PipelineError::UnableToCast( - format!("{}", left_v - right_v), - "i64".to_string(), - ), - )?; - Ok(Field::from(DozerDuration( - std::time::Duration::from_nanos(duration as u64), - TimeUnit::Nanoseconds, - ))) - } else { - Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )) - } - } - "+" | "*" | "/" | "%" => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::UInt(_) - | Field::U128(_) - | Field::Int(_) - | Field::Int8(_) - | Field::I128(_) - | Field::Float(_) - | Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Decimal(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Null => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::Float(left_v) => match right_p { - // left: Float, right: Int - Field::Int(right_v) => { - return match $op { - "/" | "%" => { - if right_v == 0_i64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - left_v, - OrderedFloat::::from_i64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - &_ => Ok(Field::Float($fct( - left_v, - OrderedFloat::::from_i64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))), - } - } - - Field::Int8(v) => { - let right_v = v as i64; - return match $op { - "/" | "%" => { - if right_v == 0_i64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - left_v, - OrderedFloat::::from_i64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - &_ => Ok(Field::Float($fct( - left_v, - OrderedFloat::::from_i64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))), - }; - } - - // left: Float, right: I128 - Field::I128(right_v) => { - return match $op { - "/" | "%" => { - if right_v == 0_i128 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - left_v, - OrderedFloat::::from_i128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - &_ => Ok(Field::Float($fct( - left_v, - OrderedFloat::::from_i128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))), - } - } - // left: Float, right: UInt - Field::UInt(right_v) => { - return match $op { - "/" | "%" => { - if right_v == 0_u64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - left_v, - OrderedFloat::::from_u64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - &_ => Ok(Field::Float($fct( - left_v, - OrderedFloat::::from_u64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))), - } - } - // left: Float, right: U128 - Field::U128(right_v) => { - return match $op { - "/" | "%" => { - if right_v == 0_u128 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - left_v, - OrderedFloat::::from_u128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - &_ => Ok(Field::Float($fct( - left_v, - OrderedFloat::::from_u128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))), - } - } - // left: Float, right: Float - Field::Float(right_v) => { - return match $op { - "/" | "%" => { - if right_v == 0_f64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct(left_v, right_v))) - } - } - &_ => Ok(Field::Float($fct(left_v, right_v))), - } - } - // left: Float, right: Decimal - Field::Decimal(right_v) => { - return match $op { - "/" => Ok(Field::Decimal( - Decimal::from_f64(*left_v) - .ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))? - .checked_div(right_v) - .ok_or(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - ))?, - )), - "%" => Ok(Field::Decimal( - Decimal::from_f64(*left_v) - .ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))? - .checked_rem(right_v) - .ok_or(PipelineError::SqlError( - OperationError::ModuloByZeroOrOverflow, - ))?, - )), - "*" => Ok(Field::Decimal( - Decimal::from_f64(*left_v) - .ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))? - .checked_mul(right_v) - .ok_or(PipelineError::SqlError( - OperationError::MultiplicationOverflow, - ))?, - )), - "+" | "-" => Ok(Field::Decimal($fct( - Decimal::from_f64(*left_v).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?, - right_v, - ))), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - } - } - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::Int(left_v) => match right_p { - // left: Int, right: Int - Field::Int(right_v) => { - return match $op { - // When Int / Int division happens - "/" => { - if right_v == 0_i64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - OrderedFloat::::from_i64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => { - Ok(Field::Int($fct(Wrapping(left_v), Wrapping(right_v)).0)) - } - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - - Field::Int8(v) => { - let right_v = v as i64; - return match $op { - // When Int / Int division happens - "/" => { - if right_v == 0_i64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - OrderedFloat::::from_i64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => { - Ok(Field::Int($fct(Wrapping(left_v), Wrapping(right_v)).0)) - } - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - - // left: Int, right: I128 - Field::I128(right_v) => { - return match $op { - // When Int / I128 division happens - "/" => { - if right_v == 0_i128 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - OrderedFloat::::from_i128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::I128( - $fct(Wrapping(left_v as i128), Wrapping(right_v)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // left: Int, right: UInt - Field::UInt(right_v) => { - return match $op { - // When Int / UInt division happens - "/" => { - if right_v == 0_u64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - OrderedFloat::::from_u64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::Int( - $fct(Wrapping(left_v), Wrapping(right_v as i64)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // left: Int, right: U128 - Field::U128(right_v) => { - return match $op { - // When Int / U128 division happens - "/" => { - if right_v == 0_u128 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - OrderedFloat::::from_u128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::I128( - $fct(Wrapping(left_v as i128), Wrapping(right_v as i128)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // left: Int, right: Float - Field::Float(right_v) => { - return match $op { - "/" | "%" => { - if right_v == 0_f64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - right_v, - ))) - } - } - &_ => Ok(Field::Float($fct( - OrderedFloat::::from_i64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - right_v, - ))), - } - } - - // left: Int, right: Decimal - Field::Decimal(right_v) => { - return match $op { - "/" => Ok(Field::Decimal( - Decimal::from_i64(left_v) - .ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))? - .checked_div(right_v) - .ok_or(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - ))?, - )), - "%" => Ok(Field::Decimal( - Decimal::from_i64(left_v) - .ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))? - .checked_rem(right_v) - .ok_or(PipelineError::SqlError( - OperationError::ModuloByZeroOrOverflow, - ))?, - )), - "*" => Ok(Field::Decimal( - Decimal::from_i64(left_v) - .ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))? - .checked_mul(right_v) - .ok_or(PipelineError::SqlError( - OperationError::MultiplicationOverflow, - ))?, - )), - "+" | "-" => Ok(Field::Decimal($fct( - Decimal::from_i64(left_v).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?, - right_v, - ))), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - } - } - // left: Int, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::Int8(v) => { - let left_v = v as i64; - return match right_p { - // left: Int, right: Int - Field::Int(right_v) => { - return match $op { - // When Int / Int division happens - "/" => { - if right_v == 0_i64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - OrderedFloat::::from_i64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => { - Ok(Field::Int($fct(Wrapping(left_v), Wrapping(right_v)).0)) - } - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - - Field::Int8(v) => { - let right_v = v as i64; - return match $op { - // When Int / Int division happens - "/" => { - if right_v == 0_i64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - OrderedFloat::::from_i64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => { - Ok(Field::Int($fct(Wrapping(left_v), Wrapping(right_v)).0)) - } - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - - // left: Int, right: I128 - Field::I128(right_v) => { - return match $op { - // When Int / I128 division happens - "/" => { - if right_v == 0_i128 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - OrderedFloat::::from_i128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::I128( - $fct(Wrapping(left_v as i128), Wrapping(right_v)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // left: Int, right: UInt - Field::UInt(right_v) => { - return match $op { - // When Int / UInt division happens - "/" => { - if right_v == 0_u64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - OrderedFloat::::from_u64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::Int( - $fct(Wrapping(left_v), Wrapping(right_v as i64)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // left: Int, right: U128 - Field::U128(right_v) => { - return match $op { - // When Int / U128 division happens - "/" => { - if right_v == 0_u128 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - OrderedFloat::::from_u128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::I128( - $fct(Wrapping(left_v as i128), Wrapping(right_v as i128)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // left: Int, right: Float - Field::Float(right_v) => { - return match $op { - "/" | "%" => { - if right_v == 0_f64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - right_v, - ))) - } - } - &_ => Ok(Field::Float($fct( - OrderedFloat::::from_i64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - right_v, - ))), - } - } - - // left: Int, right: Decimal - Field::Decimal(right_v) => { - return match $op { - "/" => Ok(Field::Decimal( - Decimal::from_i64(left_v) - .ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))? - .checked_div(right_v) - .ok_or(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - ))?, - )), - "%" => Ok(Field::Decimal( - Decimal::from_i64(left_v) - .ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))? - .checked_rem(right_v) - .ok_or(PipelineError::SqlError( - OperationError::ModuloByZeroOrOverflow, - ))?, - )), - "*" => Ok(Field::Decimal( - Decimal::from_i64(left_v) - .ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))? - .checked_mul(right_v) - .ok_or(PipelineError::SqlError( - OperationError::MultiplicationOverflow, - ))?, - )), - "+" | "-" => Ok(Field::Decimal($fct( - Decimal::from_i64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ), - )?, - right_v, - ))), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - } - } - // left: Int, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - Field::I128(left_v) => match right_p { - // left: I128, right: Int - Field::Int(right_v) => { - return match $op { - // When I128 / Int division happens - "/" => { - if right_v == 0_i64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i128(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - OrderedFloat::::from_i64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::I128( - $fct(Wrapping(left_v), Wrapping(right_v as i128)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - Field::Int8(right_v) => { - return match $op { - // When I128 / Int division happens - "/" => { - if right_v == 0_i8 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i128(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - OrderedFloat::::from_i64(right_v as i64).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::I128( - $fct(Wrapping(left_v), Wrapping(right_v as i128)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // left: I128, right: I128 - Field::I128(right_v) => { - return match $op { - // When I128 / I128 division happens - "/" => { - if right_v == 0_i128 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i128(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - OrderedFloat::::from_i128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => { - Ok(Field::I128($fct(Wrapping(left_v), Wrapping(right_v)).0)) - } - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // left: I128, right: UInt - Field::UInt(right_v) => { - return match $op { - // When I128 / UInt division happens - "/" => { - if right_v == 0_u64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i128(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - OrderedFloat::::from_u64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::I128( - $fct(Wrapping(left_v), Wrapping(right_v as i128)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // left: I128, right: U128 - Field::U128(right_v) => { - return match $op { - // When Int / U128 division happens - "/" => { - if right_v == 0_u128 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i128(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - OrderedFloat::::from_u128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - ))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::I128( - $fct(Wrapping(left_v), Wrapping(right_v as i128)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // left: I128, right: Float - Field::Float(right_v) => { - return match $op { - "/" | "%" => { - if right_v == 0_f64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_i128(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - right_v, - ))) - } - } - &_ => Ok(Field::Float($fct( - OrderedFloat::::from_i128(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - right_v, - ))), - } - } - // left: I128, right: Decimal - Field::Decimal(right_v) => { - return match $op { - "/" | "%" => { - if right_v == dozer_types::rust_decimal::Decimal::zero() { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Decimal($fct( - Decimal::from_i128(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ), - )?, - right_v, - ))) - } - } - &_ => Ok(Field::Decimal($fct( - Decimal::from_i128(left_v).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?, - right_v, - ))), - } - } - // left: I128, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::UInt(left_v) => match right_p { - // left: UInt, right: Int - Field::Int(right_v) => { - return match $op { - // When UInt / Int division happens - "/" => { - if right_v == 0_i64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float(OrderedFloat($fct( - f64::from_u64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - f64::from_i64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - )))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::Int( - $fct(Wrapping(left_v as i64), Wrapping(right_v)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - Field::Int8(right_v) => { - return match $op { - // When UInt / Int division happens - "/" => { - if right_v == 0_i8 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float(OrderedFloat($fct( - f64::from_u64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - f64::from_i64(right_v as i64).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - )))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::Int( - $fct(Wrapping(left_v as i64), Wrapping(right_v as i64)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - - // left: UInt, right: I128 - Field::I128(right_v) => { - return match $op { - // When UInt / I128 division happens - "/" => { - if right_v == 0_i128 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float(OrderedFloat($fct( - f64::from_u64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - f64::from_i128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - )))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::I128( - $fct(Wrapping(left_v as i128), Wrapping(right_v)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // left: UInt, right: UInt - Field::UInt(right_v) => { - return match $op { - // When UInt / UInt division happens - "/" => { - if right_v == 0_u64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float( - OrderedFloat::::from_f64($fct( - f64::from_u64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - f64::from_u64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - )) - .ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "OrderedFloat".to_string(), - ), - )?, - )) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => { - Ok(Field::UInt($fct(Wrapping(left_v), Wrapping(right_v)).0)) - } - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // left: UInt, right: U128 - Field::U128(right_v) => { - return match $op { - // When UInt / UInt division happens - "/" => { - if right_v == 0_u128 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float( - OrderedFloat::::from_f64($fct( - f64::from_u64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - f64::from_u128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - )) - .ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "OrderedFloat".to_string(), - ), - )?, - )) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::U128( - $fct(Wrapping(left_v as u128), Wrapping(right_v)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // left: UInt, right: Float - Field::Float(right_v) => { - return match $op { - "/" | "%" => { - if right_v == 0_f64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_u64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - right_v, - ))) - } - } - &_ => Ok(Field::Float($fct( - OrderedFloat::::from_u64(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - right_v, - ))), - } - } - // left: UInt, right: Decimal - Field::Decimal(right_v) => { - return match $op { - "/" => { - if right_v == dozer_types::rust_decimal::Decimal::zero() { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Decimal( - Decimal::from_u64(left_v) - .ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))? - .checked_div(right_v) - .ok_or(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - ))?, - )) - } - } - "%" => Ok(Field::Decimal( - Decimal::from_u64(left_v) - .ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))? - .checked_rem(right_v) - .ok_or(PipelineError::SqlError( - OperationError::ModuloByZeroOrOverflow, - ))?, - )), - "*" => Ok(Field::Decimal( - Decimal::from_u64(left_v) - .ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))? - .checked_mul(right_v) - .ok_or(PipelineError::SqlError( - OperationError::MultiplicationOverflow, - ))?, - )), - "+" | "-" => Ok(Field::Decimal($fct( - Decimal::from_u64(left_v).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?, - right_v, - ))), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - } - } - // left: UInt, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::U128(left_v) => match right_p { - // left: U128, right: Int - Field::Int(right_v) => { - return match $op { - // When U128 / Int division happens - "/" => { - if right_v == 0_i64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float(OrderedFloat($fct( - f64::from_u128(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - f64::from_i64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - )))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::I128( - $fct(Wrapping(left_v as i128), Wrapping(right_v as i128)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - - Field::Int8(right_v) => { - return match $op { - // When U128 / Int division happens - "/" => { - if right_v == 0_i8 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float(OrderedFloat($fct( - f64::from_u128(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - f64::from_i64(right_v as i64).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - )))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::I128( - $fct(Wrapping(left_v as i128), Wrapping(right_v as i128)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - - // left: U128, right: I128 - Field::I128(right_v) => { - return match $op { - // When U128 / I128 division happens - "/" => { - if right_v == 0_i128 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float(OrderedFloat($fct( - f64::from_u128(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - f64::from_i128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - )))) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::I128( - $fct(Wrapping(left_v as i128), Wrapping(right_v)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // left: U128, right: UInt - Field::UInt(right_v) => { - return match $op { - // When U128 / UInt division happens - "/" => { - if right_v == 0_u64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float( - OrderedFloat::::from_f64($fct( - f64::from_u128(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - f64::from_u64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - )) - .ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "OrderedFloat".to_string(), - ), - )?, - )) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => Ok(Field::U128( - $fct(Wrapping(left_v), Wrapping(right_v as u128)).0, - )), - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // left: U128, right: U128 - Field::U128(right_v) => { - return match $op { - // When U128 / U128 division happens - "/" => { - if right_v == 0_u128 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float( - OrderedFloat::::from_f64($fct( - f64::from_u128(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - f64::from_u128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "f64".to_string(), - ), - )?, - )) - .ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "OrderedFloat".to_string(), - ), - )?, - )) - } - } - // When it's not division operation - "+" | "-" | "*" | "%" => { - Ok(Field::U128($fct(Wrapping(left_v), Wrapping(right_v)).0)) - } - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // left: U128, right: Float - Field::Float(right_v) => { - return match $op { - "/" | "%" => { - if right_v == 0_f64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Float($fct( - OrderedFloat::::from_u128(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - right_v, - ))) - } - } - &_ => Ok(Field::Float($fct( - OrderedFloat::::from_u128(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "f64".to_string(), - ), - )?, - right_v, - ))), - } - } - // left: U128, right: Decimal - Field::Decimal(right_v) => { - return match $op { - "/" | "%" => { - if right_v == dozer_types::rust_decimal::Decimal::zero() { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Decimal($fct( - Decimal::from_u128(left_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ), - )?, - right_v, - ))) - } - } - &_ => Ok(Field::Decimal($fct( - Decimal::from_u128(left_v).ok_or(PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ))?, - right_v, - ))), - } - } - // left: U128, right: Null - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }, - Field::Decimal(left_v) => { - return match $op { - "/" => { - match right_p { - // left: Decimal, right: Int - Field::Int(right_v) => { - if right_v == 0_i64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Decimal( - left_v - .checked_div(Decimal::from_i64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - ))?, - )) - } - } - Field::Int8(v) => { - let right_v = v as i64; - if right_v == 0_i64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Decimal( - left_v - .checked_div(Decimal::from_i64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - ))?, - )) - } - } - - // left: Decimal, right: I128 - Field::I128(right_v) => { - if right_v == 0_i128 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Decimal( - left_v - .checked_div(Decimal::from_i128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - ))?, - )) - } - } - // left: Decimal, right: UInt - Field::UInt(right_v) => { - if right_v == 0_u64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Decimal( - left_v - .checked_div(Decimal::from_u64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - ))?, - )) - } - } - // left: Decimal, right: U128 - Field::U128(right_v) => { - if right_v == 0_u128 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Decimal( - left_v - .checked_div(Decimal::from_u128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - ))?, - )) - } - } - // left: Decimal, right: Float - Field::Float(right_v) => { - if right_v == 0_f64 { - Err(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - )) - } else { - Ok(Field::Decimal( - left_v - .checked_div(Decimal::from_f64(*right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - ))?, - )) - } - } - // left: Decimal, right: Null - Field::Null => Ok(Field::Null), - // left: Decimal, right: Decimal - Field::Decimal(right_v) => Ok(Field::Decimal( - left_v.checked_div(right_v).ok_or(PipelineError::SqlError( - OperationError::DivisionByZeroOrOverflow, - ))?, - )), - Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - } - } - "%" => { - match right_p { - // left: Decimal, right: Int - Field::Int(right_v) => Ok(Field::Decimal( - left_v - .checked_rem(Decimal::from_i64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::ModuloByZeroOrOverflow, - ))?, - )), - Field::Int8(right_v) => Ok(Field::Decimal( - left_v - .checked_rem(Decimal::from_i64(right_v as i64).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::ModuloByZeroOrOverflow, - ))?, - )), - // left: Decimal, right: I128 - Field::I128(right_v) => Ok(Field::Decimal( - left_v - .checked_rem(Decimal::from_i128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::ModuloByZeroOrOverflow, - ))?, - )), - // left: Decimal, right: UInt - Field::UInt(right_v) => Ok(Field::Decimal( - left_v - .checked_rem(Decimal::from_u64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::ModuloByZeroOrOverflow, - ))?, - )), - // left: Decimal, right: U128 - Field::U128(right_v) => Ok(Field::Decimal( - left_v - .checked_rem(Decimal::from_u128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::ModuloByZeroOrOverflow, - ))?, - )), - // left: Decimal, right: Float - Field::Float(right_v) => Ok(Field::Decimal( - left_v - .checked_rem(Decimal::from_f64(*right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::ModuloByZeroOrOverflow, - ))?, - )), - // left: Decimal, right: Null - Field::Null => Ok(Field::Null), - // left: Decimal, right: Decimal - Field::Decimal(right_v) => Ok(Field::Decimal( - left_v.checked_rem(right_v).ok_or(PipelineError::SqlError( - OperationError::ModuloByZeroOrOverflow, - ))?, - )), - Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - } - } - "*" => { - match right_p { - // left: Decimal, right: Int - Field::Int(right_v) => Ok(Field::Decimal( - left_v - .checked_mul(Decimal::from_i64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::MultiplicationOverflow, - ))?, - )), - Field::Int8(right_v) => Ok(Field::Decimal( - left_v - .checked_mul(Decimal::from_i64(right_v as i64).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::MultiplicationOverflow, - ))?, - )), - // left: Decimal, right: I128 - Field::I128(right_v) => Ok(Field::Decimal( - left_v - .checked_mul(Decimal::from_i128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::MultiplicationOverflow, - ))?, - )), - // left: Decimal, right: UInt - Field::UInt(right_v) => Ok(Field::Decimal( - left_v - .checked_mul(Decimal::from_u64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::MultiplicationOverflow, - ))?, - )), - // left: Decimal, right: U128 - Field::U128(right_v) => Ok(Field::Decimal( - left_v - .checked_mul(Decimal::from_u128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::MultiplicationOverflow, - ))?, - )), - // left: Decimal, right: Float - Field::Float(right_v) => Ok(Field::Decimal( - left_v - .checked_mul(Decimal::from_f64(*right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ), - )?) - .ok_or(PipelineError::SqlError( - OperationError::MultiplicationOverflow, - ))?, - )), - // left: Decimal, right: Null - Field::Null => Ok(Field::Null), - // left: Decimal, right: Decimal - Field::Decimal(right_v) => Ok(Field::Decimal( - left_v.checked_mul(right_v).ok_or(PipelineError::SqlError( - OperationError::MultiplicationOverflow, - ))?, - )), - Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - } - } - "+" | "-" => { - match right_p { - // left: Decimal, right: Int - Field::Int(right_v) => Ok(Field::Decimal($fct( - left_v, - Decimal::from_i64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ), - )?, - ))), - Field::Int8(right_v) => Ok(Field::Decimal($fct( - left_v, - Decimal::from_i64(right_v as i64).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ), - )?, - ))), - // left: Decimal, right: I128 - Field::I128(right_v) => Ok(Field::Decimal($fct( - left_v, - Decimal::from_i128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", left_v), - "Decimal".to_string(), - ), - )?, - ))), - // left: Decimal, right: UInt - Field::UInt(right_v) => Ok(Field::Decimal($fct( - left_v, - Decimal::from_u64(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ), - )?, - ))), - // left: Decimal, right: U128 - Field::U128(right_v) => Ok(Field::Decimal($fct( - left_v, - Decimal::from_u128(right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ), - )?, - ))), - // left: Decimal, right: Float - Field::Float(right_v) => Ok(Field::Decimal($fct( - left_v, - Decimal::from_f64(*right_v).ok_or( - PipelineError::UnableToCast( - format!("{}", right_v), - "Decimal".to_string(), - ), - )?, - ))), - // left: Decimal, right: Null - Field::Null => Ok(Field::Null), - // left: Decimal, right: Decimal - Field::Decimal(right_v) => { - Ok(Field::Decimal($fct(left_v, right_v))) - } - Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - } - } - &_ => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - }; - } - // right: Null, right: * - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) => Err(PipelineError::InvalidTypeComparison( - left_p, - right_p, - $op.to_string(), - )), - } - } - }; -} - -define_math_operator!(evaluate_add, "+", std::ops::Add::add, 0); -define_math_operator!(evaluate_sub, "-", std::ops::Sub::sub, 0); -define_math_operator!(evaluate_mul, "*", std::ops::Mul::mul, 0); -define_math_operator!(evaluate_div, "/", std::ops::Div::div, 1); -define_math_operator!(evaluate_mod, "%", std::ops::Rem::rem, 0); - -pub fn evaluate_plus( - schema: &Schema, - expression: &mut Expression, - record: &Record, -) -> Result { - let expression_result = expression.evaluate(record, schema)?; - match expression_result { - Field::UInt(v) => Ok(Field::UInt(v)), - Field::U128(v) => Ok(Field::U128(v)), - Field::Int(v) => Ok(Field::Int(v)), - Field::Int8(v) => Ok(Field::Int8(v)), - - Field::I128(v) => Ok(Field::I128(v)), - Field::Float(v) => Ok(Field::Float(v)), - Field::Decimal(v) => Ok(Field::Decimal(v)), - Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) - | Field::Null => Err(PipelineError::InvalidType( - expression_result, - "+".to_string(), - )), - } -} - -pub fn evaluate_minus( - schema: &Schema, - expression: &mut Expression, - record: &Record, -) -> Result { - let expression_result = expression.evaluate(record, schema)?; - match expression_result { - Field::UInt(v) => Ok(Field::UInt(v)), - Field::U128(v) => Ok(Field::U128(v)), - Field::Int(v) => Ok(Field::Int(-v)), - Field::Int8(v) => Ok(Field::Int8(-v)), - Field::I128(v) => Ok(Field::I128(-v)), - Field::Float(v) => Ok(Field::Float(-v)), - Field::Decimal(v) => Ok(Field::Decimal(v.neg())), - Field::Timestamp(dt) => Ok(Field::Timestamp(dt.checked_add_signed(chrono::Duration::nanoseconds(1)).unwrap())), - Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Binary(_) - // | Field::Timestamp(_) - | Field::Date(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) - | Field::Null => Err(PipelineError::InvalidType( - expression_result, - "-".to_string(), - )), - } -} - -#[cfg(test)] -mod tests; diff --git a/dozer-sql/expression/src/mathematical/tests.rs b/dozer-sql/expression/src/mathematical/tests.rs deleted file mode 100644 index 87ce450ceb..0000000000 --- a/dozer-sql/expression/src/mathematical/tests.rs +++ /dev/null @@ -1,2459 +0,0 @@ -use crate::tests::{ArbitraryDateTime, ArbitraryDecimal}; - -use super::*; - -use dozer_types::chrono::DateTime; -use dozer_types::types::{FieldDefinition, FieldType, Record, SourceDefinition}; -use dozer_types::{ - ordered_float::OrderedFloat, - rust_decimal::Decimal, - types::{Field, Schema}, -}; -use proptest::prelude::*; -use std::num::Wrapping; -use Expression::Literal; - -#[test] -fn test_uint_math() { - proptest!(ProptestConfig::with_cases(1000), move |(u_num1: u64, u_num2: u64, u128_num1: u128, u128_num2: u128, i_num1: i64, i_num2: i64, i128_num1: i128, i128_num2: i128, f_num1: f64, f_num2: f64, d_num1: ArbitraryDecimal, d_num2: ArbitraryDecimal)| { - let row = Record::new(vec![]); - - let mut uint1 = Box::new(Literal(Field::UInt(u_num1))); - let mut uint2 = Box::new(Literal(Field::UInt(u_num2))); - let mut u128_1 = Box::new(Literal(Field::U128(u128_num1))); - let mut u128_2 = Box::new(Literal(Field::U128(u128_num2))); - let mut int1 = Box::new(Literal(Field::Int(i_num1))); - let mut int2 = Box::new(Literal(Field::Int(i_num2))); - let mut i128_1 = Box::new(Literal(Field::I128(i128_num1))); - let mut i128_2 = Box::new(Literal(Field::I128(i128_num2))); - let mut float1 = Box::new(Literal(Field::Float(OrderedFloat(f_num1)))); - let mut float2 = Box::new(Literal(Field::Float(OrderedFloat(f_num2)))); - let mut dec1 = Box::new(Literal(Field::Decimal(d_num1.0))); - let mut dec2 = Box::new(Literal(Field::Decimal(d_num2.0))); - - let mut null = Box::new(Literal(Field::Null)); - - //// left: UInt, right: UInt - assert_eq!( - // UInt + UInt = UInt - evaluate_add(&Schema::default(), &mut uint1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::UInt((Wrapping(u_num1) + Wrapping(u_num2)).0) - ); - assert_eq!( - // UInt - UInt = UInt - evaluate_sub(&Schema::default(), &mut uint1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::UInt((Wrapping(u_num1) - Wrapping(u_num2)).0) - ); - assert_eq!( - // UInt * UInt = UInt - evaluate_mul(&Schema::default(), &mut uint2, &mut uint1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::UInt((Wrapping(u_num2) * Wrapping(u_num1)).0) - ); - assert_eq!( - // UInt / UInt = Float - evaluate_div(&Schema::default(), &mut uint2, &mut uint1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u64(u_num2).unwrap() / f64::from_u64(u_num1).unwrap())) - ); - assert_eq!( - // UInt % UInt = UInt - evaluate_mod(&Schema::default(), &mut uint1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::UInt((Wrapping(u_num1) % Wrapping(u_num2)).0) - ); - - //// left: UInt, right: U128 - assert_eq!( - // UInt + U128 = U128 - evaluate_add(&Schema::default(), &mut uint1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::U128((Wrapping(u_num1 as u128) + Wrapping(u128_num2)).0) - ); - assert_eq!( - // UInt - U128 = U128 - evaluate_sub(&Schema::default(), &mut uint1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::U128((Wrapping(u_num1 as u128) - Wrapping(u128_num2)).0) - ); - assert_eq!( - // UInt * U128 = U128 - evaluate_mul(&Schema::default(), &mut uint2, &mut u128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::U128((Wrapping(u_num2 as u128) * Wrapping(u128_num1)).0) - ); - assert_eq!( - // UInt / U128 = Float - evaluate_div(&Schema::default(), &mut uint2, &mut u128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u64(u_num2).unwrap() / f64::from_u128(u128_num1).unwrap())) - ); - assert_eq!( - // UInt % U128 = U128 - evaluate_mod(&Schema::default(), &mut uint1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::U128((Wrapping(u_num1 as u128) % Wrapping(u128_num2)).0) - ); - - //// left: UInt, right: Int - assert_eq!( - // UInt + Int = Int - evaluate_add(&Schema::default(), &mut uint1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int((Wrapping(u_num1 as i64) + Wrapping(i_num2)).0) - ); - assert_eq!( - // UInt - Int = Int - evaluate_sub(&Schema::default(), &mut uint1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int((Wrapping(u_num1 as i64) - Wrapping(i_num2)).0) - ); - assert_eq!( - // UInt * Int = Int - evaluate_mul(&Schema::default(), &mut uint2, &mut int1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int((Wrapping(u_num2 as i64) * Wrapping(i_num1)).0) - ); - assert_eq!( - // UInt / Int = Float - evaluate_div(&Schema::default(), &mut uint2, &mut int1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u64(u_num2).unwrap() / f64::from_i64(i_num1).unwrap())) - ); - assert_eq!( - // UInt % Int = Int - evaluate_mod(&Schema::default(), &mut uint1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int((Wrapping(u_num1 as i64) % Wrapping(i_num2)).0) - ); - - //// left: UInt, right: I128 - assert_eq!( - // UInt + I128 = I128 - evaluate_add(&Schema::default(), &mut uint1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(u_num1 as i128) + Wrapping(i128_num2)).0) - ); - assert_eq!( - // UInt - I128 = I128 - evaluate_sub(&Schema::default(), &mut uint1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(u_num1 as i128) - Wrapping(i128_num2)).0) - ); - assert_eq!( - // UInt * I128 = I128 - evaluate_mul(&Schema::default(), &mut uint2, &mut i128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(u_num2 as i128) * Wrapping(i128_num1)).0) - ); - assert_eq!( - // UInt / I128 = Float - evaluate_div(&Schema::default(), &mut uint2, &mut i128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u64(u_num2).unwrap() / f64::from_i128(i128_num1).unwrap())) - ); - assert_eq!( - // UInt % I128 = I128 - evaluate_mod(&Schema::default(), &mut uint1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(u_num1 as i128) % Wrapping(i128_num2)).0) - ); - - //// left: UInt, right: Float - assert_eq!( - // UInt + Float = Float - evaluate_add(&Schema::default(), &mut uint1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u64(u_num1).unwrap() + f_num2)) - ); - assert_eq!( - // UInt - Float = Float - evaluate_sub(&Schema::default(), &mut uint1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u64(u_num1).unwrap() - f_num2)) - ); - assert_eq!( - // UInt * Float = Float - evaluate_mul(&Schema::default(), &mut uint2, &mut float1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u64(u_num2).unwrap() * f_num1)) - ); - if *float1 != Literal(Field::Float(OrderedFloat(0_f64))) { - assert_eq!( - // UInt / Float = Float - evaluate_div(&Schema::default(), &mut uint2, &mut float1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u64(u_num2).unwrap() / f_num1)) - ); - } - if *float2 != Literal(Field::Float(OrderedFloat(0_f64))) { - assert_eq!( - // UInt % Float = Float - evaluate_mod(&Schema::default(), &mut uint1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u64(u_num1).unwrap() % f_num2)) - ); - } - - //// left: UInt, right: Decimal - assert_eq!( - // UInt + Decimal = Decimal - evaluate_add(&Schema::default(), &mut uint1, &mut dec2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_u64(u_num1).unwrap() + d_num2.0) - ); - assert_eq!( - // UInt - Decimal = Decimal - evaluate_sub(&Schema::default(), &mut uint1, &mut dec2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_u64(u_num1).unwrap() - d_num2.0) - ); - // UInt * Decimal = Decimal - let res = evaluate_mul(&Schema::default(), &mut uint2, &mut dec1, &row); - if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(Decimal::from_u64(u_num2).unwrap().checked_mul(d_num1.0).unwrap()) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::MultiplicationOverflow)) - )); - } - // UInt / Decimal = Decimal - let res = evaluate_div(&Schema::default(), &mut uint2, &mut dec1, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(Decimal::from_u64(u_num2).unwrap() / d_num1.0) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - // UInt % Decimal = Decimal - let res = evaluate_mod(&Schema::default(), &mut uint2, &mut dec1, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(Decimal::from_u64(u_num2).unwrap() % d_num1.0) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - - //// left: UInt, right: Null - assert_eq!( - // UInt + Null = Null - evaluate_add(&Schema::default(), &mut uint1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // UInt - Null = Null - evaluate_sub(&Schema::default(), &mut uint1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // UInt * Null = Null - evaluate_mul(&Schema::default(), &mut uint2, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // UInt / Null = Null - evaluate_div(&Schema::default(), &mut uint2, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // UInt % Null = Null - evaluate_mod(&Schema::default(), &mut uint1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - }); -} - -#[test] -fn test_u128_math() { - proptest!(ProptestConfig::with_cases(1000), move |(u_num1: u64, u_num2: u64, u128_num1: u128, u128_num2: u128, i_num1: i64, i_num2: i64, i128_num1: i128, i128_num2: i128, f_num1: f64, f_num2: f64, d_num1: ArbitraryDecimal, d_num2: ArbitraryDecimal)| { - let row = Record::new(vec![]); - - let mut uint1 = Box::new(Literal(Field::UInt(u_num1))); - let mut uint2 = Box::new(Literal(Field::UInt(u_num2))); - let mut u128_1 = Box::new(Literal(Field::U128(u128_num1))); - let mut u128_2 = Box::new(Literal(Field::U128(u128_num2))); - let mut int1 = Box::new(Literal(Field::Int(i_num1))); - let mut int2 = Box::new(Literal(Field::Int(i_num2))); - let mut i128_1 = Box::new(Literal(Field::I128(i128_num1))); - let mut i128_2 = Box::new(Literal(Field::I128(i128_num2))); - let mut float1 = Box::new(Literal(Field::Float(OrderedFloat(f_num1)))); - let mut float2 = Box::new(Literal(Field::Float(OrderedFloat(f_num2)))); - let mut dec1 = Box::new(Literal(Field::Decimal(d_num1.0))); - let mut dec2 = Box::new(Literal(Field::Decimal(d_num2.0))); - - let mut null = Box::new(Literal(Field::Null)); - - //// left: U128, right: UInt - assert_eq!( - // U128 + UInt = U128 - evaluate_add(&Schema::default(), &mut u128_1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::U128((Wrapping(u128_num1) + Wrapping(u_num2 as u128)).0) - ); - assert_eq!( - // U128 - UInt = U128 - evaluate_sub(&Schema::default(), &mut u128_1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::U128((Wrapping(u128_num1) - Wrapping(u_num2 as u128)).0) - ); - assert_eq!( - // U128 * UInt = U128 - evaluate_mul(&Schema::default(), &mut u128_2, &mut uint1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::U128((Wrapping(u128_num2) * Wrapping(u_num1 as u128)).0) - ); - assert_eq!( - // U128 / UInt = Float - evaluate_div(&Schema::default(), &mut u128_2, &mut uint1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u128(u128_num2).unwrap() / f64::from_u64(u_num1).unwrap())) - ); - assert_eq!( - // U128 % UInt = U128 - evaluate_mod(&Schema::default(), &mut u128_1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::U128((Wrapping(u128_num1) % Wrapping(u_num2 as u128)).0) - ); - - //// left: U128, right: U128 - assert_eq!( - // U128 + U128 = U128 - evaluate_add(&Schema::default(), &mut u128_1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::U128((Wrapping(u128_num1) + Wrapping(u128_num2)).0) - ); - assert_eq!( - // U128 - U128 = U128 - evaluate_sub(&Schema::default(), &mut u128_1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::U128((Wrapping(u128_num1) - Wrapping(u128_num2)).0) - ); - assert_eq!( - // U128 * U128 = U128 - evaluate_mul(&Schema::default(), &mut u128_2, &mut u128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::U128((Wrapping(u128_num2) * Wrapping(u128_num1)).0) - ); - assert_eq!( - // U128 / U128 = Float - evaluate_div(&Schema::default(), &mut u128_2, &mut u128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u128(u128_num2).unwrap() / f64::from_u128(u128_num1).unwrap())) - ); - assert_eq!( - // U128 % U128 = U128 - evaluate_mod(&Schema::default(), &mut u128_1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::U128((Wrapping(u128_num1) % Wrapping(u128_num2)).0) - ); - - //// left: U128, right: Int - assert_eq!( - // U128 + Int = I128 - evaluate_add(&Schema::default(), &mut u128_1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(u128_num1 as i128) + Wrapping(i_num2 as i128)).0) - ); - assert_eq!( - // U128 - Int = I128 - evaluate_sub(&Schema::default(), &mut u128_1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(u128_num1 as i128) - Wrapping(i_num2 as i128)).0) - ); - assert_eq!( - // U128 * Int = I128 - evaluate_mul(&Schema::default(), &mut u128_2, &mut int1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(u128_num2 as i128) * Wrapping(i_num1 as i128)).0) - ); - assert_eq!( - // U128 / Int = Float - evaluate_div(&Schema::default(), &mut u128_2, &mut int1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u128(u128_num2).unwrap() / f64::from_i64(i_num1).unwrap())) - ); - assert_eq!( - // U128 % Int = I128 - evaluate_mod(&Schema::default(), &mut u128_1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(u128_num1 as i128) % Wrapping(i_num2 as i128)).0) - ); - - //// left: U128, right: I128 - assert_eq!( - // U128 + I128 = I128 - evaluate_add(&Schema::default(), &mut u128_1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(u128_num1 as i128) + Wrapping(i128_num2)).0) - ); - assert_eq!( - // U128 - I128 = I128 - evaluate_sub(&Schema::default(), &mut u128_1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(u128_num1 as i128) - Wrapping(i128_num2)).0) - ); - assert_eq!( - // U128 * I128 = I128 - evaluate_mul(&Schema::default(), &mut u128_2, &mut i128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(u128_num2 as i128) * Wrapping(i128_num1)).0) - ); - assert_eq!( - // U128 / I128 = Float - evaluate_div(&Schema::default(), &mut u128_2, &mut i128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u128(u128_num2).unwrap() / f64::from_i128(i128_num1).unwrap())) - ); - assert_eq!( - // U128 % I128 = I128 - evaluate_mod(&Schema::default(), &mut u128_1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(u128_num1 as i128) % Wrapping(i128_num2)).0) - ); - - //// left: U128, right: Float - let res = evaluate_add(&Schema::default(), &mut u128_1, &mut float2, &row); - if res.is_ok() { - assert_eq!( - // U128 + Float = Float - evaluate_add(&Schema::default(), &mut u128_1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u128(u128_num1).unwrap() + f_num2)) - ); - } - let res = evaluate_sub(&Schema::default(), &mut u128_1, &mut float2, &row); - if res.is_ok() { - assert_eq!( - // U128 - Float = Float - evaluate_sub(&Schema::default(), &mut u128_1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u128(u128_num1).unwrap() - f_num2)) - ); - } - let res = evaluate_mul(&Schema::default(), &mut u128_2, &mut float1, &row); - if res.is_ok() { - assert_eq!( - // U128 * Float = Float - evaluate_mul(&Schema::default(), &mut u128_2, &mut float1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u128(u128_num2).unwrap() * f_num1)) - ); - } - let res = evaluate_div(&Schema::default(), &mut u128_2, &mut float1, &row); - if res.is_ok() { - assert_eq!( - // U128 / Float = Float - evaluate_div(&Schema::default(), &mut u128_2, &mut float1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u128(u128_num2).unwrap() / f_num1)) - ); - } - let res = evaluate_mod(&Schema::default(), &mut u128_1, &mut float2, &row); - if res.is_ok() { - assert_eq!( - // U128 % Float = Float - evaluate_mod(&Schema::default(), &mut u128_1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_u128(u128_num1).unwrap() % f_num2)) - ); - } - - //// left: U128, right: Decimal - let res = evaluate_add(&Schema::default(), &mut u128_1, &mut dec2, &row); - if res.is_ok() { - assert_eq!( - // U128 + Decimal = Decimal - evaluate_add(&Schema::default(), &mut u128_1, &mut dec2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_u128(u128_num1).unwrap() + d_num2.0) - ); - } - let res = evaluate_sub(&Schema::default(), &mut u128_1, &mut dec2, &row); - if res.is_ok() { - assert_eq!( - // U128 - Decimal = Decimal - evaluate_sub(&Schema::default(), &mut u128_1, &mut dec2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_u128(u128_num1).unwrap() - d_num2.0) - ); - } - // U128 * Decimal = Decimal - let res = evaluate_mul(&Schema::default(), &mut u128_2, &mut dec1, &row); - if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(Decimal::from_u128(u128_num2).unwrap().checked_mul(d_num1.0).unwrap()) - ); - } else { - assert!(res.is_err()); - if !matches!(res, Err(PipelineError::UnableToCast(_, _))) { - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::MultiplicationOverflow)) - )); - } - } - // U128 / Decimal = Decimal - let res = evaluate_div(&Schema::default(), &mut u128_2, &mut dec1, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - if !matches!(res, Err(PipelineError::UnableToCast(_, _))) { - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(Decimal::from_u128(u128_num2).unwrap() / d_num1.0) - ); - } else { - assert!(res.is_err()); - if !matches!(res, Err(PipelineError::UnableToCast(_, _))) { - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - } - // U128 % Decimal = Decimal - let res = evaluate_mod(&Schema::default(), &mut u128_1, &mut dec1, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - if !matches!(res, Err(PipelineError::UnableToCast(_, _))) { - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(Decimal::from_u128(u128_num1).unwrap() % d_num1.0) - ); - } else { - assert!(res.is_err()); - if !matches!(res, Err(PipelineError::UnableToCast(_, _))) { - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - } - - //// left: U128, right: Null - assert_eq!( - // U128 + Null = Null - evaluate_add(&Schema::default(), &mut u128_1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // U128 - Null = Null - evaluate_sub(&Schema::default(), &mut u128_1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // U128 * Null = Null - evaluate_mul(&Schema::default(), &mut u128_2, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // U128 / Null = Null - evaluate_div(&Schema::default(), &mut u128_2, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // U128 % Null = Null - evaluate_mod(&Schema::default(), &mut u128_1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - }); -} - -#[test] -fn test_int_math() { - proptest!(ProptestConfig::with_cases(1000), move |(u_num1: u64, u_num2: u64, u128_num1: u128, u128_num2: u128, i_num1: i64, i_num2: i64, i128_num1: i128, i128_num2: i128, f_num1: f64, f_num2: f64, d_num1: ArbitraryDecimal, d_num2: ArbitraryDecimal)| { - let row = Record::new(vec![]); - - let mut uint1 = Box::new(Literal(Field::UInt(u_num1))); - let mut uint2 = Box::new(Literal(Field::UInt(u_num2))); - let mut u128_1 = Box::new(Literal(Field::U128(u128_num1))); - let mut u128_2 = Box::new(Literal(Field::U128(u128_num2))); - let mut int1 = Box::new(Literal(Field::Int(i_num1))); - let mut int2 = Box::new(Literal(Field::Int(i_num2))); - let mut i128_1 = Box::new(Literal(Field::I128(i128_num1))); - let mut i128_2 = Box::new(Literal(Field::I128(i128_num2))); - let mut float1 = Box::new(Literal(Field::Float(OrderedFloat(f_num1)))); - let mut float2 = Box::new(Literal(Field::Float(OrderedFloat(f_num2)))); - let mut dec1 = Box::new(Literal(Field::Decimal(d_num1.0))); - let mut dec2 = Box::new(Literal(Field::Decimal(d_num2.0))); - - let mut null = Box::new(Literal(Field::Null)); - - //// left: Int, right: UInt - assert_eq!( - // Int + UInt = Int - evaluate_add(&Schema::default(), &mut int1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int((Wrapping(i_num1) + Wrapping(u_num2 as i64)).0) - ); - assert_eq!( - // Int - UInt = Int - evaluate_sub(&Schema::default(), &mut int1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int((Wrapping(i_num1) - Wrapping(u_num2 as i64)).0) - ); - assert_eq!( - // Int * UInt = Int - evaluate_mul(&Schema::default(), &mut int2, &mut uint1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int((Wrapping(i_num2) * Wrapping(u_num1 as i64)).0) - ); - assert_eq!( - // Int / UInt = Float - evaluate_div(&Schema::default(), &mut int2, &mut uint1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i64(i_num2).unwrap() / f64::from_u64(u_num1).unwrap())) - ); - assert_eq!( - // Int % UInt = Int - evaluate_mod(&Schema::default(), &mut int1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int((Wrapping(i_num1) % Wrapping(u_num2 as i64)).0) - ); - - //// left: Int, right: U128 - assert_eq!( - // Int + U128 = I128 - evaluate_add(&Schema::default(), &mut int1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i_num1 as i128) + Wrapping(u128_num2 as i128)).0) - ); - assert_eq!( - // Int - U128 = I128 - evaluate_sub(&Schema::default(), &mut int1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i_num1 as i128) - Wrapping(u128_num2 as i128)).0) - ); - assert_eq!( - // Int * U128 = I128 - evaluate_mul(&Schema::default(), &mut int2, &mut u128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i_num2 as i128) * Wrapping(u128_num1 as i128)).0) - ); - let res = evaluate_div(&Schema::default(), &mut int2, &mut u128_1, &row); - if res.is_ok() { - assert_eq!( - // Int / U128 = Float - evaluate_div(&Schema::default(), &mut int2, &mut u128_1, &row).unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i128(i_num2 as i128).unwrap() / f64::from_i128(u128_num1 as i128).unwrap())) - ); - } - assert_eq!( - // Int % U128 = I128 - evaluate_mod(&Schema::default(), &mut int1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i_num1 as i128) % Wrapping(u128_num2 as i128)).0) - ); - - //// left: Int, right: Int - assert_eq!( - // Int + Int = Int - evaluate_add(&Schema::default(), &mut int1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int((Wrapping(i_num1) + Wrapping(i_num2)).0) - ); - assert_eq!( - // Int - Int = Int - evaluate_sub(&Schema::default(), &mut int1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int((Wrapping(i_num1) - Wrapping(i_num2)).0) - ); - assert_eq!( - // Int * Int = Int - evaluate_mul(&Schema::default(), &mut int2, &mut int1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int((Wrapping(i_num2) * Wrapping(i_num1)).0) - ); - assert_eq!( - // Int / Int = Float - evaluate_div(&Schema::default(), &mut int2, &mut int1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i64(i_num2).unwrap() / f64::from_i64(i_num1).unwrap())) - ); - assert_eq!( - // Int % Int = Int - evaluate_mod(&Schema::default(), &mut int1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int((Wrapping(i_num1) % Wrapping(i_num2)).0) - ); - - //// left: Int, right: I128 - assert_eq!( - // Int + I128 = I128 - evaluate_add(&Schema::default(), &mut int1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i_num1 as i128) + Wrapping(i128_num2)).0) - ); - assert_eq!( - // Int - I128 = I128 - evaluate_sub(&Schema::default(), &mut int1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i_num1 as i128) - Wrapping(i128_num2)).0) - ); - assert_eq!( - // Int * I128 = I128 - evaluate_mul(&Schema::default(), &mut int2, &mut i128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i_num2 as i128) * Wrapping(i128_num1)).0) - ); - let res = evaluate_div(&Schema::default(), &mut int2, &mut i128_1, &row); - if res.is_ok() { - assert_eq!( - // Int / I128 = Float - evaluate_div(&Schema::default(), &mut int2, &mut i128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i64(i_num2).unwrap() / f64::from_i128(i128_num1).unwrap())) - ); - } - assert_eq!( - // Int % I128 = I128 - evaluate_mod(&Schema::default(), &mut int1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i_num1 as i128) % Wrapping(i128_num2)).0) - ); - - //// left: Int, right: Float - assert_eq!( - // Int + Float = Float - evaluate_add(&Schema::default(), &mut int1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i64(i_num1).unwrap() + f_num2)) - ); - assert_eq!( - // Int - Float = Float - evaluate_sub(&Schema::default(), &mut int1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i64(i_num1).unwrap() - f_num2)) - ); - assert_eq!( - // Int * Float = Float - evaluate_mul(&Schema::default(), &mut int2, &mut float1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i64(i_num2).unwrap() * f_num1)) - ); - if *float1 != Literal(Field::Float(OrderedFloat(0_f64))) { - assert_eq!( - // Int / Float = Float - evaluate_div(&Schema::default(), &mut int2, &mut float1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i64(i_num2).unwrap() / f_num1)) - ); - } - if *float2 != Literal(Field::Float(OrderedFloat(0_f64))) { - assert_eq!( - // Int % Float = Float - evaluate_mod(&Schema::default(), &mut int1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i64(i_num1).unwrap() % f_num2)) - ); - } - - //// left: Int, right: Decimal - assert_eq!( - // Int + Decimal = Decimal - evaluate_add(&Schema::default(), &mut int1, &mut dec2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_i64(i_num1).unwrap() + d_num2.0) - ); - assert_eq!( - // Int - Decimal = Decimal - evaluate_sub(&Schema::default(), &mut int1, &mut dec2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_i64(i_num1).unwrap() - d_num2.0) - ); - // Int * Decimal = Decimal - let res = evaluate_mul(&Schema::default(), &mut int2, &mut dec1, &row); - if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(Decimal::from_i64(i_num2).unwrap().checked_mul(d_num1.0).unwrap()) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::MultiplicationOverflow)) - )); - } - // Int / Decimal = Decimal - let res = evaluate_div(&Schema::default(), &mut int2, &mut dec1, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(Decimal::from_i64(i_num2).unwrap() / d_num1.0) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - // Int % Decimal = Decimal - let res = evaluate_mod(&Schema::default(), &mut int1, &mut dec2, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(Decimal::from_i64(i_num1).unwrap() % d_num2.0) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - - //// left: Int, right: Null - assert_eq!( - // Int + Null = Null - evaluate_add(&Schema::default(), &mut int1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Int - Null = Null - evaluate_sub(&Schema::default(), &mut int1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Int * Null = Null - evaluate_mul(&Schema::default(), &mut int2, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Int / Null = Null - evaluate_div(&Schema::default(), &mut int2, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Int % Null = Null - evaluate_mod(&Schema::default(), &mut int1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - }); -} - -#[test] -fn test_i128_math() { - proptest!(ProptestConfig::with_cases(1000), move |(u_num1: u64, u_num2: u64, u128_num1: u128, u128_num2: u128, i_num1: i64, i_num2: i64, i128_num1: i128, i128_num2: i128, f_num1: f64, f_num2: f64, d_num1: ArbitraryDecimal, d_num2: ArbitraryDecimal)| { - let row = Record::new(vec![]); - - let mut uint1 = Box::new(Literal(Field::UInt(u_num1))); - let mut uint2 = Box::new(Literal(Field::UInt(u_num2))); - let mut u128_1 = Box::new(Literal(Field::U128(u128_num1))); - let mut u128_2 = Box::new(Literal(Field::U128(u128_num2))); - let mut int1 = Box::new(Literal(Field::Int(i_num1))); - let mut int2 = Box::new(Literal(Field::Int(i_num2))); - let mut i128_1 = Box::new(Literal(Field::I128(i128_num1))); - let mut i128_2 = Box::new(Literal(Field::I128(i128_num2))); - let mut float1 = Box::new(Literal(Field::Float(OrderedFloat(f_num1)))); - let mut float2 = Box::new(Literal(Field::Float(OrderedFloat(f_num2)))); - let mut dec1 = Box::new(Literal(Field::Decimal(d_num1.0))); - let mut dec2 = Box::new(Literal(Field::Decimal(d_num2.0))); - - let mut null = Box::new(Literal(Field::Null)); - - //// left: I128, right: UInt - assert_eq!( - // I128 + UInt = I128 - evaluate_add(&Schema::default(), &mut i128_1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i128_num1) + Wrapping(u_num2 as i128)).0) - ); - assert_eq!( - // I128 - UInt = I128 - evaluate_sub(&Schema::default(), &mut i128_1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i128_num1) - Wrapping(u_num2 as i128)).0) - ); - assert_eq!( - // I128 * UInt = I128 - evaluate_mul(&Schema::default(), &mut i128_2, &mut uint1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i128_num2) * Wrapping(u_num1 as i128)).0) - ); - let res = evaluate_div(&Schema::default(), &mut i128_2, &mut uint1, &row); - if res.is_ok() { - assert_eq!( - // I128 / UInt = Float - evaluate_div(&Schema::default(), &mut i128_2, &mut uint1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i128(i128_num2).unwrap() / f64::from_u64(u_num1).unwrap())) - ); - } - assert_eq!( - // I128 % UInt = I128 - evaluate_mod(&Schema::default(), &mut i128_1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i128_num1) % Wrapping(u_num2 as i128)).0) - ); - - //// left: I128, right: U128 - assert_eq!( - // I128 + U128 = I128 - evaluate_add(&Schema::default(), &mut i128_1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i128_num1) + Wrapping(u128_num2 as i128)).0) - ); - assert_eq!( - // I128 - U128 = I128 - evaluate_sub(&Schema::default(), &mut i128_1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i128_num1) - Wrapping(u128_num2 as i128)).0) - ); - assert_eq!( - // I128 * U128 = I128 - evaluate_mul(&Schema::default(), &mut i128_2, &mut u128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i128_num2) * Wrapping(u128_num1 as i128)).0) - ); - let res = evaluate_div(&Schema::default(), &mut i128_2, &mut u128_1, &row); - if res.is_ok() { - assert_eq!( - // I128 / U128 = Float - evaluate_div(&Schema::default(), &mut i128_2, &mut u128_1, &row).unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i128(i128_num2).unwrap() / f64::from_i128(u128_num1 as i128).unwrap())) - ); - } - assert_eq!( - // I128 % U128 = I128 - evaluate_mod(&Schema::default(), &mut i128_1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i128_num1) % Wrapping(u128_num2 as i128)).0) - ); - - //// left: I128, right: Int - assert_eq!( - // I128 + Int = I128 - evaluate_add(&Schema::default(), &mut i128_1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i128_num1) + Wrapping(i_num2 as i128)).0) - ); - assert_eq!( - // I128 - Int = I128 - evaluate_sub(&Schema::default(), &mut i128_1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i128_num1) - Wrapping(i_num2 as i128)).0) - ); - assert_eq!( - // I128 * Int = I128 - evaluate_mul(&Schema::default(), &mut i128_2, &mut int1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i128_num2) * Wrapping(i_num1 as i128)).0) - ); - let res = evaluate_div(&Schema::default(), &mut i128_2, &mut int1, &row); - if res.is_ok() { - assert_eq!( - // I128 / Int = Float - evaluate_div(&Schema::default(), &mut i128_2, &mut int1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i128(i128_num2).unwrap() / f64::from_i64(i_num1).unwrap())) - ); - } - assert_eq!( - // I128 % Int = I128 - evaluate_mod(&Schema::default(), &mut i128_1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i128_num1) % Wrapping(i_num2 as i128)).0) - ); - - //// left: I128, right: I128 - assert_eq!( - // I128 + I128 = I128 - evaluate_add(&Schema::default(), &mut i128_1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i128_num1) + Wrapping(i128_num2)).0) - ); - assert_eq!( - // I128 - I128 = I128 - evaluate_sub(&Schema::default(), &mut i128_1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i128_num1) - Wrapping(i128_num2)).0) - ); - assert_eq!( - // I128 * I128 = I128 - evaluate_mul(&Schema::default(), &mut i128_2, &mut i128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i128_num2) * Wrapping(i128_num1)).0) - ); - let res = evaluate_div(&Schema::default(), &mut i128_2, &mut i128_1, &row); - if res.is_ok() { - assert_eq!( - // I128 / I128 = Float - evaluate_div(&Schema::default(), &mut i128_2, &mut i128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i128(i128_num2).unwrap() / f64::from_i128(i128_num1).unwrap())) - ); - } - assert_eq!( - // I128 % I128 = I128 - evaluate_mod(&Schema::default(), &mut i128_1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::I128((Wrapping(i128_num1) % Wrapping(i128_num2)).0) - ); - - //// left: I128, right: Float - let res = evaluate_add(&Schema::default(), &mut i128_1, &mut float2, &row); - if res.is_ok() { - assert_eq!( - // I128 + Float = Float - evaluate_add(&Schema::default(), &mut i128_1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i128(i128_num1).unwrap() + f_num2)) - ); - } - let res = evaluate_sub(&Schema::default(), &mut i128_1, &mut float2, &row); - if res.is_ok() { - assert_eq!( - // I128 - Float = Float - evaluate_sub(&Schema::default(), &mut i128_1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i128(i128_num1).unwrap() - f_num2)) - ); - } - let res = evaluate_mul(&Schema::default(), &mut i128_2, &mut float1, &row); - if res.is_ok() { - assert_eq!( - // I128 * Float = Float - evaluate_mul(&Schema::default(), &mut i128_2, &mut float1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i128(i128_num2).unwrap() * f_num1)) - ); - } - let res = evaluate_div(&Schema::default(), &mut i128_2, &mut float1, &row); - if res.is_ok() { - assert_eq!( - // I128 / Float = Float - evaluate_div(&Schema::default(), &mut i128_2, &mut float1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i128(i128_num2).unwrap() / f_num1)) - ); - } - let res = evaluate_mod(&Schema::default(), &mut i128_1, &mut float2, &row); - if res.is_ok() { - assert_eq!( - // I128 % Float = Float - evaluate_mod(&Schema::default(), &mut i128_1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f64::from_i128(i128_num1).unwrap() % f_num2)) - ); - } - - //// left: I128, right: Decimal - let res = evaluate_add(&Schema::default(), &mut i128_1, &mut dec2, &row); - if res.is_ok() { - assert_eq!( - // I128 + Decimal = Decimal - evaluate_add(&Schema::default(), &mut i128_1, &mut dec2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_i128(i128_num1).unwrap() + d_num2.0) - ); - } - let res = evaluate_sub(&Schema::default(), &mut i128_1, &mut dec2, &row); - if res.is_ok() { - assert_eq!( - // I128 - Decimal = Decimal - evaluate_sub(&Schema::default(), &mut i128_1, &mut dec2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_i128(i128_num1).unwrap() - d_num2.0) - ); - } - // I128 * Decimal = Decimal - let res = evaluate_mul(&Schema::default(), &mut i128_2, &mut dec1, &row); - if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(Decimal::from_i128(i128_num2).unwrap().checked_mul(d_num1.0).unwrap()) - ); - } else { - assert!(res.is_err()); - if !matches!(res, Err(PipelineError::UnableToCast(_, _))) { - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::MultiplicationOverflow)) - )); - } - } - // I128 / Decimal = Decimal - let res = evaluate_div(&Schema::default(), &mut i128_2, &mut dec1, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - if !matches!(res, Err(PipelineError::UnableToCast(_, _))) { - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(Decimal::from_i128(i128_num2).unwrap() / d_num1.0) - ); - } else { - assert!(res.is_err()); - if !matches!(res, Err(PipelineError::UnableToCast(_, _))) { - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - } - // I128 % Decimal = Decimal - let res = evaluate_mod(&Schema::default(), &mut i128_1, &mut dec2, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - if !matches!(res, Err(PipelineError::UnableToCast(_, _))) { - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(Decimal::from_i128(i128_num1).unwrap() % d_num2.0) - ); - } else { - assert!(res.is_err()); - if !matches!(res, Err(PipelineError::UnableToCast(_, _))) { - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - } - - //// left: I128, right: Null - assert_eq!( - // I128 + Null = Null - evaluate_add(&Schema::default(), &mut i128_1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // I128 - Null = Null - evaluate_sub(&Schema::default(), &mut i128_1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // I128 * Null = Null - evaluate_mul(&Schema::default(), &mut i128_2, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // I128 / Null = Null - evaluate_div(&Schema::default(), &mut i128_2, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // I128 % Null = Null - evaluate_mod(&Schema::default(), &mut i128_1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - }); -} - -#[test] -fn test_float_math() { - proptest!(ProptestConfig::with_cases(1000), move |(u_num1: u64, u_num2: u64, u128_num1: u128, u128_num2: u128, i_num1: i64, i_num2: i64, i128_num1: i128, i128_num2: i128, f_num1: f64, f_num2: f64, d_num1: ArbitraryDecimal, d_num2: ArbitraryDecimal)| { - let row = Record::new(vec![]); - - let mut uint1 = Box::new(Literal(Field::UInt(u_num1))); - let mut uint2 = Box::new(Literal(Field::UInt(u_num2))); - let mut u128_1 = Box::new(Literal(Field::U128(u128_num1))); - let mut u128_2 = Box::new(Literal(Field::U128(u128_num2))); - let mut int1 = Box::new(Literal(Field::Int(i_num1))); - let mut int2 = Box::new(Literal(Field::Int(i_num2))); - let mut i128_1 = Box::new(Literal(Field::I128(i128_num1))); - let mut i128_2 = Box::new(Literal(Field::I128(i128_num2))); - let mut float1 = Box::new(Literal(Field::Float(OrderedFloat(f_num1)))); - let mut float2 = Box::new(Literal(Field::Float(OrderedFloat(f_num2)))); - let mut dec1 = Box::new(Literal(Field::Decimal(d_num1.0))); - let mut dec2 = Box::new(Literal(Field::Decimal(d_num2.0))); - - let mut null = Box::new(Literal(Field::Null)); - - //// left: Float, right: UInt - assert_eq!( - // Float + UInt = Float - evaluate_add(&Schema::default(), &mut float1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num1) + OrderedFloat(f64::from_u64(u_num2).unwrap())) - ); - assert_eq!( - // Float - UInt = Float - evaluate_sub(&Schema::default(), &mut float1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num1) - OrderedFloat(f64::from_u64(u_num2).unwrap())) - ); - assert_eq!( - // Float * UInt = Float - evaluate_mul(&Schema::default(), &mut float2, &mut uint1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num2) * OrderedFloat(f64::from_u64(u_num1).unwrap())) - ); - assert_eq!( - // Float / UInt = Float - evaluate_div(&Schema::default(), &mut float2, &mut uint1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num2) / OrderedFloat(f64::from_u64(u_num1).unwrap())) - ); - assert_eq!( - // Float % UInt = Float - evaluate_mod(&Schema::default(), &mut float1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num1) % OrderedFloat(f64::from_u64(u_num2).unwrap())) - ); - - //// left: Float, right: U128 - let res = evaluate_add(&Schema::default(), &mut float1, &mut u128_2, &row); - if res.is_ok() { - assert_eq!( - // Float + U128 = Float - evaluate_add(&Schema::default(), &mut float1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num1) + OrderedFloat(f64::from_u128(u128_num2).unwrap())) - ); - } - let res = evaluate_sub(&Schema::default(), &mut float1, &mut u128_2, &row); - if res.is_ok() { - assert_eq!( - // Float - U128 = Float - evaluate_sub(&Schema::default(), &mut float1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num1) - OrderedFloat(f64::from_u128(u128_num2).unwrap())) - ); - } - let res = evaluate_mul(&Schema::default(), &mut float2, &mut u128_1, &row); - if res.is_ok() { - assert_eq!( - // Float * U128 = Float - evaluate_mul(&Schema::default(), &mut float2, &mut u128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num2) * OrderedFloat(f64::from_u128(u128_num1).unwrap())) - ); - } - let res = evaluate_div(&Schema::default(), &mut float2, &mut u128_1, &row); - if res.is_ok() { - assert_eq!( - // Float / U128 = Float - evaluate_div(&Schema::default(), &mut float2, &mut u128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num2) / OrderedFloat(f64::from_u128(u128_num1).unwrap())) - ); - } - let res = evaluate_mod(&Schema::default(), &mut float1, &mut u128_2, &row); - if res.is_ok() { - assert_eq!( - // Float % U128 = Float - evaluate_mod(&Schema::default(), &mut float1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num1) % OrderedFloat(f64::from_u128(u128_num2).unwrap())) - ); - } - - //// left: Float, right: Int - assert_eq!( - // Float + Int = Float - evaluate_add(&Schema::default(), &mut float1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num1) + OrderedFloat(f64::from_i64(i_num2).unwrap())) - ); - assert_eq!( - // Float - Int = Float - evaluate_sub(&Schema::default(), &mut float1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num1) - OrderedFloat(f64::from_i64(i_num2).unwrap())) - ); - assert_eq!( - // Float * Int = Float - evaluate_mul(&Schema::default(), &mut float2, &mut int1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num2) * OrderedFloat(f64::from_i64(i_num1).unwrap())) - ); - assert_eq!( - // Float / Int = Float - evaluate_div(&Schema::default(), &mut float2, &mut int1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num2) / OrderedFloat(f64::from_i64(i_num1).unwrap())) - ); - assert_eq!( - // Float % Int = Float - evaluate_mod(&Schema::default(), &mut float1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num1) % OrderedFloat(f64::from_i64(i_num2).unwrap())) - ); - - //// left: Float, right: I128 - let res = evaluate_add(&Schema::default(), &mut float1, &mut i128_2, &row); - if res.is_ok() { - assert_eq!( - // Float + I128 = Float - evaluate_add(&Schema::default(), &mut float1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num1) + OrderedFloat(f64::from_i128(i128_num2).unwrap())) - ); - } - let res = evaluate_sub(&Schema::default(), &mut float1, &mut i128_2, &row); - if res.is_ok() { - assert_eq!( - // Float - I128 = Float - evaluate_sub(&Schema::default(), &mut float1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num1) - OrderedFloat(f64::from_i128(i128_num2).unwrap())) - ); - } - let res = evaluate_mul(&Schema::default(), &mut float2, &mut i128_1, &row); - if res.is_ok() { - assert_eq!( - // Float * I128 = Float - evaluate_mul(&Schema::default(), &mut float2, &mut i128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num2) * OrderedFloat(f64::from_i128(i128_num1).unwrap())) - ); - } - let res = evaluate_div(&Schema::default(), &mut float2, &mut i128_1, &row); - if res.is_ok() { - assert_eq!( - // Float / I128 = Float - evaluate_div(&Schema::default(), &mut float2, &mut i128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num2) / OrderedFloat(f64::from_i128(i128_num1).unwrap())) - ); - } - let res = evaluate_mod(&Schema::default(), &mut float1, &mut i128_2, &row); - if res.is_ok() { - assert_eq!( - // Float % I128 = Float - evaluate_mod(&Schema::default(), &mut float1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num1) % OrderedFloat(f64::from_i128(i128_num2).unwrap())) - ); - } - - //// left: Float, right: Float - assert_eq!( - // Float + Float = Float - evaluate_add(&Schema::default(), &mut float1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num1 + f_num2)) - ); - assert_eq!( - // Float - Float = Float - evaluate_sub(&Schema::default(), &mut float1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num1 - f_num2)) - ); - assert_eq!( - // Float * Float = Float - evaluate_mul(&Schema::default(), &mut float2, &mut float1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num2 * f_num1)) - ); - if *float1 != Literal(Field::Float(OrderedFloat(0_f64))) { - assert_eq!( - // Float / Float = Float - evaluate_div(&Schema::default(), &mut float2, &mut float1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num2 / f_num1)) - ); - } - if *float2 != Literal(Field::Float(OrderedFloat(0_f64))) { - assert_eq!( - // Float % Float = Float - evaluate_mod(&Schema::default(), &mut float1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num1 % f_num2)) - ); - } - - //// left: Float, right: Decimal - let d_val1 = Decimal::from_f64(f_num1); - let d_val2 = Decimal::from_f64(f_num2); - if d_val1.is_some() && d_val2.is_some() { - assert_eq!( - // Float + Decimal = Decimal - evaluate_add(&Schema::default(), &mut float1, &mut dec2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_val1.unwrap() + d_num2.0) - ); - assert_eq!( - // Float - Decimal = Decimal - evaluate_sub(&Schema::default(), &mut float1, &mut dec2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_val1.unwrap() - d_num2.0) - ); - // Float * Decimal = Decimal - let res = evaluate_mul(&Schema::default(), &mut float2, &mut dec1, &row); - if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(d_val2.unwrap().checked_mul(d_num1.0).unwrap()) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::MultiplicationOverflow)) - )); - } - // Float / Decimal = Decimal - let res = evaluate_div(&Schema::default(), &mut float2, &mut dec1, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(d_val2.unwrap().checked_div(d_num1.0).unwrap()) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - // Float % Decimal = Decimal - let res = evaluate_mod(&Schema::default(), &mut float1, &mut dec2, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(d_val1.unwrap().checked_rem(d_num2.0).unwrap()) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - } - - //// left: Float, right: Null - assert_eq!( - // Float + Null = Null - evaluate_add(&Schema::default(), &mut float1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Float - Null = Null - evaluate_sub(&Schema::default(), &mut float1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Float * Null = Null - evaluate_mul(&Schema::default(), &mut float2, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Float / Null = Null - evaluate_div(&Schema::default(), &mut float2, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Float % Null = Null - evaluate_mod(&Schema::default(), &mut float1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - }); -} - -#[test] -fn test_decimal_math() { - proptest!(ProptestConfig::with_cases(1000), move |(u_num1: u64, u_num2: u64, u128_num1: u128, u128_num2: u128, i_num1: i64, i_num2: i64, i128_num1: i128, i128_num2: i128, f_num1: f64, f_num2: f64, d_num1: ArbitraryDecimal, d_num2: ArbitraryDecimal)| { - let row = Record::new(vec![]); - - let mut uint1 = Box::new(Literal(Field::UInt(u_num1))); - let mut uint2 = Box::new(Literal(Field::UInt(u_num2))); - let mut u128_1 = Box::new(Literal(Field::U128(u128_num1))); - let mut u128_2 = Box::new(Literal(Field::U128(u128_num2))); - let mut int1 = Box::new(Literal(Field::Int(i_num1))); - let mut int2 = Box::new(Literal(Field::Int(i_num2))); - let mut i128_1 = Box::new(Literal(Field::I128(i128_num1))); - let mut i128_2 = Box::new(Literal(Field::I128(i128_num2))); - let mut float1 = Box::new(Literal(Field::Float(OrderedFloat(f_num1)))); - let mut float2 = Box::new(Literal(Field::Float(OrderedFloat(f_num2)))); - let mut dec1 = Box::new(Literal(Field::Decimal(d_num1.0))); - let mut dec2 = Box::new(Literal(Field::Decimal(d_num2.0))); - - let mut null = Box::new(Literal(Field::Null)); - - //// left: Decimal, right: UInt - assert_eq!( - // Decimal + UInt = Decimal - evaluate_add(&Schema::default(), &mut dec1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num1.0 + Decimal::from(u_num2)) - ); - assert_eq!( - // Decimal - UInt = Decimal - evaluate_sub(&Schema::default(), &mut dec1, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num1.0 - Decimal::from(u_num2)) - ); - // Decimal * UInt = Decimal - let res = evaluate_mul(&Schema::default(), &mut dec2, &mut uint1, &row); - if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(d_num2.0 * Decimal::from(u_num1)) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::MultiplicationOverflow)) - )); - } - // Decimal / UInt = Decimal - let res = evaluate_div(&Schema::default(), &mut dec2, &mut uint1, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(d_num2.0 / Decimal::from(u_num1)) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - // Decimal % UInt = Decimal - let res = evaluate_mod(&Schema::default(), &mut dec1, &mut uint2, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(d_num1.0 % Decimal::from(u_num2)) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - - //// left: Decimal, right: U128 - let res = evaluate_add(&Schema::default(), &mut dec1, &mut u128_2, &row); - if res.is_ok() { - assert_eq!( - // Decimal + U128 = Decimal - evaluate_add(&Schema::default(), &mut dec1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num1.0 + Decimal::from_u128(u128_num2).unwrap()) - ); - } - let res = evaluate_sub(&Schema::default(), &mut dec1, &mut u128_2, &row); - if res.is_ok() { - assert_eq!( - // Decimal - U128 = Decimal - evaluate_sub(&Schema::default(), &mut dec1, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num1.0 - Decimal::from_u128(u128_num2).unwrap()) - ); - } - // Decimal * U128 = Decimal - let res = evaluate_mul(&Schema::default(), &mut dec2, &mut u128_1, &row); - if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(d_num2.0 * Decimal::from_u128(u128_num1).unwrap()) - ); - } else { - assert!(res.is_err()); - if !matches!(res, Err(PipelineError::UnableToCast(_, _))) { - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::MultiplicationOverflow)) - )); - } - } - // Decimal / U128 = Decimal - let res = evaluate_div(&Schema::default(), &mut dec2, &mut u128_1, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - if !matches!(res, Err(PipelineError::UnableToCast(_, _))) { - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(d_num2.0 / Decimal::from_u128(u128_num1).unwrap()) - ); - } else { - assert!(res.is_err()); - if !matches!(res, Err(PipelineError::UnableToCast(_, _))) { - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - } - // Decimal % U128 = Decimal - let res = evaluate_mod(&Schema::default(), &mut dec1, &mut u128_2, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - if !matches!(res, Err(PipelineError::UnableToCast(_, _))) { - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(d_num1.0 % Decimal::from_u128(u128_num2).unwrap()) - ); - } else { - assert!(res.is_err()); - if !matches!(res, Err(PipelineError::UnableToCast(_, _))) { - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - } - - //// left: Decimal, right: Int - assert_eq!( - // Decimal + Int = Decimal - evaluate_add(&Schema::default(), &mut dec1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num1.0 + Decimal::from(i_num2)) - ); - assert_eq!( - // Decimal - Int = Decimal - evaluate_sub(&Schema::default(), &mut dec1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num1.0 - Decimal::from(i_num2)) - ); - let res = evaluate_mul(&Schema::default(), &mut dec2, &mut int1, &row); - if res.is_ok() { - assert_eq!( - // Decimal * Int = Decimal - evaluate_mul(&Schema::default(), &mut dec2, &mut int1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num2.0 * Decimal::from(i_num1)) - ); - } - assert_eq!( - // Decimal / Int = Decimal - evaluate_div(&Schema::default(), &mut dec2, &mut int1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num2.0 / Decimal::from(i_num1)) - ); - assert_eq!( - // Decimal % Int = Decimal - evaluate_mod(&Schema::default(), &mut dec1, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num1.0 % Decimal::from(i_num2)) - ); - - //// left: Decimal, right: I128 - let res = evaluate_add(&Schema::default(), &mut dec1, &mut i128_2, &row); - if res.is_ok() { - assert_eq!( - // Decimal + I128 = Decimal - evaluate_add(&Schema::default(), &mut dec1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num1.0 + Decimal::from_i128(i128_num2).unwrap()) - ); - } - let res = evaluate_sub(&Schema::default(), &mut dec1, &mut i128_2, &row); - if res.is_ok() { - assert_eq!( - // Decimal - I128 = Decimal - evaluate_sub(&Schema::default(), &mut dec1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num1.0 - Decimal::from_i128(i128_num2).unwrap()) - ); - } - let res = evaluate_mul(&Schema::default(), &mut dec2, &mut i128_1, &row); - if res.is_ok() { - assert_eq!( - // Decimal * I128 = Decimal - evaluate_mul(&Schema::default(), &mut dec2, &mut i128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num2.0 * Decimal::from_i128(i128_num1).unwrap()) - ); - } - let res = evaluate_div(&Schema::default(), &mut dec2, &mut i128_1, &row); - if res.is_ok() { - assert_eq!( - // Decimal / I128 = Decimal - evaluate_div(&Schema::default(), &mut dec2, &mut i128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num2.0 / Decimal::from_i128(i128_num1).unwrap()) - ); - } - let res = evaluate_mod(&Schema::default(), &mut dec1, &mut i128_2, &row); - if res.is_ok() { - assert_eq!( - // Decimal % I128 = Decimal - evaluate_mod(&Schema::default(), &mut dec1, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num1.0 % Decimal::from_i128(i128_num2).unwrap()) - ); - } - - // left: Decimal, right: Float - let d_val1 = Decimal::from_f64(f_num1); - let d_val2 = Decimal::from_f64(f_num2); - if d_val1.is_some() && d_val2.is_some() && d_val1.unwrap() != Decimal::new(0, 0) && d_val2.unwrap() != Decimal::new(0, 0) { - assert_eq!( - // Decimal + Float = Decimal - evaluate_add(&Schema::default(), &mut dec1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num1.0 + d_val2.unwrap()) - ); - assert_eq!( - // Decimal - Float = Decimal - evaluate_sub(&Schema::default(), &mut dec1, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num1.0 - d_val2.unwrap()) - ); - // Decimal * Float = Decimal - let res = evaluate_mul(&Schema::default(), &mut dec2, &mut float1, &row); - if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(d_num2.0 * d_val1.unwrap()) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::MultiplicationOverflow)) - )); - } - // Decimal / Float = Decimal - let res = evaluate_div(&Schema::default(), &mut dec2, &mut float1, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(d_num2.0 / d_val1.unwrap()) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - // Decimal % Float = Decimal - let res = evaluate_mod(&Schema::default(), &mut dec1, &mut float2, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - else if res.is_ok() { - assert_eq!( - res.unwrap(),Field::Decimal(d_num1.0 % d_val2.unwrap()) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - } - - - //// left: Decimal, right: Decimal - assert_eq!( - // Decimal + Decimal = Decimal - evaluate_add(&Schema::default(), &mut dec1, &mut dec2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num1.0 + d_num2.0) - ); - assert_eq!( - // Decimal - Decimal = Decimal - evaluate_sub(&Schema::default(), &mut dec1, &mut dec2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(d_num1.0 - d_num2.0) - ); - // Decimal * Decimal = Decimal - let res = evaluate_mul(&Schema::default(), &mut dec2, &mut dec1, &row); - if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(d_num2.0 * d_num1.0) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::MultiplicationOverflow)) - )); - } - // Decimal / Decimal = Decimal - let res = evaluate_div(&Schema::default(), &mut dec2, &mut dec1, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(d_num2.0 / d_num1.0) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::DivisionByZeroOrOverflow)) - )); - } - // Decimal % Decimal = Decimal - let res = evaluate_mod(&Schema::default(), &mut dec1, &mut dec2, &row); - if d_num1.0 == Decimal::new(0, 0) { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - else if res.is_ok() { - assert_eq!( - res.unwrap(), Field::Decimal(d_num1.0 % d_num2.0) - ); - } else { - assert!(res.is_err()); - assert!(matches!( - res, - Err(PipelineError::SqlError(OperationError::ModuloByZeroOrOverflow)) - )); - } - - //// left: Decimal, right: Null - assert_eq!( - // Decimal + Null = Null - evaluate_add(&Schema::default(), &mut dec1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal - Null = Null - evaluate_sub(&Schema::default(), &mut dec1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal * Null = Null - evaluate_mul(&Schema::default(), &mut dec2, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal / Null = Null - evaluate_div(&Schema::default(), &mut dec2, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal % Null = Null - evaluate_mod(&Schema::default(), &mut dec1, &mut null, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - }) -} - -#[test] -fn test_null_math() { - proptest!(ProptestConfig::with_cases(1000), move |(u_num1: u64, u_num2: u64, u128_num1: u128, u128_num2: u128, i_num1: i64, i_num2: i64, i128_num1: i128, i128_num2: i128, f_num1: f64, f_num2: f64, d_num1: ArbitraryDecimal, d_num2: ArbitraryDecimal)| { - let row = Record::new(vec![]); - - let mut uint1 = Box::new(Literal(Field::UInt(u_num1))); - let mut uint2 = Box::new(Literal(Field::UInt(u_num2))); - let mut u128_1 = Box::new(Literal(Field::U128(u128_num1))); - let mut u128_2 = Box::new(Literal(Field::U128(u128_num2))); - let mut int1 = Box::new(Literal(Field::Int(i_num1))); - let mut int2 = Box::new(Literal(Field::Int(i_num2))); - let mut i128_1 = Box::new(Literal(Field::I128(i128_num1))); - let mut i128_2 = Box::new(Literal(Field::I128(i128_num2))); - let mut float1 = Box::new(Literal(Field::Float(OrderedFloat(f_num1)))); - let mut float2 = Box::new(Literal(Field::Float(OrderedFloat(f_num2)))); - let mut dec1 = Box::new(Literal(Field::Decimal(d_num1.0))); - let mut dec2 = Box::new(Literal(Field::Decimal(d_num2.0))); - - let mut null = Box::new(Literal(Field::Null)); - - //// left: Null, right: UInt - assert_eq!( - // Null + UInt = Null - evaluate_add(&Schema::default(), &mut null, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Null - UInt = Null - evaluate_sub(&Schema::default(), &mut null, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Null * UInt = Null - evaluate_mul(&Schema::default(), &mut null, &mut uint1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal / UInt = Null - evaluate_div(&Schema::default(), &mut null, &mut uint1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal % UInt = Null - evaluate_mod(&Schema::default(), &mut null, &mut uint2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - - //// left: Null, right: U128 - assert_eq!( - // Null + U128 = Null - evaluate_add(&Schema::default(), &mut null, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Null - U128 = Null - evaluate_sub(&Schema::default(), &mut null, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Null * U128 = Null - evaluate_mul(&Schema::default(), &mut null, &mut u128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal / U128 = Null - evaluate_div(&Schema::default(), &mut null, &mut u128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal % U128 = Null - evaluate_mod(&Schema::default(), &mut null, &mut u128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - - //// left: Null, right: Int - assert_eq!( - // Null + Int = Null - evaluate_add(&Schema::default(), &mut null, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Null - Int = Null - evaluate_sub(&Schema::default(), &mut null, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Null * Int = Null - evaluate_mul(&Schema::default(), &mut null, &mut int1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal / Int = Null - evaluate_div(&Schema::default(), &mut null, &mut int1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal % Int = Null - evaluate_mod(&Schema::default(), &mut null, &mut int2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - - //// left: Null, right: I128 - assert_eq!( - // Null + I128 = Null - evaluate_add(&Schema::default(), &mut null, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Null - I128 = Null - evaluate_sub(&Schema::default(), &mut null, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Null * I128 = Null - evaluate_mul(&Schema::default(), &mut null, &mut i128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal / I128 = Null - evaluate_div(&Schema::default(), &mut null, &mut i128_1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal % I128 = Null - evaluate_mod(&Schema::default(), &mut null, &mut i128_2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - - //// left: Null, right: Float - assert_eq!( - // Null + Float = Null - evaluate_add(&Schema::default(), &mut null, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Null - Float = Null - evaluate_sub(&Schema::default(), &mut null, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Null * Float = Null - evaluate_mul(&Schema::default(), &mut null, &mut float1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal / Float = Null - evaluate_div(&Schema::default(), &mut null, &mut float1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal % Float = Null - evaluate_mod(&Schema::default(), &mut null, &mut float2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - - //// left: Null, right: Decimal - assert_eq!( - // Null + Decimal = Null - evaluate_add(&Schema::default(), &mut null, &mut dec2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Null - Decimal = Null - evaluate_sub(&Schema::default(), &mut null, &mut dec2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Null * Decimal = Null - evaluate_mul(&Schema::default(), &mut null, &mut dec1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal / Decimal = Null - evaluate_div(&Schema::default(), &mut null, &mut dec1, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal % Decimal = Null - evaluate_mod(&Schema::default(), &mut null, &mut dec2, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - - //// left: Null, right: Null - let mut null_clone = null.clone(); - assert_eq!( - // Null + Null = Null - evaluate_add(&Schema::default(), &mut null, &mut null_clone, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Null - Null = Null - evaluate_sub(&Schema::default(), &mut null, &mut null_clone, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Null * Null = Null - evaluate_mul(&Schema::default(), &mut null, &mut null_clone, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal / Null = Null - evaluate_div(&Schema::default(), &mut null, &mut null_clone, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - assert_eq!( - // Decimal % Null = Null - evaluate_mod(&Schema::default(), &mut null, &mut null_clone, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - }) -} - -#[test] -fn test_timestamp_difference() { - let schema = Schema::default() - .field( - FieldDefinition::new( - String::from("a"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - true, - ) - .field( - FieldDefinition::new( - String::from("b"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(); - - let record = Record::new(vec![ - Field::Timestamp(DateTime::parse_from_rfc3339("2020-01-01T00:13:00Z").unwrap()), - Field::Timestamp(DateTime::parse_from_rfc3339("2020-01-01T00:12:10Z").unwrap()), - ]); - - let result = evaluate_sub( - &schema, - &mut Expression::Column { index: 0 }, - &mut Expression::Column { index: 1 }, - &record, - ) - .unwrap(); - assert_eq!( - result, - Field::Duration(DozerDuration( - std::time::Duration::from_nanos(50000 * 1000 * 1000), - TimeUnit::Nanoseconds - )) - ); - - let result = evaluate_sub( - &schema, - &mut Expression::Column { index: 1 }, - &mut Expression::Column { index: 0 }, - &record, - ); - assert!(result.is_err()); -} - -#[test] -fn test_duration() { - proptest!( - ProptestConfig::with_cases(1000), - move |(d1: u64, d2: u64, dt1: ArbitraryDateTime)| { - test_duration_math(d1, d2, dt1) - }); -} - -fn test_duration_math(d1: u64, d2: u64, dt1: ArbitraryDateTime) { - let row = Record::new(vec![]); - - let mut v = Expression::Literal(Field::Date(dt1.0.date_naive())); - let mut dur1 = Expression::Literal(Field::Duration(DozerDuration( - std::time::Duration::from_nanos(d1), - TimeUnit::Nanoseconds, - ))); - let mut dur2 = Expression::Literal(Field::Duration(DozerDuration( - std::time::Duration::from_nanos(d2), - TimeUnit::Nanoseconds, - ))); - - // Duration + Duration = Duration - let result = evaluate_add(&Schema::default(), &mut dur1, &mut dur2, &row); - let sum = std::time::Duration::from_nanos(d1).checked_add(std::time::Duration::from_nanos(d2)); - if result.is_ok() && sum.is_some() { - assert_eq!( - result.unwrap(), - Field::Duration(DozerDuration(sum.unwrap(), TimeUnit::Nanoseconds)) - ); - } - // Duration - Duration = Duration - let result = evaluate_sub(&Schema::default(), &mut dur1, &mut dur2, &row); - let diff = std::time::Duration::from_nanos(d1).checked_sub(std::time::Duration::from_nanos(d2)); - if result.is_ok() && diff.is_some() { - assert_eq!( - result.unwrap(), - Field::Duration(DozerDuration(diff.unwrap(), TimeUnit::Nanoseconds)) - ); - } - // Duration * Duration = Error - let result = evaluate_mul(&Schema::default(), &mut dur1, &mut dur2, &row); - assert!(result.is_err()); - // Duration / Duration = Error - let result = evaluate_div(&Schema::default(), &mut dur1, &mut dur2, &row); - assert!(result.is_err()); - // Duration % Duration = Error - let result = evaluate_mod(&Schema::default(), &mut dur1, &mut dur2, &row); - assert!(result.is_err()); - - // Duration + Timestamp = Error - let result = evaluate_add(&Schema::default(), &mut dur1, &mut v, &row); - assert!(result.is_err()); - // Duration - Timestamp = Error - let result = evaluate_sub(&Schema::default(), &mut dur1, &mut v, &row); - assert!(result.is_err()); - // Duration * Timestamp = Error - let result = evaluate_mul(&Schema::default(), &mut dur1, &mut v, &row); - assert!(result.is_err()); - // Duration / Timestamp = Error - let result = evaluate_div(&Schema::default(), &mut dur1, &mut v, &row); - assert!(result.is_err()); - // Duration % Timestamp = Error - let result = evaluate_mod(&Schema::default(), &mut dur1, &mut v, &row); - assert!(result.is_err()); - - // Timestamp + Duration = Timestamp - let result = evaluate_add(&Schema::default(), &mut v, &mut dur1, &row); - let sum = dt1 - .0 - .checked_add_signed(chrono::Duration::nanoseconds(d1 as i64)); - if result.is_ok() && sum.is_some() { - assert_eq!(result.unwrap(), Field::Timestamp(sum.unwrap())); - } - // Timestamp - Duration = Timestamp - let result = evaluate_sub(&Schema::default(), &mut v, &mut dur2, &row); - let diff = dt1 - .0 - .checked_sub_signed(chrono::Duration::nanoseconds(d2 as i64)); - if result.is_ok() && diff.is_some() { - assert_eq!(result.unwrap(), Field::Timestamp(diff.unwrap())); - } - // Timestamp * Duration = Error - let result = evaluate_mul(&Schema::default(), &mut v, &mut dur1, &row); - assert!(result.is_err()); - // Timestamp / Duration = Error - let result = evaluate_div(&Schema::default(), &mut v, &mut dur1, &row); - assert!(result.is_err()); - // Timestamp % Duration = Error - let result = evaluate_mod(&Schema::default(), &mut v, &mut dur1, &row); - assert!(result.is_err()); -} - -#[test] -fn test_decimal() { - let mut dec1 = Box::new(Literal(Field::Decimal(Decimal::from_i64(1_i64).unwrap()))); - let mut dec2 = Box::new(Literal(Field::Decimal(Decimal::from_i64(2_i64).unwrap()))); - let mut float1 = Box::new(Literal(Field::Float( - OrderedFloat::::from_i64(1_i64).unwrap(), - ))); - let mut float2 = Box::new(Literal(Field::Float( - OrderedFloat::::from_i64(2_i64).unwrap(), - ))); - let mut int1 = Box::new(Literal(Field::Int(1_i64))); - let mut int2 = Box::new(Literal(Field::Int(2_i64))); - let mut uint1 = Box::new(Literal(Field::UInt(1_u64))); - let mut uint2 = Box::new(Literal(Field::UInt(2_u64))); - - let row = Record::new(vec![]); - - // left: Int, right: Decimal - assert_eq!( - evaluate_add(&Schema::default(), &mut int1, dec1.as_mut(), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_i64(2_i64).unwrap()) - ); - assert_eq!( - evaluate_sub(&Schema::default(), &mut int1, dec1.as_mut(), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_i64(0_i64).unwrap()) - ); - assert_eq!( - evaluate_mul(&Schema::default(), &mut int2, dec1.as_mut(), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_i64(2_i64).unwrap()) - ); - assert_eq!( - evaluate_div(&Schema::default(), &mut int1, dec2.as_mut(), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_f64(0.5).unwrap()) - ); - assert_eq!( - evaluate_mod(&Schema::default(), &mut int1, dec1.as_mut(), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_i64(0_i64).unwrap()) - ); - - // left: UInt, right: Decimal - assert_eq!( - evaluate_add(&Schema::default(), &mut uint1, dec1.as_mut(), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_i64(2_i64).unwrap()) - ); - assert_eq!( - evaluate_sub(&Schema::default(), &mut uint1, dec1.as_mut(), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_i64(0_i64).unwrap()) - ); - assert_eq!( - evaluate_mul(&Schema::default(), &mut uint2, dec1.as_mut(), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_i64(2_i64).unwrap()) - ); - assert_eq!( - evaluate_div(&Schema::default(), &mut uint1, dec2.as_mut(), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_f64(0.5).unwrap()) - ); - assert_eq!( - evaluate_mod(&Schema::default(), &mut uint1, dec1.as_mut(), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_i64(0_i64).unwrap()) - ); - - // left: Float, right: Decimal - assert_eq!( - evaluate_add(&Schema::default(), &mut float1, dec1.as_mut(), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_i64(2_i64).unwrap()) - ); - assert_eq!( - evaluate_sub(&Schema::default(), &mut float1, dec1.as_mut(), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_i64(0_i64).unwrap()) - ); - assert_eq!( - evaluate_mul(&Schema::default(), &mut float2, dec1.as_mut(), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_i64(2_i64).unwrap()) - ); - assert_eq!( - evaluate_div(&Schema::default(), &mut float1, dec2.as_mut(), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_f64(0.5).unwrap()) - ); - assert_eq!( - evaluate_mod(&Schema::default(), &mut float1, dec1.as_mut(), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Decimal(Decimal::from_i64(0_i64).unwrap()) - ); -} diff --git a/dozer-sql/expression/src/onnx/error.rs b/dozer-sql/expression/src/onnx/error.rs deleted file mode 100644 index 2765f25fcb..0000000000 --- a/dozer-sql/expression/src/onnx/error.rs +++ /dev/null @@ -1,36 +0,0 @@ -use dozer_types::{ - thiserror::{self, Error}, - types::{Field, FieldType}, -}; -use ndarray::ShapeError; -use ort::{tensor::TensorElementDataType, OrtError}; - -use crate::execution::Expression; - -#[derive(Error, Debug)] -pub enum Error { - #[error("Onnx Ndarray Shape Error: {0}")] - OnnxShapeErr(#[from] ShapeError), - #[error("Onnx Runtime Error: {0}")] - OnnxOrtErr(#[from] OrtError), - #[error("Dozer expect onnx model to ingest single 1d input tensor: size of input {0}")] - OnnxInputSizeErr(usize), - #[error("Expected model input shape {0} doesn't match with actual input shape {1}")] - OnnxInputShapeErr(usize, usize), - #[error("Invalid input shape")] - OnnxInvalidInputShapeErr, - #[error("Expected model input datatype {0:?} doesn't match with actual input datatype {1}")] - OnnxInputDataTypeMismatchErr(TensorElementDataType, FieldType), - #[error("Expected model input datatype {0:?} doesn't match with actual input field {1}")] - OnnxInputDataMismatchErr(TensorElementDataType, Field), - #[error("Expected model output shape {0} doesn't match with actual output shape {1}")] - OnnxOutputShapeErr(usize, usize), - #[error("Dozer doesn't support following output datatype {0:?}")] - OnnxNotSupportedDataTypeErr(TensorElementDataType), - #[error("Dozer can't find following column in the input schema {0:?}")] - ColumnNotFound(Expression), - #[error("Dozer doesn't support non-column for onnx arguments {0:?}")] - NonColumnArgFound(Expression), - #[error("Input argument overflow for {1:?}: {0}")] - InputArgumentOverflow(Field, TensorElementDataType), -} diff --git a/dozer-sql/expression/src/onnx/mod.rs b/dozer-sql/expression/src/onnx/mod.rs deleted file mode 100644 index ce74b51382..0000000000 --- a/dozer-sql/expression/src/onnx/mod.rs +++ /dev/null @@ -1,12 +0,0 @@ -pub mod error; -pub mod udf; -pub mod utils; - -#[derive(Clone, Debug)] -pub struct DozerSession(pub std::sync::Arc); - -impl PartialEq for DozerSession { - fn eq(&self, other: &Self) -> bool { - std::ptr::eq(self as *const _, other as *const _) - } -} diff --git a/dozer-sql/expression/src/onnx/udf.rs b/dozer-sql/expression/src/onnx/udf.rs deleted file mode 100644 index 0741c31b32..0000000000 --- a/dozer-sql/expression/src/onnx/udf.rs +++ /dev/null @@ -1,838 +0,0 @@ -use super::error::Error::{ - InputArgumentOverflow, OnnxInputDataMismatchErr, OnnxInvalidInputShapeErr, - OnnxNotSupportedDataTypeErr, OnnxOrtErr, OnnxShapeErr, -}; -use crate::error::Error::{self, Onnx}; -use crate::execution::Expression; -use dozer_types::log::warn; -use dozer_types::ordered_float::OrderedFloat; -use dozer_types::types::{Field, Record, Schema}; -use half::f16; -use ndarray::Array; -use num_traits::FromPrimitive; -use ort::tensor::TensorElementDataType; -use ort::{Session, Value}; -use std::borrow::Borrow; -use std::ops::Deref; - -pub fn evaluate_onnx_udf( - schema: &Schema, - session: &Session, - args: &mut [Expression], - record: &Record, -) -> Result { - let input_values = args - .iter_mut() - .map(|arg| arg.evaluate(record, schema)) - .collect::, Error>>()?; - - let mut input_dim_prefix = false; - let mut output_dim_prefix = false; - - let mut input_shape = vec![]; - for (i, d) in session.inputs[0].dimensions().enumerate() { - if let Some(v) = d { - input_shape.push(v); - } - if i == 0 && d.is_none() { - input_dim_prefix = true; - } - } - if input_shape.is_empty() { - return Err(Onnx(OnnxInvalidInputShapeErr)); - } - let mut output_shape = vec![]; - for (j, d) in session.outputs[0].dimensions().enumerate() { - if let Some(v) = d { - output_shape.push(v); - } - if j == 0 && d.is_none() { - output_dim_prefix = true; - } - } - if output_shape.is_empty() { - return Err(Onnx(OnnxInvalidInputShapeErr)); - } - let input_type = session.inputs[0].input_type; - let return_type = session.outputs[0].output_type; - - if input_dim_prefix { - input_shape.insert(0, 1); - } - if output_dim_prefix { - output_shape.insert(0, 1); - } - - match input_type { - TensorElementDataType::Float32 => { - warn!("Precision loss is expected due to conversion to f32"); - let mut input_array = vec![]; - for field in input_values { - if let Field::Float(v) = field { - let num = match f32::from_f64(*v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::Int(v) = field { - let num = match f32::from_i64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::I128(v) = field { - let num = match f32::from_i128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::UInt(v) = field { - let num = match f32::from_u64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::U128(v) = field { - let num = match f32::from_u128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else { - return Err(Onnx(OnnxInputDataMismatchErr(input_type, field))); - } - } - let array = ndarray::CowArray::from( - Array::from_shape_vec(input_shape.clone(), input_array) - .map_err(|e| Onnx(OnnxShapeErr(e)))? - .into_dyn(), - ); - let input_tensor_values = - vec![Value::from_array(session.allocator(), &array) - .map_err(|e| Onnx(OnnxOrtErr(e)))?]; - let outputs: Vec = session - .run(input_tensor_values) - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - let output = outputs[0].borrow(); - - // number of output validation - // assert_eq!(outputs.len(), 1); - onnx_output_to_dozer(return_type, output, output_shape, output_dim_prefix) - } - TensorElementDataType::Float64 => { - let mut input_array = vec![]; - for field in input_values { - if let Field::Float(v) = field { - input_array.push(*v); - } else if let Field::Int(v) = field { - let num = match f64::from_i64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::I128(v) = field { - let num = match f64::from_i128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::UInt(v) = field { - let num = match f64::from_u64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::U128(v) = field { - let num = match f64::from_u128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else { - return Err(Onnx(OnnxInputDataMismatchErr(input_type, field))); - } - } - let array = ndarray::CowArray::from( - Array::from_shape_vec(input_shape.clone(), input_array) - .map_err(|e| Onnx(OnnxShapeErr(e)))? - .into_dyn(), - ); - let input_tensor_values = - vec![Value::from_array(session.allocator(), &array) - .map_err(|e| Onnx(OnnxOrtErr(e)))?]; - let outputs: Vec = session - .run(input_tensor_values) - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - let output = outputs[0].borrow(); - - // number of output validation - assert_eq!(outputs.len(), 1); - onnx_output_to_dozer(return_type, output, output_shape, output_dim_prefix) - } - TensorElementDataType::Uint8 => { - warn!("Precision loss is expected due to conversion to u8"); - let mut input_array = vec![]; - for field in input_values { - if let Field::UInt(v) = field { - let num = match u8::from_u64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::U128(v) = field { - let num = match u8::from_u128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::Float(v) = field { - let num = match u8::from_f64(*v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::Int(v) = field { - let num = match u8::from_i64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::I128(v) = field { - let num = match u8::from_i128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else { - return Err(Onnx(OnnxInputDataMismatchErr(input_type, field))); - } - } - let array = ndarray::CowArray::from( - Array::from_shape_vec(input_shape.clone(), input_array) - .map_err(|e| Onnx(OnnxShapeErr(e)))? - .into_dyn(), - ); - let input_tensor_values = - vec![Value::from_array(session.allocator(), &array) - .map_err(|e| Onnx(OnnxOrtErr(e)))?]; - let outputs: Vec = session - .run(input_tensor_values) - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - let output = outputs[0].borrow(); - - // number of output validation - assert_eq!(outputs.len(), 1); - onnx_output_to_dozer(return_type, output, output_shape, output_dim_prefix) - } - TensorElementDataType::Uint16 => { - let mut input_array = vec![]; - for field in input_values { - warn!("Precision loss is expected due to conversion to u16"); - if let Field::UInt(v) = field { - let num = match u16::from_u64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::U128(v) = field { - let num = match u16::from_u128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::Float(v) = field { - let num = match u16::from_f64(*v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::Int(v) = field { - let num = match u16::from_i64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::I128(v) = field { - let num = match u16::from_i128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else { - return Err(Onnx(OnnxInputDataMismatchErr(input_type, field))); - } - } - let array = ndarray::CowArray::from( - Array::from_shape_vec(input_shape.clone(), input_array) - .map_err(|e| Onnx(OnnxShapeErr(e)))? - .into_dyn(), - ); - let input_tensor_values = - vec![Value::from_array(session.allocator(), &array) - .map_err(|e| Onnx(OnnxOrtErr(e)))?]; - let outputs: Vec = session - .run(input_tensor_values) - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - let output = outputs[0].borrow(); - - // number of output validation - assert_eq!(outputs.len(), 1); - onnx_output_to_dozer(return_type, output, output_shape, output_dim_prefix) - } - TensorElementDataType::Uint32 => { - warn!("Precision loss is expected due to conversion to u32"); - let mut input_array = vec![]; - for field in input_values { - if let Field::UInt(v) = field { - let num = match u32::from_u64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::U128(v) = field { - let num = match u32::from_u128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::Float(v) = field { - let num = match u32::from_f64(*v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::Int(v) = field { - let num = match u32::from_i64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::I128(v) = field { - let num = match u32::from_i128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else { - return Err(Onnx(OnnxInputDataMismatchErr(input_type, field))); - } - } - let array = ndarray::CowArray::from( - Array::from_shape_vec(input_shape.clone(), input_array) - .map_err(|e| Onnx(OnnxShapeErr(e)))? - .into_dyn(), - ); - let input_tensor_values = - vec![Value::from_array(session.allocator(), &array) - .map_err(|e| Onnx(OnnxOrtErr(e)))?]; - let outputs: Vec = session - .run(input_tensor_values) - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - let output = outputs[0].borrow(); - - // number of output validation - assert_eq!(outputs.len(), 1); - onnx_output_to_dozer(return_type, output, output_shape, output_dim_prefix) - } - TensorElementDataType::Uint64 => { - let mut input_array = vec![]; - for field in input_values { - if let Field::UInt(v) = field { - input_array.push(v); - } else if let Field::U128(v) = field { - let num = match u64::from_u128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::Float(v) = field { - let num = match u64::from_f64(*v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::Int(v) = field { - let num = match u64::from_i64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::I128(v) = field { - let num = match u64::from_i128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else { - return Err(Onnx(OnnxInputDataMismatchErr(input_type, field))); - } - } - let array = ndarray::CowArray::from( - Array::from_shape_vec(input_shape.clone(), input_array) - .map_err(|e| Onnx(OnnxShapeErr(e)))? - .into_dyn(), - ); - let input_tensor_values = - vec![Value::from_array(session.allocator(), &array) - .map_err(|e| Onnx(OnnxOrtErr(e)))?]; - let outputs: Vec = session - .run(input_tensor_values) - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - let output = outputs[0].borrow(); - - // number of output validation - assert_eq!(outputs.len(), 1); - onnx_output_to_dozer(return_type, output, output_shape, output_dim_prefix) - } - TensorElementDataType::Int8 => { - warn!("Precision loss is expected due to conversion to i8"); - let mut input_array = vec![]; - for field in input_values { - if let Field::Int(v) = field { - let num = match i8::from_i64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::I128(v) = field { - let num = match i8::from_i128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::UInt(v) = field { - let num = match i8::from_u64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::U128(v) = field { - let num = match i8::from_u128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::Float(v) = field { - let num = match i8::from_f64(*v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else { - return Err(Onnx(OnnxInputDataMismatchErr(input_type, field))); - } - } - let array = ndarray::CowArray::from( - Array::from_shape_vec(input_shape.clone(), input_array) - .map_err(|e| Onnx(OnnxShapeErr(e)))? - .into_dyn(), - ); - let input_tensor_values = - vec![Value::from_array(session.allocator(), &array) - .map_err(|e| Onnx(OnnxOrtErr(e)))?]; - let outputs: Vec = session - .run(input_tensor_values) - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - let output = outputs[0].borrow(); - - // number of output validation - assert_eq!(outputs.len(), 1); - onnx_output_to_dozer(return_type, output, output_shape, output_dim_prefix) - } - TensorElementDataType::Int16 => { - warn!("Precision loss is expected due to conversion to i16"); - let mut input_array = vec![]; - for field in input_values { - if let Field::Int(v) = field { - let num = match i16::from_i64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::I128(v) = field { - let num = match i16::from_i128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::UInt(v) = field { - let num = match i16::from_u64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::U128(v) = field { - let num = match i16::from_u128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::Float(v) = field { - let num = match i16::from_f64(*v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else { - return Err(Onnx(OnnxInputDataMismatchErr(input_type, field))); - } - } - let array = ndarray::CowArray::from( - Array::from_shape_vec(input_shape.clone(), input_array) - .map_err(|e| Onnx(OnnxShapeErr(e)))? - .into_dyn(), - ); - let input_tensor_values = - vec![Value::from_array(session.allocator(), &array) - .map_err(|e| Onnx(OnnxOrtErr(e)))?]; - let outputs: Vec = session - .run(input_tensor_values) - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - let output = outputs[0].borrow(); - - // number of output validation - assert_eq!(outputs.len(), 1); - onnx_output_to_dozer(return_type, output, output_shape, output_dim_prefix) - } - TensorElementDataType::Int32 => { - warn!("Precision loss is expected due to conversion to i32"); - let mut input_array = vec![]; - for field in input_values { - if let Field::Int(v) = field { - let num = match i32::from_i64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::I128(v) = field { - let num = match i32::from_i128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::UInt(v) = field { - let num = match i32::from_u64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::U128(v) = field { - let num = match i32::from_u128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::Float(v) = field { - let num = match i32::from_f64(*v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else { - return Err(Onnx(OnnxInputDataMismatchErr(input_type, field))); - } - } - let array = ndarray::CowArray::from( - Array::from_shape_vec(input_shape.clone(), input_array) - .map_err(|e| Onnx(OnnxShapeErr(e)))? - .into_dyn(), - ); - let input_tensor_values = - vec![Value::from_array(session.allocator(), &array) - .map_err(|e| Onnx(OnnxOrtErr(e)))?]; - let outputs: Vec = session - .run(input_tensor_values) - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - let output = outputs[0].borrow(); - - // number of output validation - assert_eq!(outputs.len(), 1); - onnx_output_to_dozer(return_type, output, output_shape, output_dim_prefix) - } - TensorElementDataType::Int64 => { - let mut input_array = vec![]; - for field in input_values { - if let Field::Int(v) = field { - input_array.push(v); - } else if let Field::I128(v) = field { - let num = match i64::from_i128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::UInt(v) = field { - let num = match i64::from_u64(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::U128(v) = field { - let num = match i64::from_u128(v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else if let Field::Float(v) = field { - let num = match i64::from_f64(*v) { - Some(val) => val, - None => { - return Err(Onnx(InputArgumentOverflow(field.clone(), return_type))) - } - }; - input_array.push(num); - } else { - return Err(Onnx(OnnxInputDataMismatchErr(input_type, field))); - } - } - let array = ndarray::CowArray::from( - Array::from_shape_vec(input_shape.clone(), input_array) - .map_err(|e| Onnx(OnnxShapeErr(e)))? - .into_dyn(), - ); - let input_tensor_values = - vec![Value::from_array(session.allocator(), &array) - .map_err(|e| Onnx(OnnxOrtErr(e)))?]; - let outputs: Vec = session - .run(input_tensor_values) - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - let output = outputs[0].borrow(); - - // number of output validation - assert_eq!(outputs.len(), 1); - onnx_output_to_dozer(return_type, output, output_shape, output_dim_prefix) - } - TensorElementDataType::String => { - let mut input_array = vec![]; - for field in input_values { - if let Field::String(v) = field { - input_array.push(v); - } else if let Field::Text(v) = field { - input_array.push(v); - } else { - return Err(Onnx(OnnxInputDataMismatchErr(input_type, field))); - } - } - let array = ndarray::CowArray::from( - Array::from_shape_vec(input_shape.clone(), input_array) - .map_err(|e| Onnx(OnnxShapeErr(e)))? - .into_dyn(), - ); - let input_tensor_values = - vec![Value::from_array(session.allocator(), &array) - .map_err(|e| Onnx(OnnxOrtErr(e)))?]; - let outputs: Vec = session - .run(input_tensor_values) - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - let output = outputs[0].borrow(); - - // number of output validation - assert_eq!(outputs.len(), 1); - onnx_output_to_dozer(return_type, output, output_shape, output_dim_prefix) - } - TensorElementDataType::Bool => { - let mut input_array = vec![]; - for field in input_values { - if let Field::Boolean(v) = field { - input_array.push(v); - } else { - return Err(Onnx(OnnxInputDataMismatchErr(input_type, field))); - } - } - let array = ndarray::CowArray::from( - Array::from_shape_vec(input_shape.clone(), input_array) - .map_err(|e| Onnx(OnnxShapeErr(e)))? - .into_dyn(), - ); - let input_tensor_values = - vec![Value::from_array(session.allocator(), &array) - .map_err(|e| Onnx(OnnxOrtErr(e)))?]; - let outputs: Vec = session - .run(input_tensor_values) - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - let output = outputs[0].borrow(); - - // number of output validation - assert_eq!(outputs.len(), 1); - onnx_output_to_dozer(return_type, output, output_shape, output_dim_prefix) - } - _ => Err(Onnx(OnnxNotSupportedDataTypeErr(input_type))), - } -} - -fn onnx_output_to_dozer( - return_type: TensorElementDataType, - output: &Value, - output_shape: Vec, - output_dim_prefix: bool, -) -> Result { - if output_dim_prefix { - match return_type { - TensorElementDataType::Float16 => { - let output_array_view = output - .try_extract::() - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - assert_eq!(output_array_view.view().shape(), output_shape); - let view = output_array_view.view(); - match view.deref().to_slice() { - Some(v) => { - let result = v[0].into(); - Ok(Field::Float(OrderedFloat(result))) - } - None => Ok(Field::Null), - } - } - TensorElementDataType::Float32 => { - let output_array_view = output - .try_extract::() - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - assert_eq!(output_array_view.view().shape(), output_shape); - let view = output_array_view.view(); - match view.deref().to_slice() { - Some(v) => { - let result = v[0].into(); - Ok(Field::Float(OrderedFloat(result))) - } - None => Ok(Field::Null), - } - } - TensorElementDataType::Float64 => { - let output_array_view = output - .try_extract::() - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - assert_eq!(output_array_view.view().shape(), output_shape); - let view = output_array_view.view(); - match view.deref().to_slice() { - Some(v) => { - let result = v[0]; - Ok(Field::Float(OrderedFloat(result))) - } - None => Ok(Field::Null), - } - } - _ => Err(Onnx(OnnxNotSupportedDataTypeErr(return_type))), - } - } else { - match return_type { - TensorElementDataType::Float16 => { - let output_array_view = output - .try_extract::() - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - assert_eq!(output_array_view.view().shape(), output_shape); - Ok(Field::Float(OrderedFloat( - output_array_view.view().deref()[0].into(), - ))) - } - TensorElementDataType::Float32 => { - let output_array_view = output - .try_extract::() - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - assert_eq!(output_array_view.view().shape(), output_shape); - let view = output_array_view.view(); - let result = view.deref()[0].into(); - Ok(Field::Float(OrderedFloat(result))) - } - TensorElementDataType::Float64 => { - let output_array_view = output - .try_extract::() - .map_err(|e| Onnx(OnnxOrtErr(e)))?; - assert_eq!(output_array_view.view().shape(), output_shape); - Ok(Field::Float(OrderedFloat( - output_array_view.view().deref()[0], - ))) - } - _ => Err(Onnx(OnnxNotSupportedDataTypeErr(return_type))), - } - } -} diff --git a/dozer-sql/expression/src/onnx/utils.rs b/dozer-sql/expression/src/onnx/utils.rs deleted file mode 100644 index 126ac2dc12..0000000000 --- a/dozer-sql/expression/src/onnx/utils.rs +++ /dev/null @@ -1,132 +0,0 @@ -use super::error::Error::{ - ColumnNotFound, NonColumnArgFound, OnnxInputDataTypeMismatchErr, OnnxInputShapeErr, - OnnxInputSizeErr, OnnxNotSupportedDataTypeErr, OnnxOutputShapeErr, -}; -use crate::error::Error::{self, Onnx}; -use crate::execution::Expression; -use dozer_types::arrow::datatypes::ArrowNativeTypeOp; -use dozer_types::types::{FieldType, Schema}; -use ort::session::{Input, Output}; -use ort::tensor::TensorElementDataType; - -pub fn onnx_input_validation( - schema: &Schema, - args: &[Expression], - inputs: &[Input], -) -> Result<(), Error> { - // 1. number of input & input shape check - if inputs.len() != 1 { - return Err(Onnx(OnnxInputSizeErr(inputs.len()))); - } - let mut flattened = 1_u32; - let dim = inputs[0].dimensions.clone(); - for d in dim { - match d { - None => continue, - Some(v) => { - flattened = flattened.mul_wrapping(v); - } - } - } - if flattened as usize != args.len() || inputs.len() != 1 { - return Err(Onnx(OnnxInputShapeErr(flattened as usize, args.len()))); - } - // 2. input datatype check - for (input, arg) in inputs.iter().zip(args) { - match arg { - Expression::Column { index } => match schema.fields.get(*index) { - Some(def) => match input.input_type { - TensorElementDataType::Float32 | TensorElementDataType::Float64 => { - if def.typ != FieldType::Float { - return Err(Onnx(OnnxInputDataTypeMismatchErr( - input.input_type, - def.typ, - ))); - } - } - TensorElementDataType::Uint8 - | TensorElementDataType::Uint16 - | TensorElementDataType::Uint32 - | TensorElementDataType::Uint64 => { - if def.typ != FieldType::UInt && def.typ != FieldType::U128 { - return Err(Onnx(OnnxInputDataTypeMismatchErr( - input.input_type, - def.typ, - ))); - } - } - TensorElementDataType::Int8 - | TensorElementDataType::Int16 - | TensorElementDataType::Int32 - | TensorElementDataType::Int64 => { - if def.typ != FieldType::Int && def.typ != FieldType::I128 { - return Err(Onnx(OnnxInputDataTypeMismatchErr( - input.input_type, - def.typ, - ))); - } - } - TensorElementDataType::String => { - if def.typ != FieldType::String && def.typ != FieldType::Text { - return Err(Onnx(OnnxInputDataTypeMismatchErr( - input.input_type, - def.typ, - ))); - } - } - TensorElementDataType::Bool => { - if def.typ != FieldType::Boolean { - return Err(Onnx(OnnxInputDataTypeMismatchErr( - input.input_type, - def.typ, - ))); - } - } - _ => return Err(Onnx(OnnxNotSupportedDataTypeErr(input.input_type))), - }, - None => return Err(Onnx(ColumnNotFound(arg.clone()))), - }, - _ => return Err(Onnx(NonColumnArgFound(arg.clone()))), - } - } - Ok(()) -} - -pub fn onnx_output_validation(outputs: &Vec) -> Result<(), Error> { - // 1. number of output & output shape check - let mut flattened = 1_u32; - for output_shape in outputs { - let dim = output_shape.dimensions.clone(); - for d in dim { - match d { - None => continue, - Some(v) => { - flattened = flattened.mul_wrapping(v); - } - } - } - } - // output needs to be 1d single dim tensor - if flattened as usize != 1_usize { - return Err(Onnx(OnnxOutputShapeErr(flattened as usize, 1_usize))); - } - // 2. output datatype check - for output in outputs { - match output.output_type { - TensorElementDataType::Float32 - | TensorElementDataType::Float64 - | TensorElementDataType::Uint8 - | TensorElementDataType::Uint16 - | TensorElementDataType::Uint32 - | TensorElementDataType::Uint64 - | TensorElementDataType::Int8 - | TensorElementDataType::Int16 - | TensorElementDataType::Int32 - | TensorElementDataType::Int64 - | TensorElementDataType::String - | TensorElementDataType::Bool => continue, - _ => return Err(Onnx(OnnxNotSupportedDataTypeErr(output.output_type))), - } - } - Ok(()) -} diff --git a/dozer-sql/expression/src/operator.rs b/dozer-sql/expression/src/operator.rs deleted file mode 100644 index 55889dbfda..0000000000 --- a/dozer-sql/expression/src/operator.rs +++ /dev/null @@ -1,110 +0,0 @@ -use crate::comparison::*; -use crate::error::Error; -use crate::execution::Expression; -use crate::logical::*; -use crate::mathematical::*; -use dozer_types::types::Record; -use dozer_types::types::{Field, Schema}; -use std::fmt::{Display, Formatter}; - -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Hash)] -pub enum UnaryOperatorType { - Not, - Plus, - Minus, -} - -impl Display for UnaryOperatorType { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - match self { - UnaryOperatorType::Not => f.write_str("!"), - UnaryOperatorType::Plus => f.write_str("+"), - UnaryOperatorType::Minus => f.write_str("-"), - } - } -} - -impl UnaryOperatorType { - pub fn evaluate( - &self, - schema: &Schema, - value: &mut Expression, - record: &Record, - ) -> Result { - match self { - UnaryOperatorType::Not => evaluate_not(schema, value, record), - UnaryOperatorType::Plus => evaluate_plus(schema, value, record), - UnaryOperatorType::Minus => evaluate_minus(schema, value, record), - } - } -} - -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Hash)] -pub enum BinaryOperatorType { - // Comparison - Eq, - Ne, - Gt, - Gte, - Lt, - Lte, - - // Logical - And, - Or, - - // Mathematical - Add, - Sub, - Mul, - Div, - Mod, -} - -impl Display for BinaryOperatorType { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - match self { - BinaryOperatorType::Eq => f.write_str("="), - BinaryOperatorType::Ne => f.write_str("!="), - BinaryOperatorType::Gt => f.write_str(">"), - BinaryOperatorType::Gte => f.write_str(">="), - BinaryOperatorType::Lt => f.write_str("<"), - BinaryOperatorType::Lte => f.write_str("<="), - BinaryOperatorType::And => f.write_str(" AND "), - BinaryOperatorType::Or => f.write_str(" OR "), - BinaryOperatorType::Add => f.write_str("+"), - BinaryOperatorType::Sub => f.write_str("-"), - BinaryOperatorType::Mul => f.write_str("*"), - BinaryOperatorType::Div => f.write_str("/"), - BinaryOperatorType::Mod => f.write_str("%"), - } - } -} - -impl BinaryOperatorType { - pub fn evaluate( - &self, - schema: &Schema, - left: &mut Expression, - right: &mut Expression, - record: &Record, - ) -> Result { - match self { - BinaryOperatorType::Eq => evaluate_eq(schema, left, right, record), - BinaryOperatorType::Ne => evaluate_ne(schema, left, right, record), - BinaryOperatorType::Gt => evaluate_gt(schema, left, right, record), - BinaryOperatorType::Gte => evaluate_gte(schema, left, right, record), - BinaryOperatorType::Lt => evaluate_lt(schema, left, right, record), - BinaryOperatorType::Lte => evaluate_lte(schema, left, right, record), - - BinaryOperatorType::And => evaluate_and(schema, left, right, record), - BinaryOperatorType::Or => evaluate_or(schema, left, right, record), - - BinaryOperatorType::Add => evaluate_add(schema, left, right, record), - BinaryOperatorType::Sub => evaluate_sub(schema, left, right, record), - BinaryOperatorType::Mul => evaluate_mul(schema, left, right, record), - BinaryOperatorType::Div => evaluate_div(schema, left, right, record), - BinaryOperatorType::Mod => evaluate_mod(schema, left, right, record), - } - } -} diff --git a/dozer-sql/expression/src/python_udf.rs b/dozer-sql/expression/src/python_udf.rs deleted file mode 100644 index 25f5394f04..0000000000 --- a/dozer-sql/expression/src/python_udf.rs +++ /dev/null @@ -1,81 +0,0 @@ -use crate::execution::Expression; -use dozer_types::ordered_float::OrderedFloat; -use dozer_types::pyo3::types::PyTuple; -use dozer_types::pyo3::Python; -use dozer_types::thiserror::{self, Error}; -use dozer_types::types::Record; -use dozer_types::types::{Field, FieldType, Schema}; -use std::env; -use std::path::PathBuf; - -const MODULE_NAME: &str = "python_udf"; - -#[derive(Debug, Error)] -pub enum Error { - #[error( - "Python UDF must have a return type. The syntax is: function_name(arguments)" - )] - MissingReturnType, - #[error("Missing 'VIRTUAL_ENV' environment var")] - MissingVirtualEnv, - #[error("PyO3 error: {0}")] - PyO3(#[from] dozer_types::pyo3::PyErr), - #[error("Unsupported return type: {0}")] - UnsupportedReturnType(FieldType), - #[error("Failed to parse return type: {0}")] - FailedToParseReturnType(String), -} - -pub fn evaluate_py_udf( - schema: &Schema, - name: &str, - args: &mut [Expression], - return_type: &FieldType, - record: &Record, -) -> Result { - let values = args - .iter_mut() - .map(|arg| arg.evaluate(record, schema)) - .collect::, crate::error::Error>>()?; - - // Get the path of the Python interpreter in your virtual environment - let env_path = env::var("VIRTUAL_ENV").map_err(|_| Error::MissingVirtualEnv)?; - let py_path = format!("{env_path}/bin/python"); - // Set the `PYTHON_SYS_EXECUTABLE` environment variable - env::set_var("PYTHON_SYS_EXECUTABLE", py_path); - - Python::with_gil(|py| -> Result { - // Get the directory containing the module - let module_dir = PathBuf::from(env_path); - // Import the `sys` module and append the module directory to the system path - let sys = py.import("sys")?; - let path = sys.getattr("path")?; - path.call_method1("append", (module_dir.to_string_lossy(),))?; - - let module = py.import(MODULE_NAME)?; - let function = module.getattr(name)?; - - let args = PyTuple::new(py, values); - let res = function.call1(args)?; - - Ok(match return_type { - FieldType::UInt => Field::UInt(res.extract::()?), - FieldType::U128 => Field::U128(res.extract::()?), - FieldType::Int => Field::Int(res.extract::()?), - FieldType::Int8 => Field::Int8(res.extract::()?), - FieldType::I128 => Field::I128(res.extract::()?), - FieldType::Float => Field::Float(OrderedFloat::from(res.extract::()?)), - FieldType::Boolean => Field::Boolean(res.extract::()?), - FieldType::String => Field::String(res.extract::()?), - FieldType::Text => Field::Text(res.extract::()?), - FieldType::Binary => Field::Binary(res.extract::>()?), - FieldType::Decimal - | FieldType::Date - | FieldType::Timestamp - | FieldType::Point - | FieldType::Duration - | FieldType::Json => return Err(Error::UnsupportedReturnType(*return_type)), - }) - }) - .map_err(Into::into) -} diff --git a/dozer-sql/expression/src/scalar/common.rs b/dozer-sql/expression/src/scalar/common.rs deleted file mode 100644 index 1698f11809..0000000000 --- a/dozer-sql/expression/src/scalar/common.rs +++ /dev/null @@ -1,199 +0,0 @@ -use crate::arg_utils::{validate_num_arguments, validate_one_argument, validate_two_arguments}; -use crate::error::Error; -use crate::execution::{Expression, ExpressionType}; -use crate::scalar::field::evaluate_nvl; -use crate::scalar::number::{evaluate_abs, evaluate_round}; -use crate::scalar::string::{ - evaluate_concat, evaluate_length, evaluate_to_char, evaluate_ucase, validate_concat, - validate_ucase, -}; -use dozer_types::types::Record; -use dozer_types::types::{Field, FieldType, Schema}; -use std::fmt::{Display, Formatter}; - -use super::field::{evaluate_decode, validate_decode}; -use super::string::{ - evaluate_chr, evaluate_replace, evaluate_substr, validate_replace, validate_substr, -}; - -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Hash)] -pub enum ScalarFunctionType { - Abs, - Round, - Ucase, - Concat, - Length, - ToChar, - Chr, - Substr, - Nvl, - Replace, - Decode, -} - -impl Display for ScalarFunctionType { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - match self { - ScalarFunctionType::Abs => f.write_str("ABS"), - ScalarFunctionType::Round => f.write_str("ROUND"), - ScalarFunctionType::Ucase => f.write_str("UCASE"), - ScalarFunctionType::Concat => f.write_str("CONCAT"), - ScalarFunctionType::Length => f.write_str("LENGTH"), - ScalarFunctionType::ToChar => f.write_str("TO_CHAR"), - ScalarFunctionType::Chr => f.write_str("CHR"), - ScalarFunctionType::Substr => f.write_str("SUBSTR"), - ScalarFunctionType::Nvl => f.write_str("NVL"), - ScalarFunctionType::Replace => f.write_str("REPLACE"), - ScalarFunctionType::Decode => f.write_str("DECODE"), - } - } -} - -pub(crate) fn get_scalar_function_type( - function: &ScalarFunctionType, - args: &[Expression], - schema: &Schema, -) -> Result { - match function { - ScalarFunctionType::Abs => validate_one_argument(args, schema, ScalarFunctionType::Abs), - ScalarFunctionType::Round => { - let return_type = if args.len() == 1 { - validate_one_argument(args, schema, ScalarFunctionType::Round)?.return_type - } else { - validate_two_arguments(args, schema, ScalarFunctionType::Round)? - .0 - .return_type - }; - Ok(ExpressionType::new( - return_type, - true, - dozer_types::types::SourceDefinition::Dynamic, - false, - )) - } - ScalarFunctionType::Ucase => { - validate_num_arguments(1..2, args.len(), ScalarFunctionType::Ucase)?; - validate_ucase(&args[0], schema) - } - ScalarFunctionType::Concat => validate_concat(args, schema), - ScalarFunctionType::Length => Ok(ExpressionType::new( - FieldType::UInt, - false, - dozer_types::types::SourceDefinition::Dynamic, - false, - )), - ScalarFunctionType::ToChar => { - if args.len() == 1 { - validate_one_argument(args, schema, ScalarFunctionType::ToChar) - } else { - Ok(validate_two_arguments(args, schema, ScalarFunctionType::ToChar)?.0) - } - } - ScalarFunctionType::Chr => validate_one_argument(args, schema, ScalarFunctionType::Chr), - ScalarFunctionType::Substr => validate_substr(args, schema), - ScalarFunctionType::Nvl => { - Ok(validate_two_arguments(args, schema, ScalarFunctionType::Nvl)?.0) - } - ScalarFunctionType::Replace => validate_replace(args, schema), - ScalarFunctionType::Decode => validate_decode(args, schema), - } -} - -impl ScalarFunctionType { - pub fn new(name: &str) -> Option { - match name { - "abs" => Some(ScalarFunctionType::Abs), - "round" => Some(ScalarFunctionType::Round), - "ucase" => Some(ScalarFunctionType::Ucase), - "concat" => Some(ScalarFunctionType::Concat), - "decode" => Some(ScalarFunctionType::Decode), - "length" => Some(ScalarFunctionType::Length), - "to_char" => Some(ScalarFunctionType::ToChar), - "chr" => Some(ScalarFunctionType::Chr), - "substr" => Some(ScalarFunctionType::Substr), - "replace" => Some(ScalarFunctionType::Replace), - "nvl" => Some(ScalarFunctionType::Nvl), - _ => None, - } - } - - pub(crate) fn evaluate( - &self, - schema: &Schema, - args: &mut [Expression], - record: &Record, - ) -> Result { - match self { - ScalarFunctionType::Abs => { - validate_num_arguments(1..2, args.len(), ScalarFunctionType::Abs)?; - evaluate_abs(schema, &mut args[0], record) - } - ScalarFunctionType::Round => { - validate_num_arguments(1..3, args.len(), ScalarFunctionType::Round)?; - let (arg0, arg1) = args.split_at_mut(1); - evaluate_round(schema, &mut arg0[0], arg1.get_mut(0), record) - } - ScalarFunctionType::Ucase => { - validate_num_arguments(1..2, args.len(), ScalarFunctionType::Ucase)?; - evaluate_ucase(schema, &mut args[0], record) - } - ScalarFunctionType::Concat => evaluate_concat(schema, args, record), - ScalarFunctionType::Length => { - validate_num_arguments(1..2, args.len(), ScalarFunctionType::Length)?; - evaluate_length(schema, &mut args[0], record) - } - ScalarFunctionType::ToChar => { - validate_num_arguments(2..3, args.len(), ScalarFunctionType::ToChar)?; - let (arg0, arg1) = args.split_at_mut(1); - evaluate_to_char(schema, &mut arg0[0], &mut arg1[0], record) - } - ScalarFunctionType::Chr => { - validate_one_argument(args, schema, ScalarFunctionType::Chr)?; - evaluate_chr(schema, &mut args[0], record) - } - ScalarFunctionType::Substr => { - validate_num_arguments(2..3, args.len(), ScalarFunctionType::Substr)?; - let mut arg2 = args.get(2).map(|arg| Box::new(arg.clone())); - evaluate_substr( - schema, - &mut args[0].clone(), - &mut args[1].clone(), - &mut arg2, - record, - ) - } - ScalarFunctionType::Nvl => { - validate_two_arguments(args, schema, ScalarFunctionType::Nvl)?; - evaluate_nvl(schema, &mut args[0].clone(), &mut args[1].clone(), record) - } - ScalarFunctionType::Replace => { - validate_replace(args, schema)?; - evaluate_replace( - schema, - &mut args[0].clone(), - &mut args[1].clone(), - &mut args[2].clone(), - record, - ) - } - ScalarFunctionType::Decode => { - validate_decode(args, schema)?; - - let (arg0, results) = args.split_at_mut(1); - let (results, default) = if results.len() % 2 == 0 { - results.split_at_mut(results.len() - 1) - } else { - results.split_at_mut(results.len()) - }; - - let default = if default.is_empty() { - None - } else { - Some(default[0].clone()) - }; - - evaluate_decode(schema, &mut arg0[0], results, default, record) - } - } - } -} diff --git a/dozer-sql/expression/src/scalar/field.rs b/dozer-sql/expression/src/scalar/field.rs deleted file mode 100644 index 6ce434c706..0000000000 --- a/dozer-sql/expression/src/scalar/field.rs +++ /dev/null @@ -1,111 +0,0 @@ -use crate::error::Error; -use crate::execution::{Expression, ExpressionType}; -use dozer_types::types::Record; -use dozer_types::types::{Field, Schema}; - -pub(crate) fn evaluate_nvl( - schema: &Schema, - arg: &mut Expression, - replacement: &mut Expression, - record: &Record, -) -> Result { - let arg_field = arg.evaluate(record, schema)?; - let replacement_field = replacement.evaluate(record, schema)?; - if replacement_field.as_string().is_some() && arg_field == Field::Null { - Ok(replacement_field) - } else { - Ok(arg_field) - } -} - -pub fn validate_decode(args: &[Expression], schema: &Schema) -> Result { - if args.len() < 3 { - return Err(Error::InvalidNumberOfArguments { - function_name: "decode".to_string(), - expected: 3..usize::MAX, - actual: args.len(), - }); - } - - let ret_type = args[2].get_type(schema)?.return_type; - - Ok(ExpressionType::new( - ret_type, - false, - dozer_types::types::SourceDefinition::Dynamic, - false, - )) -} - -pub(crate) fn evaluate_decode( - schema: &Schema, - arg: &mut Expression, - results: &mut [Expression], - default: Option, - record: &Record, -) -> Result { - let arg_field = arg.evaluate(record, schema)?; - - for chunk in results.chunks_exact_mut(2) { - let coded = &mut chunk[0]; - let coded_field = coded.evaluate(record, schema)?; - let decoded = &mut chunk[1]; - - if coded_field == arg_field { - return decoded.evaluate(record, schema); - } - } - - if let Some(mut default) = default { - default.evaluate(record, schema) - } else { - Ok(Field::Null) - } -} - -#[cfg(test)] -mod tests { - use dozer_types::types::{Field, Record, Schema}; - - use crate::{execution::Expression, scalar::field::evaluate_decode}; - - #[test] - fn test_decode() { - let row = Record::new(vec![]); - let mut value = Box::new(Expression::Literal(Field::Int(2))); - let mut results = [ - Expression::Literal(Field::Int(1)), - Expression::Literal(Field::String("Southlake".to_owned())), - Expression::Literal(Field::Int(2)), - Expression::Literal(Field::String("San Francisco".to_owned())), - Expression::Literal(Field::Int(3)), - Expression::Literal(Field::String("New Jersey".to_owned())), - Expression::Literal(Field::Int(4)), - Expression::Literal(Field::String("Seattle".to_owned())), - ]; - let default = Some(Expression::Literal(Field::String( - "Non domestic".to_owned(), - ))); - let result = Field::String("San Francisco".to_owned()); - - assert_eq!( - evaluate_decode( - &Schema::default(), - &mut value, - &mut results, - default.clone(), - &row - ) - .unwrap(), - result - ); - - let mut value = Box::new(Expression::Literal(Field::Int(5))); - let result = Field::String("Non domestic".to_owned()); - - assert_eq!( - evaluate_decode(&Schema::default(), &mut value, &mut results, default, &row).unwrap(), - result - ); - } -} diff --git a/dozer-sql/expression/src/scalar/mod.rs b/dozer-sql/expression/src/scalar/mod.rs deleted file mode 100644 index 08358522c3..0000000000 --- a/dozer-sql/expression/src/scalar/mod.rs +++ /dev/null @@ -1,4 +0,0 @@ -pub mod common; -pub mod field; -pub mod number; -pub mod string; diff --git a/dozer-sql/expression/src/scalar/number.rs b/dozer-sql/expression/src/scalar/number.rs deleted file mode 100644 index 787d425907..0000000000 --- a/dozer-sql/expression/src/scalar/number.rs +++ /dev/null @@ -1,195 +0,0 @@ -use crate::error::Error; -use crate::execution::Expression; -use crate::scalar::common::ScalarFunctionType; -use dozer_types::ordered_float::OrderedFloat; -use dozer_types::types::Record; -use dozer_types::types::{Field, FieldType, Schema}; -use num_traits::{Float, ToPrimitive}; - -pub(crate) fn evaluate_abs( - schema: &Schema, - arg: &mut Expression, - record: &Record, -) -> Result { - let value = arg.evaluate(record, schema)?; - match value { - Field::UInt(u) => Ok(Field::UInt(u)), - Field::U128(u) => Ok(Field::U128(u)), - Field::Int(i) => Ok(Field::Int(i.abs())), - Field::Int8(i) => Ok(Field::Int8(i.abs())), - Field::I128(i) => Ok(Field::I128(i.abs())), - Field::Float(f) => Ok(Field::Float(f.abs())), - Field::Decimal(d) => Ok(Field::Decimal(d.abs())), - Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Date(_) - | Field::Timestamp(_) - | Field::Binary(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) - | Field::Null => Err(Error::InvalidFunctionArgument { - function_name: ScalarFunctionType::Abs.to_string(), - argument_index: 0, - argument: value, - }), - } -} - -pub(crate) fn evaluate_round( - schema: &Schema, - arg: &mut Expression, - decimals: Option<&mut Expression>, - record: &Record, -) -> Result { - let value = arg.evaluate(record, schema)?; - let mut places = 0; - if let Some(expression) = decimals { - let field = expression.evaluate(record, schema)?; - match field { - Field::UInt(u) => places = u as i32, - Field::U128(u) => places = u as i32, - Field::Int(i) => places = i as i32, - Field::Int8(i) => places = i as i32, - Field::I128(i) => places = i as i32, - Field::Float(f) => places = f.round().0 as i32, - Field::Decimal(d) => { - places = d - .to_i32() - .ok_or(Error::InvalidCast { - from: field, - to: FieldType::Decimal, - }) - .unwrap() - } - Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Date(_) - | Field::Timestamp(_) - | Field::Binary(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) - | Field::Null => {} // Truncate value to 0 decimals - } - } - let order = OrderedFloat(10.0_f64.powi(places)); - - match value { - Field::UInt(u) => Ok(Field::UInt(u)), - Field::U128(u) => Ok(Field::U128(u)), - Field::Int(i) => Ok(Field::Int(i)), - Field::Int8(i) => Ok(Field::Int8(i)), - Field::I128(i) => Ok(Field::I128(i)), - Field::Float(f) => Ok(Field::Float((f * order).round() / order)), - Field::Decimal(d) => Ok(Field::Decimal(d.round_dp(places as u32))), - Field::Null => Ok(Field::Null), - Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Date(_) - | Field::Timestamp(_) - | Field::Binary(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) => Err(Error::InvalidFunctionArgument { - function_name: ScalarFunctionType::Round.to_string(), - argument_index: 0, - argument: value, - }), - } -} - -#[cfg(test)] -mod tests { - use super::*; - - use dozer_types::ordered_float::OrderedFloat; - use dozer_types::types::Record; - use dozer_types::types::{Field, Schema}; - use proptest::prelude::*; - use std::ops::Neg; - use Expression::Literal; - - #[test] - fn test_abs() { - proptest!(ProptestConfig::with_cases(1000), |(i_num in 0i64..100000000i64, f_num in 0f64..100000000f64)| { - let row = Record::new(vec![]); - - let mut v = Box::new(Literal(Field::Int(i_num.neg()))); - assert_eq!( - evaluate_abs(&Schema::default(), &mut v, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int(i_num) - ); - - let row = Record::new(vec![]); - - let mut v = Box::new(Literal(Field::Float(OrderedFloat(f_num.neg())))); - assert_eq!( - evaluate_abs(&Schema::default(), &mut v, &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num)) - ); - }); - } - - #[test] - fn test_round() { - proptest!(ProptestConfig::with_cases(1000), |(i_num: i64, f_num: f64, i_pow: i32, f_pow: f32)| { - let row = Record::new(vec![]); - - let mut v = Box::new(Literal(Field::Int(i_num))); - let d = &mut Box::new(Literal(Field::Int(0))); - assert_eq!( - evaluate_round(&Schema::default(), &mut v, Some(d), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int(i_num) - ); - - let mut v = Box::new(Literal(Field::Float(OrderedFloat(f_num)))); - let d = &mut Box::new(Literal(Field::Int(0))); - assert_eq!( - evaluate_round(&Schema::default(), &mut v, Some(d), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num.round())) - ); - - let mut v = Box::new(Literal(Field::Float(OrderedFloat(f_num)))); - let d = &mut Box::new(Literal(Field::Int(i_pow as i64))); - let order = 10.0_f64.powi(i_pow); - assert_eq!( - evaluate_round(&Schema::default(), &mut v, Some(d), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat((f_num * order).round() / order)) - ); - - let mut v = Box::new(Literal(Field::Float(OrderedFloat(f_num)))); - let d = &mut Box::new(Literal(Field::Float(OrderedFloat(f_pow as f64)))); - let order = 10.0_f64.powi(f_pow.round() as i32); - assert_eq!( - evaluate_round(&Schema::default(), &mut v, Some(d), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat((f_num * order).round() / order)) - ); - - let mut v = Box::new(Literal(Field::Float(OrderedFloat(f_num)))); - let d = &mut Box::new(Literal(Field::String(f_pow.to_string()))); - assert_eq!( - evaluate_round(&Schema::default(), &mut v, Some(d), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(f_num.round())) - ); - - let mut v = Box::new(Literal(Field::Null)); - let d = &mut Box::new(Literal(Field::String(i_pow.to_string()))); - assert_eq!( - evaluate_round(&Schema::default(), &mut v, Some(d), &row) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Null - ); - }); - } -} diff --git a/dozer-sql/expression/src/scalar/string.rs b/dozer-sql/expression/src/scalar/string.rs deleted file mode 100644 index 7e557fe63c..0000000000 --- a/dozer-sql/expression/src/scalar/string.rs +++ /dev/null @@ -1,908 +0,0 @@ -use crate::error::Error; -use std::fmt::Write; -use std::fmt::{Display, Formatter}; - -use crate::execution::{Expression, ExpressionType}; - -use crate::arg_utils::{validate_arg_type, validate_num_arguments}; -use crate::scalar::common::ScalarFunctionType; - -use dozer_types::log; -use dozer_types::types::Record; -use dozer_types::types::{Field, FieldType, Schema}; -use like::{Escape, Like}; - -pub(crate) fn validate_ucase(arg: &Expression, schema: &Schema) -> Result { - validate_arg_type( - arg, - vec![FieldType::String, FieldType::Text], - schema, - ScalarFunctionType::Ucase, - 0, - ) -} - -pub fn evaluate_ucase( - schema: &Schema, - arg: &mut Expression, - record: &Record, -) -> Result { - let f = arg.evaluate(record, schema)?; - let v = f.to_string(); - let ret = v.to_uppercase(); - - Ok(match arg.get_type(schema)?.return_type { - FieldType::String => Field::String(ret), - FieldType::UInt - | FieldType::U128 - | FieldType::Int - | FieldType::Int8 - | FieldType::I128 - | FieldType::Float - | FieldType::Decimal - | FieldType::Boolean - | FieldType::Text - | FieldType::Date - | FieldType::Timestamp - | FieldType::Binary - | FieldType::Json - | FieldType::Point - | FieldType::Duration => Field::Text(ret), - }) -} - -pub fn validate_concat(args: &[Expression], schema: &Schema) -> Result { - let mut ret_type = FieldType::String; - for exp in args { - let r = validate_arg_type( - exp, - vec![FieldType::String, FieldType::Text], - schema, - ScalarFunctionType::Concat, - 0, - )?; - if matches!(r.return_type, FieldType::Text) { - ret_type = FieldType::Text; - } - } - Ok(ExpressionType::new( - ret_type, - false, - dozer_types::types::SourceDefinition::Dynamic, - false, - )) -} - -pub fn evaluate_concat( - schema: &Schema, - args: &mut [Expression], - record: &Record, -) -> Result { - let mut res_type = FieldType::String; - let mut res_vec: Vec = Vec::with_capacity(args.len()); - - for e in args { - if matches!(e.get_type(schema)?.return_type, FieldType::Text) { - res_type = FieldType::Text; - } - let f = e.evaluate(record, schema)?; - let val = f.to_string(); - res_vec.push(val); - } - - let res_str = res_vec.iter().fold(String::new(), |a, b| a + b.as_str()); - Ok(match res_type { - FieldType::Text => Field::Text(res_str), - FieldType::UInt - | FieldType::U128 - | FieldType::Int - | FieldType::Int8 - | FieldType::I128 - | FieldType::Float - | FieldType::Decimal - | FieldType::Boolean - | FieldType::String - | FieldType::Date - | FieldType::Timestamp - | FieldType::Binary - | FieldType::Json - | FieldType::Point - | FieldType::Duration => Field::String(res_str), - }) -} - -pub(crate) fn evaluate_length( - schema: &Schema, - arg0: &mut Expression, - record: &Record, -) -> Result { - let f0 = arg0.evaluate(record, schema)?; - let v0 = f0.to_string(); - Ok(Field::UInt(v0.len() as u64)) -} - -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum TrimType { - Trailing, - Leading, - Both, -} - -impl Display for TrimType { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - match self { - TrimType::Trailing => f.write_str("TRAILING "), - TrimType::Leading => f.write_str("LEADING "), - TrimType::Both => f.write_str("BOTH "), - } - } -} - -pub fn validate_trim(arg: &Expression, schema: &Schema) -> Result { - validate_arg_type( - arg, - vec![FieldType::String, FieldType::Text], - schema, - ScalarFunctionType::Concat, - 0, - ) -} - -pub fn evaluate_trim( - schema: &Schema, - arg: &mut Expression, - what: &mut Option>, - typ: &Option, - record: &Record, -) -> Result { - let arg_field = arg.evaluate(record, schema)?; - let arg_value = arg_field.to_string(); - - let v1: Vec<_> = match what { - Some(e) => { - let f = e.evaluate(record, schema)?; - f.to_string().chars().collect() - } - _ => vec![' '], - }; - - let retval = match typ { - Some(TrimType::Both) => arg_value.trim_matches::<&[char]>(&v1).to_string(), - Some(TrimType::Leading) => arg_value.trim_start_matches::<&[char]>(&v1).to_string(), - Some(TrimType::Trailing) => arg_value.trim_end_matches::<&[char]>(&v1).to_string(), - None => arg_value.trim_matches::<&[char]>(&v1).to_string(), - }; - - Ok(match arg.get_type(schema)?.return_type { - FieldType::String => Field::String(retval), - FieldType::UInt - | FieldType::U128 - | FieldType::Int - | FieldType::Int8 - | FieldType::I128 - | FieldType::Float - | FieldType::Decimal - | FieldType::Boolean - | FieldType::Text - | FieldType::Date - | FieldType::Timestamp - | FieldType::Binary - | FieldType::Json - | FieldType::Point - | FieldType::Duration => Field::Text(retval), - }) -} - -pub(crate) fn get_like_operator_type( - arg: &Expression, - pattern: &Expression, - schema: &Schema, -) -> Result { - validate_arg_type( - pattern, - vec![FieldType::String, FieldType::Text], - schema, - ScalarFunctionType::Concat, - 0, - )?; - - validate_arg_type( - arg, - vec![FieldType::String, FieldType::Text], - schema, - ScalarFunctionType::Concat, - 0, - ) -} - -pub fn evaluate_like( - schema: &Schema, - arg: &mut Expression, - pattern: &mut Expression, - escape: Option, - record: &Record, -) -> Result { - let arg_field = arg.evaluate(record, schema)?; - let arg_value = arg_field.to_string(); - let arg_string = arg_value.as_str(); - - let pattern_field = pattern.evaluate(record, schema)?; - let pattern_value = pattern_field.to_string(); - let pattern_string = pattern_value.as_str(); - - if let Some(escape_char) = escape { - let arg_escape = &arg_string.escape(&escape_char.to_string())?; - let result = - Like::::like(arg_escape.as_str(), pattern_string).map(Field::Boolean)?; - return Ok(result); - } - - let result = Like::::like(arg_string, pattern_string).map(Field::Boolean)?; - Ok(result) -} - -pub(crate) fn evaluate_to_char( - schema: &Schema, - arg: &mut Expression, - pattern: &mut Expression, - record: &Record, -) -> Result { - let arg_field = arg.evaluate(record, schema)?; - - let pattern_field = pattern.evaluate(record, schema)?; - let pattern_value = pattern_field.to_string(); - - let output = match arg_field { - Field::Timestamp(value) => value.format(pattern_value.as_str()).to_string(), - Field::Date(value) => { - let mut formatted = String::new(); - let format_result = write!(formatted, "{}", value.format(pattern_value.as_str())); - if format_result.is_ok() { - formatted - } else { - pattern_value - } - } - Field::Null => return Ok(Field::Null), - _ => { - return Err(Error::InvalidFunctionArgument { - function_name: "TO_CHAR".to_string(), - argument_index: 0, - argument: arg_field, - }); - } - }; - - Ok(Field::String(output)) -} - -pub(crate) fn evaluate_chr( - schema: &Schema, - arg: &mut Expression, - record: &Record, -) -> Result { - let value = arg.evaluate(record, schema)?; - match value { - Field::UInt(u) => Ok(Field::String((((u % 256) as u8) as char).to_string())), - Field::U128(u) => Ok(Field::String((((u % 256) as u8) as char).to_string())), - Field::Int(i) => { - if (0..256).contains(&i) { - Ok(Field::String(((i as u8) as char).to_string())) - } else if i > 255 { - log::warn!( - "Values greater than 255 are not supported in CHR function: {}", - i - ); - Ok(Field::String((((i % 256) as u8) as char).to_string())) - } else { - Err(Error::InvalidFunctionArgument { - function_name: ScalarFunctionType::Chr.to_string(), - argument_index: 0, - argument: value, - }) - } - } - Field::Int8(i) => Ok(Field::String(((i as u8) as char).to_string())), - Field::I128(i) => { - if i >= 0 { - Ok(Field::String((((i % 256) as u8) as char).to_string())) - } else { - Err(Error::InvalidFunctionArgument { - function_name: ScalarFunctionType::Chr.to_string(), - argument_index: 0, - argument: value, - }) - } - } - Field::Float(_) - | Field::Decimal(_) - | Field::Boolean(_) - | Field::String(_) - | Field::Text(_) - | Field::Date(_) - | Field::Timestamp(_) - | Field::Binary(_) - | Field::Json(_) - | Field::Point(_) - | Field::Duration(_) - | Field::Null => Err(Error::InvalidFunctionArgument { - function_name: ScalarFunctionType::Chr.to_string(), - argument_index: 0, - argument: value, - }), - } -} - -pub fn validate_substr(args: &[Expression], schema: &Schema) -> Result { - validate_num_arguments(2..4, args.len(), ScalarFunctionType::Substr)?; - - if args.len() == 2 { - validate_arg_type( - &args[0], - vec![FieldType::String, FieldType::Text], - schema, - ScalarFunctionType::Substr, - 0, - )?; - validate_arg_type( - &args[1], - vec![ - FieldType::UInt, - FieldType::U128, - FieldType::Int, - FieldType::I128, - ], - schema, - ScalarFunctionType::Substr, - 1, - )?; - } else { - validate_arg_type( - &args[0], - vec![FieldType::String, FieldType::Text], - schema, - ScalarFunctionType::Substr, - 0, - )?; - validate_arg_type( - &args[1], - vec![ - FieldType::UInt, - FieldType::U128, - FieldType::Int, - FieldType::I128, - ], - schema, - ScalarFunctionType::Substr, - 1, - )?; - validate_arg_type( - &args[2], - vec![ - FieldType::UInt, - FieldType::U128, - FieldType::Int, - FieldType::I128, - ], - schema, - ScalarFunctionType::Substr, - 2, - )?; - } - - let ret_type = FieldType::String; - - Ok(ExpressionType::new( - ret_type, - false, - dozer_types::types::SourceDefinition::Dynamic, - false, - )) -} - -pub(crate) fn evaluate_substr( - schema: &Schema, - arg: &mut Expression, - position: &mut Expression, - length: &mut Option>, - record: &Record, -) -> Result { - let arg_field = arg.evaluate(record, schema)?; - let arg_value = arg_field.to_string(); - - let position_field = position.evaluate(record, schema)?; - let position_value = position_field - .to_int() - .ok_or_else(|| Error::InvalidFunctionArgument { - function_name: "SUBSTR".to_string(), - argument_index: 1, - argument: position_field, - })?; - // 0 is treated as 1 - let position_value_normalized = if position_value == 0 { - 1 - } else { - position_value - }; - - let length_value = match length { - Some(length_expr) => { - let length_field = length_expr.evaluate(record, schema)?; - let length = length_field - .to_int() - .ok_or_else(|| Error::InvalidFunctionArgument { - function_name: "SUBSTR".to_string(), - argument_index: 2, - argument: length_field, - })?; - if length < 1 { - return Ok(Field::Null); - } - length as usize - } - None => arg_value.len(), - }; - - let start = if position_value_normalized >= 1 { - arg_value - .char_indices() - .nth(position_value_normalized as usize - 1) - .map_or(arg_value.len(), |(i, _)| i) - } else { - arg_value - .char_indices() - .nth_back((-position_value_normalized) as usize - 1) - .map_or(0, |(i, _)| i) - }; - - let remainder = &arg_value[start..]; - Ok(Field::String( - remainder.chars().take(length_value).collect(), - )) -} - -pub fn validate_replace(args: &[Expression], schema: &Schema) -> Result { - if args.len() != 3 { - return Err(Error::InvalidFunctionArgument { - function_name: ScalarFunctionType::Replace.to_string(), - argument_index: 0, - argument: Field::Null, - }); - } - - let mut ret_type = FieldType::String; - for exp in args { - let r = validate_arg_type( - exp, - vec![FieldType::String, FieldType::Text], - schema, - ScalarFunctionType::Replace, - 0, - )?; - if matches!(r.return_type, FieldType::Text) { - ret_type = FieldType::Text; - } - } - - Ok(ExpressionType::new( - ret_type, - false, - dozer_types::types::SourceDefinition::Dynamic, - false, - )) -} - -pub(crate) fn evaluate_replace( - schema: &Schema, - arg: &mut Expression, - search: &mut Expression, - replace: &mut Expression, - record: &Record, -) -> Result { - let arg_field = arg.evaluate(record, schema)?; - let arg_value = arg_field.to_string(); - - let search_field = search.evaluate(record, schema)?; - let search_value = search_field.to_string(); - - let replace_field = replace.evaluate(record, schema)?; - let replace_value = replace_field.to_string(); - - let result = arg_value.replace(search_value.as_str(), replace_value.as_str()); - - Ok(Field::String(result)) -} - -#[cfg(test)] -mod tests { - use super::*; - use Expression::Literal; - - use proptest::prelude::*; - - #[test] - fn test_string() { - proptest!( - ProptestConfig::with_cases(1000), - move |(s_val in ".+", s_val1 in ".*", s_val2 in ".*", c_val: char) | { - test_like(&s_val, c_val); - test_ucase(&s_val, c_val); - test_concat(&s_val1, &s_val2, c_val); - test_trim(&s_val, c_val); - }); - } - - fn test_like(s_val: &str, c_val: char) { - let row = Record::new(vec![]); - - // Field::String - let mut value = Box::new(Literal(Field::String(format!("Hello{}", s_val)))); - let mut pattern = Box::new(Literal(Field::String("Hello%".to_owned()))); - - assert_eq!( - evaluate_like(&Schema::default(), &mut value, &mut pattern, None, &row).unwrap(), - Field::Boolean(true) - ); - - let mut value = Box::new(Literal(Field::String(format!("Hello, {}orld!", c_val)))); - let mut pattern = Box::new(Literal(Field::String("Hello, _orld!".to_owned()))); - - assert_eq!( - evaluate_like(&Schema::default(), &mut value, &mut pattern, None, &row).unwrap(), - Field::Boolean(true) - ); - - let mut value = Box::new(Literal(Field::String(s_val.to_string()))); - let mut pattern = Box::new(Literal(Field::String("Hello%".to_owned()))); - - assert_eq!( - evaluate_like(&Schema::default(), &mut value, &mut pattern, None, &row).unwrap(), - Field::Boolean(false) - ); - - let c_value = &s_val[0..0]; - let mut value = Box::new(Literal(Field::String(format!("Hello, {}!", c_value)))); - let mut pattern = Box::new(Literal(Field::String("Hello, _!".to_owned()))); - - assert_eq!( - evaluate_like(&Schema::default(), &mut value, &mut pattern, None, &row).unwrap(), - Field::Boolean(false) - ); - - // todo: should find the way to generate escape character using proptest - // let mut value = Box::new(Literal(Field::String(format!("Hello, {}%", c_val)))); - // let mut pattern = Box::new(Literal(Field::String("Hello, %".to_owned()))); - // let escape = Some(c_val); - // - // assert_eq!( - // evaluate_like(&Schema::default(), &mut value, &mut pattern, escape, &row).unwrap(), - // Field::Boolean(true) - // ); - - // Field::Text - let mut value = Box::new(Literal(Field::Text(format!("Hello{}", s_val)))); - let mut pattern = Box::new(Literal(Field::Text("Hello%".to_owned()))); - - assert_eq!( - evaluate_like(&Schema::default(), &mut value, &mut pattern, None, &row).unwrap(), - Field::Boolean(true) - ); - - let mut value = Box::new(Literal(Field::Text(format!("Hello, {}orld!", c_val)))); - let mut pattern = Box::new(Literal(Field::Text("Hello, _orld!".to_owned()))); - - assert_eq!( - evaluate_like(&Schema::default(), &mut value, &mut pattern, None, &row).unwrap(), - Field::Boolean(true) - ); - - let mut value = Box::new(Literal(Field::Text(s_val.to_string()))); - let mut pattern = Box::new(Literal(Field::Text("Hello%".to_owned()))); - - assert_eq!( - evaluate_like(&Schema::default(), &mut value, &mut pattern, None, &row).unwrap(), - Field::Boolean(false) - ); - - let c_value = &s_val[0..0]; - let mut value = Box::new(Literal(Field::Text(format!("Hello, {}!", c_value)))); - let mut pattern = Box::new(Literal(Field::Text("Hello, _!".to_owned()))); - - assert_eq!( - evaluate_like(&Schema::default(), &mut value, &mut pattern, None, &row).unwrap(), - Field::Boolean(false) - ); - - // todo: should find the way to generate escape character using proptest - // let mut value = Box::new(Literal(Field::Text(format!("Hello, {}%", c_val)))); - // let mut pattern = Box::new(Literal(Field::Text("Hello, %".to_owned()))); - // let escape = Some(c_val); - // - // assert_eq!( - // evaluate_like(&Schema::default(), &mut value, &mut pattern, escape, &row).unwrap(), - // Field::Boolean(true) - // ); - } - - fn test_ucase(s_val: &str, c_val: char) { - let row = Record::new(vec![]); - - // Field::String - let mut value = Box::new(Literal(Field::String(s_val.to_string()))); - assert_eq!( - evaluate_ucase(&Schema::default(), &mut value, &row).unwrap(), - Field::String(s_val.to_uppercase()) - ); - - let mut value = Box::new(Literal(Field::String(c_val.to_string()))); - assert_eq!( - evaluate_ucase(&Schema::default(), &mut value, &row).unwrap(), - Field::String(c_val.to_uppercase().to_string()) - ); - - // Field::Text - let mut value = Box::new(Literal(Field::Text(s_val.to_string()))); - assert_eq!( - evaluate_ucase(&Schema::default(), &mut value, &row).unwrap(), - Field::Text(s_val.to_uppercase()) - ); - - let mut value = Box::new(Literal(Field::Text(c_val.to_string()))); - assert_eq!( - evaluate_ucase(&Schema::default(), &mut value, &row).unwrap(), - Field::Text(c_val.to_uppercase().to_string()) - ); - } - - fn test_concat(s_val1: &str, s_val2: &str, c_val: char) { - let row = Record::new(vec![]); - - // Field::String - let val1 = Literal(Field::String(s_val1.to_string())); - let val2 = Literal(Field::String(s_val2.to_string())); - - if validate_concat(&[val1.clone(), val2.clone()], &Schema::default()).is_ok() { - assert_eq!( - evaluate_concat(&Schema::default(), &mut [val1, val2], &row).unwrap(), - Field::String(s_val1.to_string() + s_val2) - ); - } - - let val1 = Literal(Field::String(s_val2.to_string())); - let val2 = Literal(Field::String(s_val1.to_string())); - - if validate_concat(&[val1.clone(), val2.clone()], &Schema::default()).is_ok() { - assert_eq!( - evaluate_concat(&Schema::default(), &mut [val1, val2], &row).unwrap(), - Field::String(s_val2.to_string() + s_val1) - ); - } - - let val1 = Literal(Field::String(s_val1.to_string())); - let val2 = Literal(Field::String(c_val.to_string())); - - if validate_concat(&[val1.clone(), val2.clone()], &Schema::default()).is_ok() { - assert_eq!( - evaluate_concat(&Schema::default(), &mut [val1, val2], &row).unwrap(), - Field::String(s_val1.to_string() + c_val.to_string().as_str()) - ); - } - - let val1 = Literal(Field::String(c_val.to_string())); - let val2 = Literal(Field::String(s_val1.to_string())); - - if validate_concat(&[val1.clone(), val2.clone()], &Schema::default()).is_ok() { - assert_eq!( - evaluate_concat(&Schema::default(), &mut [val1, val2], &row).unwrap(), - Field::String(c_val.to_string() + s_val1) - ); - } - - // Field::Text - let val1 = Literal(Field::Text(s_val1.to_string())); - let val2 = Literal(Field::Text(s_val2.to_string())); - - if validate_concat(&[val1.clone(), val2.clone()], &Schema::default()).is_ok() { - assert_eq!( - evaluate_concat(&Schema::default(), &mut [val1, val2], &row).unwrap(), - Field::Text(s_val1.to_string() + s_val2) - ); - } - - let val1 = Literal(Field::Text(s_val2.to_string())); - let val2 = Literal(Field::Text(s_val1.to_string())); - - if validate_concat(&[val1.clone(), val2.clone()], &Schema::default()).is_ok() { - assert_eq!( - evaluate_concat(&Schema::default(), &mut [val1, val2], &row).unwrap(), - Field::Text(s_val2.to_string() + s_val1) - ); - } - - let val1 = Literal(Field::Text(s_val1.to_string())); - let val2 = Literal(Field::Text(c_val.to_string())); - - if validate_concat(&[val1.clone(), val2.clone()], &Schema::default()).is_ok() { - assert_eq!( - evaluate_concat(&Schema::default(), &mut [val1, val2], &row).unwrap(), - Field::Text(s_val1.to_string() + c_val.to_string().as_str()) - ); - } - - let val1 = Literal(Field::Text(c_val.to_string())); - let val2 = Literal(Field::Text(s_val1.to_string())); - - if validate_concat(&[val1.clone(), val2.clone()], &Schema::default()).is_ok() { - assert_eq!( - evaluate_concat(&Schema::default(), &mut [val1, val2], &row).unwrap(), - Field::Text(c_val.to_string() + s_val1) - ); - } - } - - fn test_trim(s_val1: &str, c_val: char) { - let row = Record::new(vec![]); - - // Field::String - let mut value = Literal(Field::String(s_val1.to_string())); - let what = ' '; - - if validate_trim(&value, &Schema::default()).is_ok() { - assert_eq!( - evaluate_trim(&Schema::default(), &mut value, &mut None, &None, &row).unwrap(), - Field::String(s_val1.trim_matches(what).to_string()) - ); - assert_eq!( - evaluate_trim( - &Schema::default(), - &mut value, - &mut None, - &Some(TrimType::Trailing), - &row - ) - .unwrap(), - Field::String(s_val1.trim_end_matches(what).to_string()) - ); - assert_eq!( - evaluate_trim( - &Schema::default(), - &mut value, - &mut None, - &Some(TrimType::Leading), - &row - ) - .unwrap(), - Field::String(s_val1.trim_start_matches(what).to_string()) - ); - assert_eq!( - evaluate_trim( - &Schema::default(), - &mut value, - &mut None, - &Some(TrimType::Both), - &row - ) - .unwrap(), - Field::String(s_val1.trim_matches(what).to_string()) - ); - } - - let mut value = Literal(Field::String(s_val1.to_string())); - let mut what = Some(Box::new(Literal(Field::String(c_val.to_string())))); - - if validate_trim(&value, &Schema::default()).is_ok() { - assert_eq!( - evaluate_trim(&Schema::default(), &mut value, &mut what, &None, &row).unwrap(), - Field::String(s_val1.trim_matches(c_val).to_string()) - ); - assert_eq!( - evaluate_trim( - &Schema::default(), - &mut value, - &mut what, - &Some(TrimType::Trailing), - &row - ) - .unwrap(), - Field::String(s_val1.trim_end_matches(c_val).to_string()) - ); - assert_eq!( - evaluate_trim( - &Schema::default(), - &mut value, - &mut what, - &Some(TrimType::Leading), - &row - ) - .unwrap(), - Field::String(s_val1.trim_start_matches(c_val).to_string()) - ); - assert_eq!( - evaluate_trim( - &Schema::default(), - &mut value, - &mut what, - &Some(TrimType::Both), - &row - ) - .unwrap(), - Field::String(s_val1.trim_matches(c_val).to_string()) - ); - } - } - - #[test] - fn test_chr() { - let row = Record::new(vec![]); - - let mut value = Box::new(Literal(Field::Int(65))); - assert_eq!( - evaluate_chr(&Schema::default(), &mut value, &row).unwrap(), - Field::String("A".to_owned()) - ); - let mut value = Box::new(Literal(Field::Int(321))); - assert_eq!( - evaluate_chr(&Schema::default(), &mut value, &row).unwrap(), - Field::String("A".to_owned()) - ); - } - - #[test] - fn test_substr() { - let row = Record::new(vec![]); - - let mut value = Box::new(Literal(Field::String("ABCDEFG".to_owned()))); - let mut position = Box::new(Literal(Field::Int(3))); - let mut length = Some(Box::new(Literal(Field::Int(4)))); - let result = Field::String("CDEF".to_owned()); - - assert_eq!( - evaluate_substr( - &Schema::default(), - &mut value, - &mut position, - &mut length, - &row - ) - .unwrap(), - result - ); - - let mut value = Box::new(Literal(Field::String("ABCDEFG".to_owned()))); - let mut position = Box::new(Literal(Field::Int(-5))); - let mut length = Some(Box::new(Literal(Field::Int(4)))); - let result = Field::String("CDEF".to_owned()); - - assert_eq!( - evaluate_substr( - &Schema::default(), - &mut value, - &mut position, - &mut length, - &row - ) - .unwrap(), - result - ); - } - - #[test] - fn test_replace() { - let row = Record::new(vec![]); - let mut value = Box::new(Literal(Field::String("JACK AND JUE".to_owned()))); - let mut search = Box::new(Literal(Field::String("J".to_owned()))); - let mut replace = Box::new(Literal(Field::String("BL".to_owned()))); - - assert_eq!( - evaluate_replace( - &Schema::default(), - &mut value, - &mut search, - &mut replace, - &row - ) - .unwrap(), - Field::String("BLACK AND BLUE".to_owned()) - ); - } -} diff --git a/dozer-sql/jsonpath/Cargo.toml b/dozer-sql/jsonpath/Cargo.toml deleted file mode 100644 index 82bab34296..0000000000 --- a/dozer-sql/jsonpath/Cargo.toml +++ /dev/null @@ -1,21 +0,0 @@ -[package] -name = "jsonpath" -description = "The library provides the basic functionality to find the set of the data according to the filtering query." -version = "0.2.6" -authors = ["BorisZhguchev "] -edition = "2018" -license-file = "LICENSE" -homepage = "https://github.com/besok/jsonpath-rust" -repository = "https://github.com/besok/jsonpath-rust" -readme = "README.md" -keywords = ["json", "json-path", "jsonpath", "jsonpath-rust", "xpath"] -categories = ["development-tools", "parsing", "text-processing"] - -[dependencies] -dozer-types = { path = "../../dozer-types" } -regex = "1" -pest = "2.0" -pest_derive = "2.0" - -[dev-dependencies] -lazy_static = "1.0" diff --git a/dozer-sql/jsonpath/LICENSE b/dozer-sql/jsonpath/LICENSE deleted file mode 100644 index 3e7627bcf9..0000000000 --- a/dozer-sql/jsonpath/LICENSE +++ /dev/null @@ -1,21 +0,0 @@ -MIT License - -Copyright (c) [2021] [Boris Zhguchev] - -Permission is hereby granted, free of charge, to any person obtaining a copy -of this software and associated documentation files (the "Software"), to deal -in the Software without restriction, including without limitation the rights -to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -copies of the Software, and to permit persons to whom the Software is -furnished to do so, subject to the following conditions: - -The above copyright notice and this permission notice shall be included in all -copies or substantial portions of the Software. - -THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -SOFTWARE. diff --git a/dozer-sql/jsonpath/README.md b/dozer-sql/jsonpath/README.md deleted file mode 100644 index 63d6c29056..0000000000 --- a/dozer-sql/jsonpath/README.md +++ /dev/null @@ -1,3 +0,0 @@ -## Disclaimer - -This `jsonpath` folder has been forked from [jsonpath-rust](https://github.com/besok/jsonpath-rust) to integrate with Dozer native `JsonValue`. diff --git a/dozer-sql/jsonpath/src/lib.rs b/dozer-sql/jsonpath/src/lib.rs deleted file mode 100644 index 9a145b9d1d..0000000000 --- a/dozer-sql/jsonpath/src/lib.rs +++ /dev/null @@ -1,219 +0,0 @@ -use crate::parser::model::JsonPath; -use crate::path::{json_path_instance, PathInstance}; -use crate::JsonPathValue::{NewValue, NoValue, Slice}; -use dozer_types::json_types::{json_from_str, JsonArray, JsonValue}; -use std::convert::TryInto; -use std::fmt::Debug; -use std::str::FromStr; - -pub mod parser; -pub mod path; - -pub trait JsonPathQuery { - fn path(self, query: &str) -> Result; -} - -pub struct JsonPathInst { - inner: JsonPath, -} - -impl FromStr for JsonPathInst { - type Err = String; - - fn from_str(s: &str) -> Result { - Ok(JsonPathInst { - inner: s.try_into()?, - }) - } -} - -impl JsonPathQuery for Box { - fn path(self, query: &str) -> Result { - let p = JsonPathInst::from_str(query)?; - Ok(JsonPathFinder::new(self, Box::new(p)).find()) - } -} - -impl JsonPathQuery for JsonValue { - fn path(self, query: &str) -> Result { - let p = JsonPathInst::from_str(query)?; - Ok(JsonPathFinder::new(Box::new(self), Box::new(p)).find()) - } -} - -#[macro_export] -macro_rules! json_path_value { - (&$v:expr) =>{ - JsonPathValue::Slice(&$v) - }; - - ($(&$v:expr),+ $(,)?) =>{ - { - let mut res = Vec::new(); - $( - res.push(json_path_value!(&$v)); - )+ - res - } - }; - ($v:expr) =>{ - JsonPathValue::NewValue($v) - }; - -} - -#[derive(Debug, PartialEq, Copy, Clone)] -pub enum JsonPathValue<'a, Data> { - /// The slice of the initial json data - Slice(&'a Data), - /// The new data that was generated from the input data (like length operator) - NewValue(Data), - /// The absent value that indicates the input data is not matched to the given json path (like the absent fields) - NoValue, -} - -impl<'a, Data: Clone + Debug + Default> JsonPathValue<'a, Data> { - pub fn to_data(self) -> Data { - match self { - Slice(r) => r.clone(), - NewValue(val) => val, - NoValue => Data::default(), - } - } -} - -impl<'a, Data> From<&'a Data> for JsonPathValue<'a, Data> { - fn from(data: &'a Data) -> Self { - Slice(data) - } -} - -impl<'a, Data> JsonPathValue<'a, Data> { - fn only_no_value(input: &[JsonPathValue<'a, Data>]) -> bool { - !input.is_empty() && input.iter().filter(|v| v.has_value()).count() == 0 - } - - fn map_vec(data: Vec<&'a Data>) -> Vec> { - data.into_iter().map(|v| v.into()).collect() - } - - fn map_slice(self, mapper: F) -> Vec> - where - F: FnOnce(&'a Data) -> Vec<&'a Data>, - { - match self { - Slice(r) => mapper(r).into_iter().map(Slice).collect(), - NewValue(_) => vec![], - no_v => vec![no_v], - } - } - - fn flat_map_slice(self, mapper: F) -> Vec> - where - F: FnOnce(&'a Data) -> Vec>, - { - match self { - Slice(r) => mapper(r), - _ => vec![NoValue], - } - } - - pub fn has_value(&self) -> bool { - !matches!(self, NoValue) - } - - pub fn into_data(input: Vec>) -> Vec<&'a Data> { - input - .into_iter() - .filter_map(|v| match v { - Slice(el) => Some(el), - _ => None, - }) - .collect() - } - - /// moves a pointer (from slice) out or provides a default value when the value was generated - pub fn slice_or(self, default: &'a Data) -> &'a Data { - match self { - Slice(r) => r, - NewValue(_) | NoValue => default, - } - } -} - -/// The base structure stitching the json instance and jsonpath instance -pub struct JsonPathFinder { - json: Box, - path: Box, -} - -impl JsonPathFinder { - /// creates a new instance of [JsonPathFinder] - pub fn new(json: Box, path: Box) -> Self { - JsonPathFinder { json, path } - } - - /// updates a path with a new one - pub fn set_path(&mut self, path: Box) { - self.path = path - } - /// updates a json with a new one - pub fn set_json(&mut self, json: Box) { - self.json = json - } - /// updates a json from string and therefore can be some parsing errors - pub fn set_json_str(&mut self, json: &str) -> Result<(), String> { - self.json = Box::from(json_from_str(json).map_err(|e| e.to_string())?); - Ok(()) - } - /// updates a path from string and therefore can be some parsing errors - pub fn set_path_str(&mut self, path: &str) -> Result<(), String> { - self.path = Box::new(JsonPathInst::from_str(path)?); - Ok(()) - } - - /// create a new instance from string and therefore can be some parsing errors - pub fn from_str(json: &str, path: &str) -> Result { - let json = json_from_str(json).map_err(|e| e.to_string())?; - let path = Box::new(JsonPathInst::from_str(path)?); - Ok(JsonPathFinder::new(Box::from(json), path)) - } - - /// creates an instance to find a json slice from the json - pub fn instance(&self) -> PathInstance { - json_path_instance(&self.path.inner, &self.json) - } - /// finds a slice of data in the set json. - /// The result is a vector of references to the incoming structure. - pub fn find_slice(&self) -> Vec> { - let res = self.instance().find((&(*self.json)).into()); - let has_v: Vec> = - res.into_iter().filter(|v| v.has_value()).collect(); - - if has_v.is_empty() { - vec![NoValue] - } else { - has_v - } - } - - /// finds a slice of data and wrap it with Value::Array by cloning the data. - /// Returns either an array of elements or Json::Null if the match is incorrect. - pub fn find(&self) -> JsonValue { - let slice = self.find_slice(); - if !slice.is_empty() { - if JsonPathValue::only_no_value(&slice) { - JsonValue::NULL - } else { - self.find_slice() - .into_iter() - .filter(|v| v.has_value()) - .map(|v| v.to_data()) - .collect::() - .into() - } - } else { - JsonArray::new().into() - } - } -} diff --git a/dozer-sql/jsonpath/src/parser/grammar/json_path.pest b/dozer-sql/jsonpath/src/parser/grammar/json_path.pest deleted file mode 100644 index a0ed9594b6..0000000000 --- a/dozer-sql/jsonpath/src/parser/grammar/json_path.pest +++ /dev/null @@ -1,53 +0,0 @@ -WHITESPACE = _{ " " | "\t" | "\r\n" | "\n"} - -boolean = {"true" | "false"} -null = {"null"} - -min = _{"-"} -col = _{":"} -dot = _{ "." } -word = _{ ('a'..'z' | 'A'..'Z')+ } -specs = _{ "_" | "-" | "/" | "\\" | "#" } -number = @{"-"? ~ ("0" | ASCII_NONZERO_DIGIT ~ ASCII_DIGIT*) ~ ("." ~ ASCII_DIGIT+)? ~ (^"e" ~ ("+" | "-")? ~ ASCII_DIGIT+)?} - -string_qt = ${ "\'" ~ inner ~ "\'" } -inner = @{ char* } -char = _{ - !("\"" | "\\" | "\'") ~ ANY - | "\\" ~ ("\"" | "\'" | "\\" | "/" | "b" | "f" | "n" | "r" | "t") - | "\\" ~ ("u" ~ ASCII_HEX_DIGIT{4}) -} -root = {"$"} -sign = { "==" | "!=" | "~=" | ">=" | ">" | "<=" | "<" | "in" | "nin" | "size" | "noneOf" | "anyOf" | "subsetOf"} -key_lim = {!"length()" ~ (word | ASCII_DIGIT | specs)+} -key_unlim = {"[" ~ string_qt ~ "]"} -key = ${key_lim | key_unlim} - -descent = {dot ~ dot ~ key} -descent_w = {dot ~ dot ~ "*"} // refactor afterwards -wildcard = {dot? ~ "[" ~"*"~"]" | dot ~ "*"} -current = {"@" ~ chain?} -field = ${dot? ~ key_unlim | dot ~ key_lim } -function = { dot ~ "length" ~ "(" ~ ")"} -unsigned = {("0" | ASCII_NONZERO_DIGIT ~ ASCII_DIGIT*)} -signed = {min? ~ unsigned} -start_slice = {signed} -end_slice = {signed} -step_slice = {col ~ unsigned} -slice = {start_slice? ~ col ~ end_slice? ~ step_slice? } - -unit_keys = { string_qt ~ ("," ~ string_qt)+ } -unit_indexes = { number ~ ("," ~ number)+ } -filter = {"?"~ "(" ~ logic ~ ")"} - -logic = {logic_and ~ ("||" ~ logic_and)*} -logic_and = {logic_atom ~ ("&&" ~ logic_atom)*} -logic_atom = {atom ~ (sign ~ atom)? | "(" ~ logic ~ ")"} - -atom = {chain | string_qt | number | boolean | null} - -index = {dot? ~ "["~ (unit_keys | unit_indexes | slice | unsigned |filter) ~ "]" } - -chain = {(root | descent | descent_w | wildcard | current | field | index | function)+} - -path = {SOI ~ chain ~ EOI } \ No newline at end of file diff --git a/dozer-sql/jsonpath/src/parser/macros.rs b/dozer-sql/jsonpath/src/parser/macros.rs deleted file mode 100644 index e8847005aa..0000000000 --- a/dozer-sql/jsonpath/src/parser/macros.rs +++ /dev/null @@ -1,83 +0,0 @@ -#[macro_export] -macro_rules! filter { - () => {FilterExpression::Atom(op!,FilterSign::new(""),op!())}; - ( $left:expr, $s:literal, $right:expr) => { - FilterExpression::Atom($left,FilterSign::new($s),$right) - }; - ( $left:expr,||, $right:expr) => {FilterExpression::Or(Box::new($left),Box::new($right)) }; - ( $left:expr,&&, $right:expr) => {FilterExpression::And(Box::new($left),Box::new($right)) }; -} -#[macro_export] -macro_rules! op { - ( ) => { - Operand::Dynamic(Box::new(JsonPath::Empty)) - }; - ( $s:literal) => { - Operand::Static(json!($s)) - }; - ( s $s:expr) => { - Operand::Static(json!($s)) - }; - ( $s:expr) => { - Operand::Dynamic(Box::new($s)) - }; -} - -#[macro_export] -macro_rules! idx { - ( $s:literal) => {JsonPathIndex::Single(json!($s))}; - ( idx $($ss:literal),+) => {{ - let mut ss_vec = Vec::new(); - $( ss_vec.push(json!($ss)) ; )+ - JsonPathIndex::UnionIndex(ss_vec) - }}; - ( $($ss:literal),+) => {{ - let mut ss_vec = Vec::new(); - $( ss_vec.push($ss.to_string()) ; )+ - JsonPathIndex::UnionKeys(ss_vec) - }}; - ( $s:literal) => {JsonPathIndex::Single(json!($s))}; - ( ? $s:expr) => {JsonPathIndex::Filter($s)}; - ( [$l:literal;$m:literal;$r:literal]) => {JsonPathIndex::Slice($l,$m,$r)}; - ( [$l:literal;$m:literal;]) => {JsonPathIndex::Slice($l,$m,1)}; - ( [$l:literal;;$m:literal]) => {JsonPathIndex::Slice($l,0,$m)}; - ( [;$l:literal;$m:literal]) => {JsonPathIndex::Slice(0,$l,$m)}; - ( [;;$m:literal]) => {JsonPathIndex::Slice(0,0,$m)}; - ( [;$m:literal;]) => {JsonPathIndex::Slice(0,$m,1)}; - ( [$m:literal;;]) => {JsonPathIndex::Slice($m,0,1)}; - ( [;;]) => {JsonPathIndex::Slice(0,0,1)}; -} - -#[macro_export] -macro_rules! chain { - ($($ss:expr),+) => {{ - let mut ss_vec = Vec::new(); - $( ss_vec.push($ss) ; )+ - JsonPath::Chain(ss_vec) - }}; -} - -#[macro_export] -macro_rules! path { - ( ) => {JsonPath::Empty}; - (*) => {JsonPath::Wildcard}; - ($) => {JsonPath::Root}; - (@) => {JsonPath::Current(Box::new(JsonPath::Empty))}; - (@$e:expr) => {JsonPath::Current(Box::new($e))}; - (@,$($ss:expr),+) => {{ - let mut ss_vec = Vec::new(); - $( ss_vec.push($ss) ; )+ - let chain = JsonPath::Chain(ss_vec); - JsonPath::Current(Box::new(chain)) - }}; - (..$e:literal) => {JsonPath::Descent($e.to_string())}; - (..*) => {JsonPath::DescentW}; - ($e:literal) => {JsonPath::Field($e.to_string())}; - ($e:expr) => {JsonPath::Index($e)}; -} -#[macro_export] -macro_rules! function { - (length) => { - JsonPath::Fn(Function::Length) - }; -} diff --git a/dozer-sql/jsonpath/src/parser/mod.rs b/dozer-sql/jsonpath/src/parser/mod.rs deleted file mode 100644 index 9077ad6c64..0000000000 --- a/dozer-sql/jsonpath/src/parser/mod.rs +++ /dev/null @@ -1,7 +0,0 @@ -//! The parser for the jsonpath. -//! The module grammar denotes the structure of the parsing grammar - -mod macros; -pub mod model; -#[allow(clippy::module_inception)] -pub mod parser; diff --git a/dozer-sql/jsonpath/src/parser/model.rs b/dozer-sql/jsonpath/src/parser/model.rs deleted file mode 100644 index 094c6e566c..0000000000 --- a/dozer-sql/jsonpath/src/parser/model.rs +++ /dev/null @@ -1,177 +0,0 @@ -use crate::parser::parser::parse_json_path; -use dozer_types::json_types::JsonValue; -use std::convert::TryFrom; - -/// The basic structures for parsing json paths. -/// The common logic of the structures pursues to correspond the internal parsing structure. -#[derive(Debug, Clone)] -pub enum JsonPath { - /// The $ operator - Root, - /// Field represents key - Field(String), - /// The whole chain of the path. - Chain(Vec), - /// The .. operator - Descent(String), - /// The ..* operator - DescentW, - /// The indexes for array - Index(JsonPathIndex), - /// The @ operator - Current(Box), - /// The * operator - Wildcard, - /// The item uses to define the unresolved state - Empty, - /// Functions that can calculate some expressions - Fn(Function), -} - -impl TryFrom<&str> for JsonPath { - type Error = String; - - fn try_from(value: &str) -> Result { - parse_json_path(value).map_err(|e| e.to_string()) - } -} - -#[derive(Debug, PartialEq, Clone)] -pub enum Function { - /// length() - Length, -} -#[derive(Debug, Clone)] -pub enum JsonPathIndex { - /// A single element in array - Single(JsonValue), - /// Union represents a several indexes - UnionIndex(Vec), - /// Union represents a several keys - UnionKeys(Vec), - /// DEfault slice where the items are start/end/step respectively - Slice(i32, i32, usize), - /// Filter ?() - Filter(FilterExpression), -} - -#[derive(Debug, Clone, PartialEq)] -pub enum FilterExpression { - /// a single expression like a > 2 - Atom(Operand, FilterSign, Operand), - /// and with && - And(Box, Box), - /// or with || - Or(Box, Box), -} - -impl FilterExpression { - pub fn exists(op: Operand) -> Self { - FilterExpression::Atom( - op, - FilterSign::Exists, - Operand::Dynamic(Box::new(JsonPath::Empty)), - ) - } -} - -/// Operand for filtering expressions -#[derive(Debug, Clone)] -pub enum Operand { - Static(JsonValue), - Dynamic(Box), -} - -#[allow(dead_code)] -impl Operand { - pub fn val(v: JsonValue) -> Self { - Operand::Static(v) - } -} - -/// The operators for filtering functions -#[derive(Debug, Clone, PartialEq)] -pub enum FilterSign { - Equal, - Unequal, - Less, - Greater, - LeOrEq, - GrOrEq, - Regex, - In, - Nin, - Size, - NoneOf, - AnyOf, - SubSetOf, - Exists, -} - -impl FilterSign { - pub fn new(key: &str) -> Self { - match key { - "==" => FilterSign::Equal, - "!=" => FilterSign::Unequal, - "<" => FilterSign::Less, - ">" => FilterSign::Greater, - "<=" => FilterSign::LeOrEq, - ">=" => FilterSign::GrOrEq, - "~=" => FilterSign::Regex, - "in" => FilterSign::In, - "nin" => FilterSign::Nin, - "size" => FilterSign::Size, - "noneOf" => FilterSign::NoneOf, - "anyOf" => FilterSign::AnyOf, - "subsetOf" => FilterSign::SubSetOf, - _ => FilterSign::Exists, - } - } -} - -impl PartialEq for JsonPath { - fn eq(&self, other: &Self) -> bool { - match (self, other) { - (JsonPath::Root, JsonPath::Root) => true, - (JsonPath::Descent(k1), JsonPath::Descent(k2)) => k1 == k2, - (JsonPath::DescentW, JsonPath::DescentW) => true, - (JsonPath::Field(k1), JsonPath::Field(k2)) => k1 == k2, - (JsonPath::Wildcard, JsonPath::Wildcard) => true, - (JsonPath::Empty, JsonPath::Empty) => true, - (JsonPath::Current(jp1), JsonPath::Current(jp2)) => jp1 == jp2, - (JsonPath::Chain(ch1), JsonPath::Chain(ch2)) => ch1 == ch2, - (JsonPath::Index(idx1), JsonPath::Index(idx2)) => idx1 == idx2, - (JsonPath::Fn(fn1), JsonPath::Fn(fn2)) => fn2 == fn1, - (_, _) => false, - } - } -} - -impl PartialEq for JsonPathIndex { - fn eq(&self, other: &Self) -> bool { - match (self, other) { - (JsonPathIndex::Slice(s1, e1, st1), JsonPathIndex::Slice(s2, e2, st2)) => { - s1 == s2 && e1 == e2 && st1 == st2 - } - (JsonPathIndex::Single(el1), JsonPathIndex::Single(el2)) => el1 == el2, - (JsonPathIndex::UnionIndex(elems1), JsonPathIndex::UnionIndex(elems2)) => { - elems1 == elems2 - } - (JsonPathIndex::UnionKeys(elems1), JsonPathIndex::UnionKeys(elems2)) => { - elems1 == elems2 - } - (JsonPathIndex::Filter(left), JsonPathIndex::Filter(right)) => left.eq(right), - (_, _) => false, - } - } -} - -impl PartialEq for Operand { - fn eq(&self, other: &Self) -> bool { - match (self, other) { - (Operand::Static(v1), Operand::Static(v2)) => v1 == v2, - (Operand::Dynamic(jp1), Operand::Dynamic(jp2)) => jp1 == jp2, - (_, _) => false, - } - } -} diff --git a/dozer-sql/jsonpath/src/parser/parser.rs b/dozer-sql/jsonpath/src/parser/parser.rs deleted file mode 100644 index c66db97001..0000000000 --- a/dozer-sql/jsonpath/src/parser/parser.rs +++ /dev/null @@ -1,195 +0,0 @@ -use crate::parser::model::FilterExpression::{And, Or}; -use crate::parser::model::{ - FilterExpression, FilterSign, Function, JsonPath, JsonPathIndex, Operand, -}; -use dozer_types::json_types::JsonValue; -use pest::error::Error; -use pest::iterators::{Pair, Pairs}; -use pest::Parser; -use pest_derive::Parser; - -#[derive(Parser)] -#[grammar = "parser/grammar/json_path.pest"] -pub struct JsonPathParser; - -/// the parsing function. -/// Since the parsing can finish with error the result is [[Result]] -#[allow(clippy::result_large_err)] -pub fn parse_json_path(jp_str: &str) -> Result> { - Ok(parse_internal( - JsonPathParser::parse(Rule::path, jp_str)?.next().unwrap(), - )) -} - -/// Internal function takes care of the logic by parsing the operators and unrolling the string into the final result. -fn parse_internal(rule: Pair) -> JsonPath { - match rule.as_rule() { - Rule::path => rule - .into_inner() - .next() - .map(parse_internal) - .unwrap_or(JsonPath::Empty), - Rule::current => JsonPath::Current(Box::new( - rule.into_inner() - .next() - .map(parse_internal) - .unwrap_or(JsonPath::Empty), - )), - Rule::chain => JsonPath::Chain(rule.into_inner().map(parse_internal).collect()), - Rule::root => JsonPath::Root, - Rule::wildcard => JsonPath::Wildcard, - Rule::descent => parse_key(down(rule)) - .map(JsonPath::Descent) - .unwrap_or(JsonPath::Empty), - Rule::descent_w => JsonPath::DescentW, - Rule::function => JsonPath::Fn(Function::Length), - Rule::field => parse_key(down(rule)) - .map(JsonPath::Field) - .unwrap_or(JsonPath::Empty), - Rule::index => JsonPath::Index(parse_index(rule)), - _ => JsonPath::Empty, - } -} - -/// parsing the rule 'key' with the structures either .key or .]'key'[ -fn parse_key(rule: Pair) -> Option { - match rule.as_rule() { - Rule::key | Rule::key_unlim | Rule::string_qt => parse_key(down(rule)), - Rule::key_lim | Rule::inner => Some(String::from(rule.as_str())), - _ => None, - } -} - -fn parse_slice(mut pairs: Pairs) -> JsonPathIndex { - let mut start = 0; - let mut end = 0; - let mut step = 1; - while pairs.peek().is_some() { - let in_pair = pairs.next().unwrap(); - match in_pair.as_rule() { - Rule::start_slice => start = in_pair.as_str().parse::().unwrap_or(start), - Rule::end_slice => end = in_pair.as_str().parse::().unwrap_or(end), - Rule::step_slice => step = down(in_pair).as_str().parse::().unwrap_or(step), - _ => (), - } - } - JsonPathIndex::Slice(start, end, step) -} - -fn parse_unit_keys(mut pairs: Pairs) -> JsonPathIndex { - let mut keys = vec![]; - - while pairs.peek().is_some() { - keys.push(String::from(down(pairs.next().unwrap()).as_str())); - } - JsonPathIndex::UnionKeys(keys) -} - -fn number_to_value(number: &str) -> JsonValue { - number.parse::().ok().map(JsonValue::from).unwrap() -} - -fn parse_unit_indexes(mut pairs: Pairs) -> JsonPathIndex { - let mut keys = vec![]; - - while pairs.peek().is_some() { - keys.push(number_to_value(pairs.next().unwrap().as_str())); - } - JsonPathIndex::UnionIndex(keys) -} - -fn parse_chain_in_operand(rule: Pair) -> Operand { - match parse_internal(rule) { - JsonPath::Chain(elems) => { - if elems.len() == 1 { - match elems.first() { - Some(JsonPath::Index(JsonPathIndex::UnionKeys(keys))) => { - Operand::val(JsonValue::from(keys.clone())) - } - Some(JsonPath::Index(JsonPathIndex::UnionIndex(keys))) => { - Operand::val(JsonValue::from(keys.clone())) - } - Some(JsonPath::Field(f)) => Operand::val(vec![f].into()), - _ => Operand::Dynamic(Box::new(JsonPath::Chain(elems))), - } - } else { - Operand::Dynamic(Box::new(JsonPath::Chain(elems))) - } - } - jp => Operand::Dynamic(Box::new(jp)), - } -} - -fn parse_filter_index(pair: Pair) -> JsonPathIndex { - JsonPathIndex::Filter(parse_logic(pair.into_inner())) -} - -fn parse_logic(mut pairs: Pairs) -> FilterExpression { - let mut expr: Option = None; - while pairs.peek().is_some() { - let next_expr = parse_logic_and(pairs.next().unwrap().into_inner()); - match expr { - None => expr = Some(next_expr), - Some(e) => expr = Some(Or(Box::new(e), Box::new(next_expr))), - } - } - expr.unwrap() -} - -fn parse_logic_and(mut pairs: Pairs) -> FilterExpression { - let mut expr: Option = None; - - while pairs.peek().is_some() { - let next_expr = parse_logic_atom(pairs.next().unwrap().into_inner()); - match expr { - None => expr = Some(next_expr), - Some(e) => expr = Some(And(Box::new(e), Box::new(next_expr))), - } - } - expr.unwrap() -} - -fn parse_logic_atom(mut pairs: Pairs) -> FilterExpression { - match pairs.peek().map(|x| x.as_rule()) { - Some(Rule::logic) => parse_logic(pairs.next().unwrap().into_inner()), - Some(Rule::atom) => { - let left: Operand = parse_atom(pairs.next().unwrap()); - if pairs.peek().is_none() { - FilterExpression::exists(left) - } else { - let sign: FilterSign = FilterSign::new(pairs.next().unwrap().as_str()); - let right: Operand = parse_atom(pairs.next().unwrap()); - FilterExpression::Atom(left, sign, right) - } - } - Some(x) => panic!("unexpected => {:?}", x), - None => panic!("unexpected none"), - } -} - -fn parse_atom(rule: Pair) -> Operand { - let atom = down(rule.clone()); - match atom.as_rule() { - Rule::number => Operand::Static(number_to_value(rule.as_str())), - Rule::string_qt => Operand::Static(JsonValue::from(down(atom).as_str())), - Rule::chain => parse_chain_in_operand(down(rule)), - Rule::boolean => Operand::Static(rule.as_str().parse::().unwrap().into()), - _ => Operand::Static(JsonValue::NULL), - } -} - -fn parse_index(rule: Pair) -> JsonPathIndex { - let next = down(rule); - match next.as_rule() { - Rule::unsigned => JsonPathIndex::Single(number_to_value(next.as_str())), - Rule::slice => parse_slice(next.into_inner()), - Rule::unit_indexes => parse_unit_indexes(next.into_inner()), - Rule::unit_keys => parse_unit_keys(next.into_inner()), - Rule::filter => parse_filter_index(down(next)), - _ => JsonPathIndex::Single(number_to_value(next.as_str())), - } -} - -fn down(rule: Pair) -> Pair { - rule.into_inner().next().unwrap() -} diff --git a/dozer-sql/jsonpath/src/path/index.rs b/dozer-sql/jsonpath/src/path/index.rs deleted file mode 100644 index 50289374d1..0000000000 --- a/dozer-sql/jsonpath/src/path/index.rs +++ /dev/null @@ -1,326 +0,0 @@ -use crate::parser::model::{FilterExpression, FilterSign, JsonPath}; -use crate::path::json::{any_of, eq, inside, less, regex, size, sub_set_of}; -use crate::path::top::ObjectField; -use crate::path::{json_path_instance, process_operand, Path, PathInstance}; -use crate::JsonPathValue; -use crate::JsonPathValue::{NoValue, Slice}; -use dozer_types::json_types::JsonValue; - -/// process the slice like [start:end:step] -#[derive(Debug)] -pub(crate) struct ArraySlice { - start_index: i32, - end_index: i32, - step: usize, -} - -impl ArraySlice { - pub(crate) fn new(start_index: i32, end_index: i32, step: usize) -> ArraySlice { - ArraySlice { - start_index, - end_index, - step, - } - } - - fn end(&self, len: i32) -> Option { - if self.end_index >= 0 { - if self.end_index > len { - None - } else { - Some(self.end_index as usize) - } - } else if self.end_index < -len { - None - } else { - Some((len - (-self.end_index)) as usize) - } - } - - fn start(&self, len: i32) -> Option { - if self.start_index >= 0 { - if self.start_index > len { - None - } else { - Some(self.start_index as usize) - } - } else if self.start_index < -len { - None - } else { - Some((len - -self.start_index) as usize) - } - } - - fn process<'a, T>(&self, elements: &'a [T]) -> Vec<&'a T> { - let len = elements.len() as i32; - let mut filtered_elems: Vec<&T> = vec![]; - match (self.start(len), self.end(len)) { - (Some(start_idx), Some(end_idx)) => { - let end_idx = if end_idx == 0 { - elements.len() - } else { - end_idx - }; - for idx in (start_idx..end_idx).step_by(self.step) { - if let Some(v) = elements.get(idx) { - filtered_elems.push(v) - } - } - filtered_elems - } - _ => filtered_elems, - } - } -} - -impl<'a> Path<'a> for ArraySlice { - type Data = JsonValue; - - fn find(&self, input: JsonPathValue<'a, Self::Data>) -> Vec> { - input.flat_map_slice(|data| { - data.as_array() - .map(|elems| self.process(elems)) - .and_then(|v| { - if v.is_empty() { - None - } else { - Some(JsonPathValue::map_vec(v)) - } - }) - .unwrap_or_else(|| vec![NoValue]) - }) - } -} - -/// process the simple index like [index] -pub(crate) struct ArrayIndex { - index: usize, -} - -impl ArrayIndex { - pub(crate) fn new(index: usize) -> Self { - ArrayIndex { index } - } -} - -impl<'a> Path<'a> for ArrayIndex { - type Data = JsonValue; - - fn find(&self, input: JsonPathValue<'a, Self::Data>) -> Vec> { - input.flat_map_slice(|data| { - data.as_array() - .and_then(|elems| elems.get(self.index)) - .map(|e| vec![e.into()]) - .unwrap_or_else(|| vec![NoValue]) - }) - } -} - -/// process @ element -pub(crate) struct Current<'a> { - tail: Option>, -} - -impl<'a> Current<'a> { - pub(crate) fn from(jp: &'a JsonPath, root: &'a JsonValue) -> Self { - match jp { - JsonPath::Empty => Current::none(), - tail => Current::new(json_path_instance(tail, root)), - } - } - pub(crate) fn new(tail: PathInstance<'a>) -> Self { - Current { tail: Some(tail) } - } - pub(crate) fn none() -> Self { - Current { tail: None } - } -} - -impl<'a> Path<'a> for Current<'a> { - type Data = JsonValue; - - fn find(&self, input: JsonPathValue<'a, Self::Data>) -> Vec> { - self.tail - .as_ref() - .map(|p| p.find(input.clone())) - .unwrap_or_else(|| vec![input]) - } -} - -/// the list of indexes like [1,2,3] -pub(crate) struct UnionIndex<'a> { - indexes: Vec>, -} - -impl<'a> UnionIndex<'a> { - pub fn from_indexes(elems: &'a [JsonValue]) -> Self { - let mut indexes: Vec> = vec![]; - - for idx in elems.iter() { - indexes.push(Box::new(ArrayIndex::new(idx.to_u64().unwrap() as usize))) - } - - UnionIndex::new(indexes) - } - pub fn from_keys(elems: &'a [String]) -> Self { - let mut indexes: Vec> = vec![]; - - for key in elems.iter() { - indexes.push(Box::new(ObjectField::new(key))) - } - - UnionIndex::new(indexes) - } - - pub fn new(indexes: Vec>) -> Self { - UnionIndex { indexes } - } -} - -impl<'a> Path<'a> for UnionIndex<'a> { - type Data = JsonValue; - - fn find(&self, input: JsonPathValue<'a, Self::Data>) -> Vec> { - self.indexes - .iter() - .flat_map(|e| e.find(input.clone())) - .collect() - } -} - -/// process filter element like [?(op sign op)] -pub enum FilterPath<'a> { - Filter { - left: PathInstance<'a>, - right: PathInstance<'a>, - op: &'a FilterSign, - }, - Or { - left: PathInstance<'a>, - right: PathInstance<'a>, - }, - And { - left: PathInstance<'a>, - right: PathInstance<'a>, - }, -} - -impl<'a> FilterPath<'a> { - pub(crate) fn new(expr: &'a FilterExpression, root: &'a JsonValue) -> Self { - match expr { - FilterExpression::Atom(left, op, right) => FilterPath::Filter { - left: process_operand(left, root), - right: process_operand(right, root), - op, - }, - FilterExpression::And(l, r) => FilterPath::And { - left: Box::new(FilterPath::new(l, root)), - right: Box::new(FilterPath::new(r, root)), - }, - FilterExpression::Or(l, r) => FilterPath::Or { - left: Box::new(FilterPath::new(l, root)), - right: Box::new(FilterPath::new(r, root)), - }, - } - } - fn compound( - one: &'a FilterSign, - two: &'a FilterSign, - left: Vec>, - right: Vec>, - ) -> bool { - FilterPath::process_atom(one, left.clone(), right.clone()) - || FilterPath::process_atom(two, left, right) - } - fn process_atom( - op: &'a FilterSign, - left: Vec>, - right: Vec>, - ) -> bool { - match op { - FilterSign::Equal => eq( - JsonPathValue::into_data(left), - JsonPathValue::into_data(right), - ), - FilterSign::Unequal => !FilterPath::process_atom(&FilterSign::Equal, left, right), - FilterSign::Less => less( - JsonPathValue::into_data(left), - JsonPathValue::into_data(right), - ), - FilterSign::LeOrEq => { - FilterPath::compound(&FilterSign::Less, &FilterSign::Equal, left, right) - } - FilterSign::Greater => !FilterPath::process_atom(&FilterSign::LeOrEq, left, right), - FilterSign::GrOrEq => !FilterPath::process_atom(&FilterSign::Less, left, right), - FilterSign::Regex => regex( - JsonPathValue::into_data(left), - JsonPathValue::into_data(right), - ), - FilterSign::In => inside( - JsonPathValue::into_data(left), - JsonPathValue::into_data(right), - ), - FilterSign::Nin => !FilterPath::process_atom(&FilterSign::In, left, right), - FilterSign::NoneOf => !FilterPath::process_atom(&FilterSign::AnyOf, left, right), - FilterSign::AnyOf => any_of( - JsonPathValue::into_data(left), - JsonPathValue::into_data(right), - ), - FilterSign::SubSetOf => sub_set_of( - JsonPathValue::into_data(left), - JsonPathValue::into_data(right), - ), - FilterSign::Exists => !JsonPathValue::into_data(left).is_empty(), - FilterSign::Size => size( - JsonPathValue::into_data(left), - JsonPathValue::into_data(right), - ), - } - } - - fn process(&self, curr_el: &'a JsonValue) -> bool { - match self { - FilterPath::Filter { left, right, op } => { - FilterPath::process_atom(op, left.find(Slice(curr_el)), right.find(Slice(curr_el))) - } - FilterPath::Or { left, right } => { - if !JsonPathValue::into_data(left.find(Slice(curr_el))).is_empty() { - true - } else { - !JsonPathValue::into_data(right.find(Slice(curr_el))).is_empty() - } - } - FilterPath::And { left, right } => { - if JsonPathValue::into_data(left.find(Slice(curr_el))).is_empty() { - false - } else { - !JsonPathValue::into_data(right.find(Slice(curr_el))).is_empty() - } - } - } - } -} - -impl<'a> Path<'a> for FilterPath<'a> { - type Data = JsonValue; - - fn find(&self, input: JsonPathValue<'a, Self::Data>) -> Vec> { - input.flat_map_slice(|data| { - let mut res = vec![]; - if let Some(elems) = data.as_array() { - for el in elems.iter() { - if self.process(el) { - res.push(Slice(el)) - } - } - } else if self.process(data) { - res.push(Slice(data)) - } - if res.is_empty() { - vec![NoValue] - } else { - res - } - }) - } -} diff --git a/dozer-sql/jsonpath/src/path/json.rs b/dozer-sql/jsonpath/src/path/json.rs deleted file mode 100644 index f805f7525c..0000000000 --- a/dozer-sql/jsonpath/src/path/json.rs +++ /dev/null @@ -1,157 +0,0 @@ -use dozer_types::json_types::{DestructuredJsonRef, JsonValue}; -use regex::Regex; - -/// compare sizes of json elements -/// The method expects to get a number on the right side and array or string or object on the left -/// where the number of characters, elements or fields will be compared respectively. -pub fn size(left: Vec<&JsonValue>, right: Vec<&JsonValue>) -> bool { - let Some(n) = right.first().and_then(|v| v.to_usize()) else { - return false; - }; - left.iter().all(|el| match el.destructure_ref() { - DestructuredJsonRef::String(v) => v.len() == n, - DestructuredJsonRef::Array(elems) => elems.len() == n, - DestructuredJsonRef::Object(fields) => fields.len() == n, - _ => false, - }) -} - -/// ensure the array on the left side is a subset of the array on the right side. -pub fn sub_set_of(left: Vec<&JsonValue>, right: Vec<&JsonValue>) -> bool { - if left.is_empty() { - return true; - } - if right.is_empty() { - return false; - } - - if let Some(elems) = left.first().and_then(|e| (*e).as_array()) { - if let Some(right_elems) = right.first().and_then(|v| v.as_array()) { - if right_elems.is_empty() { - return false; - } - - for el in elems { - let mut res = false; - - for r in right_elems.iter() { - if el.eq(r) { - res = true - } - } - if !res { - return false; - } - } - return true; - } - } - false -} - -/// ensure at least one element in the array on the left side belongs to the array on the right side. -pub fn any_of(left: Vec<&JsonValue>, right: Vec<&JsonValue>) -> bool { - if left.is_empty() { - return true; - } - if right.is_empty() { - return false; - } - - let Some(elems) = right.first().and_then(|v| v.as_array()) else { - return false; - }; - if elems.is_empty() { - return false; - } - - for el in left.iter() { - if let Some(left_elems) = el.as_array() { - for l in left_elems.iter() { - for r in elems.iter() { - if l.eq(r) { - return true; - } - } - } - } else { - for r in elems.iter() { - if el.eq(&r) { - return true; - } - } - } - } - false -} - -/// ensure that the element on the left sides mathes the regex on the right side -pub fn regex(left: Vec<&JsonValue>, right: Vec<&JsonValue>) -> bool { - if left.is_empty() || right.is_empty() { - return false; - } - - let Some(str) = right.first().and_then(|v| v.as_string()) else { - return false; - }; - if let Ok(regex) = Regex::new(str) { - for el in left.iter() { - if let Some(v) = el.as_string() { - if regex.is_match(v) { - return true; - } - } - } - } - false -} - -/// ensure that the element on the left side belongs to the array on the right side. -pub fn inside(left: Vec<&JsonValue>, right: Vec<&JsonValue>) -> bool { - if left.is_empty() { - return false; - } - - let Some(first) = right.first() else { - return false; - }; - if let Some(elems) = first.as_array() { - for el in left.iter() { - if elems.contains(el) { - return true; - } - } - } else if let Some(elems) = first.as_object() { - for el in left.iter() { - for r in elems.values() { - if el.eq(&r) { - return true; - } - } - } - } - false -} - -/// ensure the number on the left side is less the number on the right side -pub fn less(left: Vec<&JsonValue>, right: Vec<&JsonValue>) -> bool { - if left.len() == 1 && right.len() == 1 { - let left_no = left.first().and_then(|v| v.as_number()); - let right_no = right.first().and_then(|v| v.as_number()); - match (left_no, right_no) { - (Some(l), Some(r)) => l < r, - _ => false, - } - } else { - false - } -} - -/// compare elements -pub fn eq(left: Vec<&JsonValue>, right: Vec<&JsonValue>) -> bool { - if left.len() != right.len() { - false - } else { - left.iter().zip(right).map(|(a, b)| a.eq(&b)).all(|a| a) - } -} diff --git a/dozer-sql/jsonpath/src/path/mod.rs b/dozer-sql/jsonpath/src/path/mod.rs deleted file mode 100644 index edbf8ad4ad..0000000000 --- a/dozer-sql/jsonpath/src/path/mod.rs +++ /dev/null @@ -1,79 +0,0 @@ -use crate::parser::model::{Function, JsonPath, JsonPathIndex, Operand}; -use crate::path::index::{ArrayIndex, ArraySlice, Current, FilterPath, UnionIndex}; -use crate::path::top::{ - Chain, DescentObject, DescentWildcard, FnPath, IdentityPath, ObjectField, RootPointer, Wildcard, -}; -use crate::JsonPathValue; -use dozer_types::json_types::JsonValue; - -/// The module is in charge of processing [[JsonPathIndex]] elements -mod index; -/// The module is a helper module providing the set of helping functions to process a json elements -mod json; -/// The module is responsible for processing of the [[JsonPath]] elements -mod top; - -/// The trait defining the behaviour of processing every separated element. -/// type Data usually stands for json [[JsonValue]] -/// The trait also requires to have a root json to process. -/// It needs in case if in the filter there will be a pointer to the absolute path -pub trait Path<'a> { - type Data; - /// when every element needs to handle independently - fn find(&self, input: JsonPathValue<'a, Self::Data>) -> Vec> { - vec![input] - } - /// when the whole output needs to handle - fn flat_find( - &self, - input: Vec>, - _is_search_length: bool, - ) -> Vec> { - input.into_iter().flat_map(|d| self.find(d)).collect() - } - /// defines when we need to invoke `find` or `flat_find` - fn needs_all(&self) -> bool { - false - } -} - -/// The basic type for instances. -pub type PathInstance<'a> = Box + 'a>; - -/// The major method to process the top part of json part -pub fn json_path_instance<'a>(json_path: &'a JsonPath, root: &'a JsonValue) -> PathInstance<'a> { - match json_path { - JsonPath::Root => Box::new(RootPointer::new(root)), - JsonPath::Field(key) => Box::new(ObjectField::new(key)), - JsonPath::Chain(chain) => Box::new(Chain::from(chain, root)), - JsonPath::Wildcard => Box::new(Wildcard {}), - JsonPath::Descent(key) => Box::new(DescentObject::new(key)), - JsonPath::DescentW => Box::new(DescentWildcard), - JsonPath::Current(value) => Box::new(Current::from(value, root)), - JsonPath::Index(index) => process_index(index, root), - JsonPath::Empty => Box::new(IdentityPath {}), - JsonPath::Fn(Function::Length) => Box::new(FnPath::Size), - } -} - -/// The method processes the indexes(all expressions indie []) -fn process_index<'a>(json_path_index: &'a JsonPathIndex, root: &'a JsonValue) -> PathInstance<'a> { - match json_path_index { - // We roundtrip through isize, because of a bug when the f64 representation is used - JsonPathIndex::Single(index) => { - Box::new(ArrayIndex::new(index.to_isize().unwrap() as usize)) - } - JsonPathIndex::Slice(s, e, step) => Box::new(ArraySlice::new(*s, *e, *step)), - JsonPathIndex::UnionKeys(elems) => Box::new(UnionIndex::from_keys(elems)), - JsonPathIndex::UnionIndex(elems) => Box::new(UnionIndex::from_indexes(elems)), - JsonPathIndex::Filter(fe) => Box::new(FilterPath::new(fe, root)), - } -} - -/// The method processes the operand inside the filter expressions -fn process_operand<'a>(op: &'a Operand, root: &'a JsonValue) -> PathInstance<'a> { - match op { - Operand::Static(v) => json_path_instance(&JsonPath::Root, v), - Operand::Dynamic(jp) => json_path_instance(jp, root), - } -} diff --git a/dozer-sql/jsonpath/src/path/top.rs b/dozer-sql/jsonpath/src/path/top.rs deleted file mode 100644 index 4cd5611fcd..0000000000 --- a/dozer-sql/jsonpath/src/path/top.rs +++ /dev/null @@ -1,284 +0,0 @@ -use crate::parser::model::*; -use crate::path::JsonPathValue::{NewValue, NoValue, Slice}; -use crate::path::{json_path_instance, JsonPathValue, Path, PathInstance}; -use dozer_types::json_types::{json, JsonValue}; - -/// to process the element [*] -pub(crate) struct Wildcard {} - -impl<'a> Path<'a> for Wildcard { - type Data = JsonValue; - - fn find(&self, data: JsonPathValue<'a, Self::Data>) -> Vec> { - data.flat_map_slice(|data| { - if let Some(elems) = data.as_array() { - elems.iter().map(Slice).collect() - } else if let Some(elems) = data.as_object() { - elems.values().map(Slice).collect() - } else { - vec![NoValue] - } - }) - } -} - -/// empty path. Returns incoming data. -pub(crate) struct IdentityPath {} - -impl<'a> Path<'a> for IdentityPath { - type Data = JsonValue; - - fn find(&self, data: JsonPathValue<'a, Self::Data>) -> Vec> { - vec![data] - } -} - -pub(crate) struct EmptyPath {} - -impl<'a> Path<'a> for EmptyPath { - type Data = JsonValue; - - fn find(&self, _data: JsonPathValue<'a, Self::Data>) -> Vec> { - vec![] - } -} - -/// process $ element -pub(crate) struct RootPointer<'a, T> { - root: &'a T, -} - -impl<'a, T> RootPointer<'a, T> { - pub(crate) fn new(root: &'a T) -> RootPointer<'a, T> { - RootPointer { root } - } -} - -impl<'a> Path<'a> for RootPointer<'a, JsonValue> { - type Data = JsonValue; - - fn find(&self, _data: JsonPathValue<'a, Self::Data>) -> Vec> { - vec![Slice(self.root)] - } -} - -/// process object fields like ['key'] or .key -pub(crate) struct ObjectField<'a> { - key: &'a str, -} - -impl<'a> ObjectField<'a> { - pub(crate) fn new(key: &'a str) -> ObjectField<'a> { - ObjectField { key } - } -} - -impl<'a> Clone for ObjectField<'a> { - fn clone(&self) -> Self { - ObjectField::new(self.key) - } -} - -impl<'a> Path<'a> for FnPath { - type Data = JsonValue; - - fn flat_find( - &self, - input: Vec>, - is_search_length: bool, - ) -> Vec> { - if JsonPathValue::only_no_value(&input) { - return vec![NoValue]; - } - - let res = if is_search_length { - NewValue(json!(input.iter().filter(|v| v.has_value()).count())) - } else { - let take_len = |v: &JsonValue| { - if let Some(elems) = v.as_array() { - NewValue(json!(elems.len())) - } else { - NoValue - } - }; - - match input.first() { - Some(v) => match v { - NewValue(d) => take_len(d), - Slice(s) => take_len(s), - NoValue => NoValue, - }, - None => NoValue, - } - }; - vec![res] - } - - fn needs_all(&self) -> bool { - true - } -} - -pub(crate) enum FnPath { - Size, -} - -impl<'a> Path<'a> for ObjectField<'a> { - type Data = JsonValue; - - fn find(&self, data: JsonPathValue<'a, Self::Data>) -> Vec> { - let take_field = |v: &'a JsonValue| v.as_object()?.get(self.key); - - let res = match data { - Slice(js) => take_field(js).map(Slice).unwrap_or_else(|| NoValue), - _ => NoValue, - }; - vec![res] - } -} -/// the top method of the processing ..* -pub(crate) struct DescentWildcard; - -impl<'a> Path<'a> for DescentWildcard { - type Data = JsonValue; - - fn find(&self, data: JsonPathValue<'a, Self::Data>) -> Vec> { - data.map_slice(deep_flatten) - } -} - -fn deep_flatten(data: &JsonValue) -> Vec<&JsonValue> { - let mut acc = vec![]; - if let Some(elems) = data.as_array() { - for v in elems.iter() { - acc.push(v); - acc.append(&mut deep_flatten(v)); - } - } else if let Some(elems) = data.as_object() { - for v in elems.values() { - acc.push(v); - acc.append(&mut deep_flatten(v)); - } - } - acc -} - -fn deep_path_by_key<'a>(data: &'a JsonValue, key: ObjectField<'a>) -> Vec<&'a JsonValue> { - let mut level: Vec<&JsonValue> = JsonPathValue::into_data(key.find(data.into())); - if let Some(elems) = data.as_object() { - let mut next_levels: Vec<&JsonValue> = elems - .values() - .flat_map(|v| deep_path_by_key(v, key.clone())) - .collect(); - level.append(&mut next_levels); - } else if let Some(elems) = data.as_array() { - let mut next_levels: Vec<&JsonValue> = elems - .iter() - .flat_map(|v| deep_path_by_key(v, key.clone())) - .collect(); - level.append(&mut next_levels); - } - level -} - -/// processes decent object like .. -pub(crate) struct DescentObject<'a> { - key: &'a str, -} - -impl<'a> Path<'a> for DescentObject<'a> { - type Data = JsonValue; - - fn find(&self, data: JsonPathValue<'a, Self::Data>) -> Vec> { - data.flat_map_slice(|data| { - let res_col = deep_path_by_key(data, ObjectField::new(self.key)); - if res_col.is_empty() { - vec![NoValue] - } else { - JsonPathValue::map_vec(res_col) - } - }) - } -} - -impl<'a> DescentObject<'a> { - pub fn new(key: &'a str) -> Self { - DescentObject { key } - } -} - -/// the top method of the processing representing the chain of other operators -pub(crate) struct Chain<'a> { - chain: Vec>, - is_search_length: bool, -} - -impl<'a> Chain<'a> { - pub fn new(chain: Vec>, is_search_length: bool) -> Self { - Chain { - chain, - is_search_length, - } - } - pub fn from(chain: &'a [JsonPath], root: &'a JsonValue) -> Self { - let chain_len = chain.len(); - let is_search_length = if chain_len > 2 { - let mut res = false; - // if the result of the slice expected to be a slice, union or filter - - // length should return length of resulted array - // In all other cases, including single index, we should fetch item from resulting array - // and return length of that item - res = match chain.get(chain_len - 1).expect("chain element disappeared") { - JsonPath::Fn(Function::Length) => { - for item in chain.iter() { - match (item, res) { - // if we found union, slice, filter or wildcard - set search to true - ( - JsonPath::Index(JsonPathIndex::UnionIndex(_)) - | JsonPath::Index(JsonPathIndex::UnionKeys(_)) - | JsonPath::Index(JsonPathIndex::Slice(_, _, _)) - | JsonPath::Index(JsonPathIndex::Filter(_)) - | JsonPath::Wildcard, - false, - ) => { - res = true; - } - // if we found a fetching of single index - reset search to false - (JsonPath::Index(JsonPathIndex::Single(_)), true) => { - res = false; - } - (_, _) => {} - } - } - res - } - _ => false, - }; - res - } else { - false - }; - - Chain::new( - chain.iter().map(|p| json_path_instance(p, root)).collect(), - is_search_length, - ) - } -} - -impl<'a> Path<'a> for Chain<'a> { - type Data = JsonValue; - - fn find(&self, data: JsonPathValue<'a, Self::Data>) -> Vec> { - let mut res = vec![data]; - - for inst in self.chain.iter() { - if inst.needs_all() { - res = inst.flat_find(res, self.is_search_length) - } else { - res = res.into_iter().flat_map(|d| inst.find(d)).collect() - } - } - res - } -} diff --git a/dozer-sql/src/aggregation/aggregator.rs b/dozer-sql/src/aggregation/aggregator.rs deleted file mode 100644 index bf5c4521c4..0000000000 --- a/dozer-sql/src/aggregation/aggregator.rs +++ /dev/null @@ -1,559 +0,0 @@ -#![allow(clippy::enum_variant_names)] - -use crate::aggregation::avg::AvgAggregator; -use crate::aggregation::count::CountAggregator; -use crate::aggregation::max::MaxAggregator; -use crate::aggregation::min::MinAggregator; -use crate::aggregation::sum::SumAggregator; -use crate::calculate_err; -use crate::errors::PipelineError; -use dozer_types::chrono::{DateTime, FixedOffset, NaiveDate}; -use dozer_types::ordered_float::OrderedFloat; -use dozer_types::rust_decimal::Decimal; - -use enum_dispatch::enum_dispatch; -use std::collections::BTreeMap; - -use dozer_sql_expression::aggregate::AggregateFunctionType; -use dozer_sql_expression::execution::Expression; - -use crate::aggregation::max_append_only::MaxAppendOnlyAggregator; -use crate::aggregation::max_value::MaxValueAggregator; -use crate::aggregation::min_append_only::MinAppendOnlyAggregator; -use crate::aggregation::min_value::MinValueAggregator; -use crate::errors::PipelineError::{InvalidFunctionArgument, InvalidValue}; -use dozer_sql_expression::aggregate::AggregateFunctionType::MaxValue; -use dozer_types::types::{DozerDuration, Field, FieldType, Schema}; -use std::fmt::{Debug, Display, Formatter}; - -#[enum_dispatch] -pub trait Aggregator: Send + Sync + bincode::Encode + bincode::Decode { - fn init(&mut self, return_type: FieldType); - fn update(&mut self, old: &[Field], new: &[Field]) -> Result; - fn delete(&mut self, old: &[Field]) -> Result; - fn insert(&mut self, new: &[Field]) -> Result; -} - -#[enum_dispatch(Aggregator)] -#[derive(Debug, bincode::Encode, bincode::Decode)] -pub enum AggregatorEnum { - AvgAggregator, - MinAggregator, - MinAppendOnlyAggregator, - MinValueAggregator, - MaxAggregator, - MaxAppendOnlyAggregator, - MaxValueAggregator, - SumAggregator, - CountAggregator, -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Hash)] -pub enum AggregatorType { - Avg, - Count, - Max, - MaxAppendOnly, - MaxValue, - Min, - MinAppendOnly, - MinValue, - Sum, -} - -#[derive(Debug, bincode::Encode, bincode::Decode)] -pub(crate) struct OrderedAggregatorState { - function_type: AggregateFunctionType, - inner: OrderedAggregatorStateInner, -} - -#[derive(Debug, bincode::Encode, bincode::Decode)] -enum OrderedAggregatorStateInner { - UInt(BTreeMap), - U128(BTreeMap), - Int(BTreeMap), - I128(BTreeMap), - Float(#[bincode(with_serde)] BTreeMap, u64>), - Decimal(#[bincode(with_serde)] BTreeMap), - Timestamp(#[bincode(with_serde)] BTreeMap, u64>), - Date(#[bincode(with_serde)] BTreeMap), - Duration(BTreeMap), -} - -impl OrderedAggregatorState { - fn max_in_map(map: &BTreeMap) -> Option { - let (value, _count) = map.last_key_value()?; - Some(value.clone()) - } - fn min_in_map(map: &BTreeMap) -> Option { - Some(map.first_key_value()?.0.clone()) - } - - fn update_for_map( - map: &mut BTreeMap, - for_value: T, - incr: bool, - ) { - let amount = map.entry(for_value.clone()).or_insert(0); - if incr { - *amount += 1; - } else { - *amount -= 1; - } - if *amount == 0 { - map.remove(&for_value); - } - } - - fn update(&mut self, for_value: &Field, incr: bool) -> Result<(), PipelineError> { - if for_value == &Field::Null { - return Ok(()); - } - match &mut self.inner { - OrderedAggregatorStateInner::UInt(map) => Self::update_for_map( - map, - calculate_err!(for_value.as_uint(), self.function_type), - incr, - ), - OrderedAggregatorStateInner::U128(map) => Self::update_for_map( - map, - calculate_err!(for_value.as_u128(), self.function_type), - incr, - ), - - OrderedAggregatorStateInner::Int(map) => Self::update_for_map( - map, - calculate_err!(for_value.as_int(), self.function_type), - incr, - ), - - OrderedAggregatorStateInner::I128(map) => Self::update_for_map( - map, - calculate_err!(for_value.as_i128(), self.function_type), - incr, - ), - - OrderedAggregatorStateInner::Float(map) => Self::update_for_map( - map, - OrderedFloat(calculate_err!(for_value.as_float(), self.function_type)), - incr, - ), - - OrderedAggregatorStateInner::Decimal(map) => Self::update_for_map( - map, - calculate_err!(for_value.as_decimal(), self.function_type), - incr, - ), - - OrderedAggregatorStateInner::Timestamp(map) => Self::update_for_map( - map, - calculate_err!(for_value.as_timestamp(), self.function_type), - incr, - ), - - OrderedAggregatorStateInner::Date(map) => Self::update_for_map( - map, - calculate_err!(for_value.as_date(), self.function_type), - incr, - ), - - OrderedAggregatorStateInner::Duration(map) => Self::update_for_map( - map, - calculate_err!(for_value.as_duration(), self.function_type), - incr, - ), - } - Ok(()) - } - - #[inline] - pub(crate) fn incr(&mut self, value: &Field) -> Result<(), PipelineError> { - self.update(value, true)?; - Ok(()) - } - - #[inline] - pub(crate) fn decr(&mut self, value: &Field) -> Result<(), PipelineError> { - self.update(value, false)?; - Ok(()) - } - - pub(crate) fn new(function_type: AggregateFunctionType, field_type: FieldType) -> Option { - let inner = match field_type { - FieldType::UInt => OrderedAggregatorStateInner::UInt(Default::default()), - FieldType::U128 => OrderedAggregatorStateInner::U128(Default::default()), - FieldType::Int => OrderedAggregatorStateInner::Int(Default::default()), - FieldType::I128 => OrderedAggregatorStateInner::I128(Default::default()), - FieldType::Float => OrderedAggregatorStateInner::Float(Default::default()), - FieldType::Decimal => OrderedAggregatorStateInner::Decimal(Default::default()), - FieldType::Timestamp => OrderedAggregatorStateInner::Timestamp(Default::default()), - FieldType::Date => OrderedAggregatorStateInner::Date(Default::default()), - FieldType::Duration => OrderedAggregatorStateInner::Duration(Default::default()), - _ => return None, - }; - Some(Self { - function_type, - inner, - }) - } - - fn get_min_opt(&self) -> Option { - let field = match &self.inner { - OrderedAggregatorStateInner::UInt(map) => Field::UInt(Self::min_in_map(map)?), - OrderedAggregatorStateInner::U128(map) => Field::U128(Self::min_in_map(map)?), - OrderedAggregatorStateInner::Int(map) => Field::Int(Self::min_in_map(map)?), - OrderedAggregatorStateInner::I128(map) => Field::I128(Self::min_in_map(map)?), - OrderedAggregatorStateInner::Float(map) => Field::Float(Self::min_in_map(map)?), - OrderedAggregatorStateInner::Decimal(map) => Field::Decimal(Self::min_in_map(map)?), - OrderedAggregatorStateInner::Timestamp(map) => Field::Timestamp(Self::min_in_map(map)?), - OrderedAggregatorStateInner::Date(map) => Field::Date(Self::min_in_map(map)?), - OrderedAggregatorStateInner::Duration(map) => Field::Duration(Self::min_in_map(map)?), - }; - Some(field) - } - - #[inline] - pub(crate) fn get_min(&self) -> Field { - self.get_min_opt().unwrap_or(Field::Null) - } - - fn get_max_opt(&self) -> Option { - let field = match &self.inner { - OrderedAggregatorStateInner::UInt(map) => Field::UInt(Self::max_in_map(map)?), - OrderedAggregatorStateInner::U128(map) => Field::U128(Self::max_in_map(map)?), - OrderedAggregatorStateInner::Int(map) => Field::Int(Self::max_in_map(map)?), - OrderedAggregatorStateInner::I128(map) => Field::I128(Self::max_in_map(map)?), - OrderedAggregatorStateInner::Float(map) => Field::Float(Self::max_in_map(map)?), - OrderedAggregatorStateInner::Decimal(map) => Field::Decimal(Self::max_in_map(map)?), - OrderedAggregatorStateInner::Timestamp(map) => Field::Timestamp(Self::max_in_map(map)?), - OrderedAggregatorStateInner::Date(map) => Field::Date(Self::max_in_map(map)?), - OrderedAggregatorStateInner::Duration(map) => Field::Duration(Self::max_in_map(map)?), - }; - Some(field) - } - - #[inline] - pub(crate) fn get_max(&self) -> Field { - self.get_max_opt().unwrap_or(Field::Null) - } -} - -impl Display for AggregatorType { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - match self { - AggregatorType::Avg => f.write_str("avg"), - AggregatorType::Count => f.write_str("count"), - AggregatorType::Max => f.write_str("max"), - AggregatorType::MaxAppendOnly => f.write_str("max_append_only"), - AggregatorType::MaxValue => f.write_str("max_value"), - AggregatorType::Min => f.write_str("min"), - AggregatorType::MinAppendOnly => f.write_str("min_append_only"), - AggregatorType::MinValue => f.write_str("min_value"), - AggregatorType::Sum => f.write_str("sum"), - } - } -} - -pub fn get_aggregator_from_aggregator_type(typ: AggregatorType) -> AggregatorEnum { - match typ { - AggregatorType::Avg => AvgAggregator::new().into(), - AggregatorType::Count => CountAggregator::new().into(), - AggregatorType::Max => MaxAggregator::new().into(), - AggregatorType::MaxAppendOnly => MaxAppendOnlyAggregator::new().into(), - AggregatorType::MaxValue => MaxValueAggregator::new().into(), - AggregatorType::Min => MinAggregator::new().into(), - AggregatorType::MinAppendOnly => MinAppendOnlyAggregator::new().into(), - AggregatorType::MinValue => MinValueAggregator::new().into(), - AggregatorType::Sum => SumAggregator::new().into(), - } -} - -pub fn get_aggregator_type_from_aggregation_expression( - e: &Expression, - schema: &Schema, -) -> Result<(Vec, AggregatorType), PipelineError> { - match e { - Expression::AggregateFunction { - fun: AggregateFunctionType::Sum, - args, - } => Ok(( - vec![args - .first() - .ok_or_else(|| { - PipelineError::NotEnoughArguments(AggregateFunctionType::Sum.to_string()) - })? - .clone()], - AggregatorType::Sum, - )), - Expression::AggregateFunction { - fun: AggregateFunctionType::Min, - args, - } => Ok(( - vec![args - .first() - .ok_or_else(|| { - PipelineError::NotEnoughArguments(AggregateFunctionType::Min.to_string()) - })? - .clone()], - AggregatorType::Min, - )), - Expression::AggregateFunction { - fun: AggregateFunctionType::MinAppendOnly, - args, - } => Ok(( - vec![args - .first() - .ok_or_else(|| { - PipelineError::NotEnoughArguments( - AggregateFunctionType::MinAppendOnly.to_string(), - ) - })? - .clone()], - AggregatorType::MinAppendOnly, - )), - Expression::AggregateFunction { - fun: AggregateFunctionType::Max, - args, - } => Ok(( - vec![args - .first() - .ok_or_else(|| { - PipelineError::NotEnoughArguments(AggregateFunctionType::Max.to_string()) - })? - .clone()], - AggregatorType::Max, - )), - Expression::AggregateFunction { - fun: AggregateFunctionType::MaxAppendOnly, - args, - } => Ok(( - vec![args - .first() - .ok_or_else(|| { - PipelineError::NotEnoughArguments( - AggregateFunctionType::MaxAppendOnly.to_string(), - ) - })? - .clone()], - AggregatorType::MaxAppendOnly, - )), - Expression::AggregateFunction { - fun: AggregateFunctionType::MaxValue, - args, - } => Ok(( - vec![ - args.first() - .ok_or_else(|| { - PipelineError::NotEnoughArguments( - AggregateFunctionType::MaxValue.to_string(), - ) - })? - .clone(), - args.get(1) - .ok_or_else(|| { - PipelineError::NotEnoughArguments( - AggregateFunctionType::MaxValue.to_string(), - ) - })? - .clone(), - ], - AggregatorType::MaxValue, - )), - Expression::AggregateFunction { - fun: AggregateFunctionType::MinValue, - args, - } => Ok(( - vec![ - args.first() - .ok_or_else(|| { - PipelineError::NotEnoughArguments( - AggregateFunctionType::MinValue.to_string(), - ) - })? - .clone(), - args.get(1) - .ok_or_else(|| { - PipelineError::NotEnoughArguments( - AggregateFunctionType::MinValue.to_string(), - ) - })? - .clone(), - ], - AggregatorType::MinValue, - )), - Expression::AggregateFunction { - fun: AggregateFunctionType::Avg, - args, - } => Ok(( - vec![args - .first() - .ok_or_else(|| { - PipelineError::NotEnoughArguments(AggregateFunctionType::Avg.to_string()) - })? - .clone()], - AggregatorType::Avg, - )), - Expression::AggregateFunction { - fun: AggregateFunctionType::Count, - args, - } => Ok(( - vec![args - .first() - .ok_or_else(|| { - PipelineError::NotEnoughArguments(AggregateFunctionType::Count.to_string()) - })? - .clone()], - AggregatorType::Count, - )), - _ => Err(PipelineError::InvalidFunction(e.to_string(schema))), - } -} - -pub fn update_val_map( - fields: &[Field], - val_delta: u64, - decr: bool, - field_map: &mut BTreeMap, - return_map: &mut BTreeMap>, -) -> Result<(), PipelineError> { - let field = match fields.first() { - Some(v) => v, - None => { - return Err(InvalidFunctionArgument( - MaxValue.to_string(), - Field::Null, - 0, - )) - } - }; - let return_field = match fields.get(1) { - Some(v) => v.clone(), - None => { - return Err(InvalidFunctionArgument( - MaxValue.to_string(), - Field::Null, - 0, - )) - } - }; - - if field == &Field::Null { - return Ok(()); - } - - let get_prev_count = field_map.get(field); - let prev_count = match get_prev_count { - Some(v) => *v, - None => 0_u64, - }; - let mut new_count = prev_count; - if decr { - new_count = new_count.wrapping_sub(val_delta); - } else { - new_count = new_count.wrapping_add(val_delta); - } - if new_count < 1 { - field_map.remove(field); - } else if field_map.contains_key(field) { - if let Some(val) = field_map.get_mut(field) { - *val = new_count; - } - } else { - field_map.insert(field.clone(), new_count); - } - - if !decr { - if return_map.contains_key(field) { - if let Some(val) = return_map.get_mut(field) { - val.insert(0, return_field); - } - } else { - return_map.insert(field.clone(), vec![return_field]); - } - } else if return_map.contains_key(field) { - if let Some(val) = return_map.get_mut(field) { - let idx = match val.iter().position(|r| r.clone() == return_field) { - Some(id) => id, - None => return Err(InvalidValue(format!("{:?}", val))), - }; - val.remove(idx); - if val.is_empty() { - return_map.remove(field); - } - } - } - - let _res = format!("{:?}", field_map); - let _res_return = format!("{:?}", return_map); - - Ok(()) -} - -#[macro_export] -macro_rules! deserialize_u8 { - ($stmt:expr) => { - match $stmt { - Some(v) => u8::from_be_bytes(deserialize!(v)), - None => 0_u8, - } - }; -} - -#[macro_export] -macro_rules! check_nan_f64 { - ($stmt:expr) => { - if $stmt.is_nan() { - 0_f64 - } else { - $stmt - } - }; -} - -#[macro_export] -macro_rules! check_nan_decimal { - ($stmt:expr) => { - if $stmt.is_nan() { - dozer_types::rust_decimal::Decimal::zero() - } else { - $stmt - } - }; -} - -#[macro_export] -macro_rules! try_unwrap { - ($stmt:expr) => { - $stmt.unwrap_or_else(|e| panic!("{}", e.to_string())) - }; -} - -#[macro_export] -macro_rules! calculate_err { - ($stmt:expr, $aggr:expr) => { - $stmt.ok_or(PipelineError::InvalidReturnType(format!( - "Failed to calculate {}", - $aggr - )))? - }; -} - -#[macro_export] -macro_rules! calculate_err_field { - ($stmt:expr, $aggr:expr, $field:expr) => { - $stmt.ok_or(PipelineError::InvalidReturnType(format!( - "Failed to calculate {} while parsing {}", - $aggr, $field - )))? - }; -} - -#[macro_export] -macro_rules! calculate_err_type { - ($stmt:expr, $aggr:expr, $return_type:expr) => { - $stmt.ok_or(PipelineError::InvalidReturnType(format!( - "Failed to calculate {} while casting {}", - $aggr, $return_type - )))? - }; -} diff --git a/dozer-sql/src/aggregation/avg.rs b/dozer-sql/src/aggregation/avg.rs deleted file mode 100644 index 92878cd5d7..0000000000 --- a/dozer-sql/src/aggregation/avg.rs +++ /dev/null @@ -1,173 +0,0 @@ -use crate::aggregation::aggregator::Aggregator; -use crate::aggregation::sum::{get_sum, SumState}; -use crate::errors::PipelineError; -use crate::errors::PipelineError::InvalidValue; -use dozer_sql_expression::aggregate::AggregateFunctionType::Avg; -use dozer_sql_expression::num_traits::FromPrimitive; -use dozer_types::arrow::datatypes::ArrowNativeTypeOp; -use dozer_types::ordered_float::OrderedFloat; -use dozer_types::rust_decimal::Decimal; - -use dozer_types::types::{DozerDuration, Field, FieldType, TimeUnit}; - -use std::ops::Div; - -#[derive(Debug, bincode::Encode, bincode::Decode)] -pub struct AvgAggregator { - current_state: SumState, - current_count: u64, - return_type: Option, -} - -impl AvgAggregator { - pub fn new() -> Self { - Self { - current_state: SumState { - int_state: 0_i64, - int8_state: 0_i8, - i128_state: 0_i128, - uint_state: 0_u64, - u128_state: 0_u128, - float_state: 0_f64, - decimal_state: Decimal::from_f64(0_f64).unwrap(), - duration_state: std::time::Duration::new(0, 0), - }, - current_count: 0_u64, - return_type: None, - } - } -} - -impl Aggregator for AvgAggregator { - fn init(&mut self, return_type: FieldType) { - self.return_type = Some(return_type); - } - - fn update(&mut self, old: &[Field], new: &[Field]) -> Result { - self.delete(old)?; - self.insert(new) - } - - fn delete(&mut self, old: &[Field]) -> Result { - self.current_count -= old.len() as u64; - get_average( - old, - &mut self.current_state, - &mut self.current_count, - self.return_type, - true, - ) - } - - fn insert(&mut self, new: &[Field]) -> Result { - self.current_count += new.len() as u64; - get_average( - new, - &mut self.current_state, - &mut self.current_count, - self.return_type, - false, - ) - } -} - -fn get_average( - field: &[Field], - current_sum: &mut SumState, - current_count: &mut u64, - return_type: Option, - decr: bool, -) -> Result { - let sum = get_sum(field, current_sum, return_type, decr)?; - - match return_type { - Some(typ) => match typ { - FieldType::UInt => { - if *current_count == 0 { - return Ok(Field::Null); - } - let u_sum = sum.to_uint().ok_or(InvalidValue(sum.to_string())).unwrap(); - Ok(Field::UInt(u_sum.div_wrapping(*current_count))) - } - FieldType::U128 => { - if *current_count == 0 { - return Ok(Field::Null); - } - let u_sum = sum.to_u128().ok_or(InvalidValue(sum.to_string())).unwrap(); - Ok(Field::U128(u_sum.wrapping_div(*current_count as u128))) - } - FieldType::Int => { - if *current_count == 0 { - return Ok(Field::Null); - } - let i_sum = sum.to_int().ok_or(InvalidValue(sum.to_string())).unwrap(); - Ok(Field::Int(i_sum.div_wrapping(*current_count as i64))) - } - FieldType::Int8 => { - if *current_count == 0 { - return Ok(Field::Null); - } - let i_sum = sum.to_int8().ok_or(InvalidValue(sum.to_string())).unwrap(); - Ok(Field::Int8(i_sum.div_wrapping(*current_count as i8))) - } - FieldType::I128 => { - if *current_count == 0 { - return Ok(Field::Null); - } - let i_sum = sum.to_i128().ok_or(InvalidValue(sum.to_string())).unwrap(); - Ok(Field::I128(i_sum.div_wrapping(*current_count as i128))) - } - FieldType::Float => { - if *current_count == 0 { - return Ok(Field::Null); - } - let f_sum = sum.to_float().ok_or(InvalidValue(sum.to_string())).unwrap(); - Ok(Field::Float(OrderedFloat( - f_sum.div_wrapping(*current_count as f64), - ))) - } - FieldType::Decimal => { - if *current_count == 0 { - return Ok(Field::Null); - } - let d_sum = sum - .to_decimal() - .ok_or(InvalidValue(sum.to_string())) - .unwrap(); - Ok(Field::Decimal(d_sum.div(Decimal::from(*current_count)))) - } - FieldType::Duration => { - if *current_count == 0 { - return Ok(Field::Null); - } - let str_dur = format!("{:?}", sum.to_duration().unwrap().0); - let d_sum = sum - .to_duration() - .ok_or(InvalidValue(str_dur.clone())) - .unwrap(); - - Ok(Field::Duration(DozerDuration( - d_sum - .0 - .checked_div((*current_count) as u32) - .ok_or(InvalidValue(str_dur)) - .unwrap(), - TimeUnit::Nanoseconds, - ))) - } - FieldType::Boolean - | FieldType::String - | FieldType::Text - | FieldType::Date - | FieldType::Timestamp - | FieldType::Binary - | FieldType::Json - | FieldType::Point => Err(PipelineError::InvalidReturnType(format!( - "Not supported return type {typ} for {Avg}" - ))), - }, - None => Err(PipelineError::InvalidReturnType(format!( - "Not supported None return type for {Avg}" - ))), - } -} diff --git a/dozer-sql/src/aggregation/count.rs b/dozer-sql/src/aggregation/count.rs deleted file mode 100644 index d653f68ba7..0000000000 --- a/dozer-sql/src/aggregation/count.rs +++ /dev/null @@ -1,76 +0,0 @@ -use crate::aggregation::aggregator::Aggregator; -use crate::calculate_err_type; -use crate::errors::PipelineError; -use dozer_sql_expression::aggregate::AggregateFunctionType::Count; -use dozer_sql_expression::num_traits::FromPrimitive; -use dozer_types::ordered_float::OrderedFloat; -use dozer_types::rust_decimal::Decimal; -use dozer_types::types::{Field, FieldType}; - -#[derive(Debug, bincode::Encode, bincode::Decode)] -pub struct CountAggregator { - current_state: u64, - return_type: Option, -} - -impl CountAggregator { - pub fn new() -> Self { - Self { - current_state: 0_u64, - return_type: None, - } - } -} - -impl Aggregator for CountAggregator { - fn init(&mut self, return_type: FieldType) { - self.return_type = Some(return_type); - } - - fn update(&mut self, old: &[Field], new: &[Field]) -> Result { - self.delete(old)?; - self.insert(new) - } - - fn delete(&mut self, old: &[Field]) -> Result { - self.current_state -= old.len() as u64; - get_count(self.current_state, self.return_type) - } - - fn insert(&mut self, new: &[Field]) -> Result { - self.current_state += new.len() as u64; - get_count(self.current_state, self.return_type) - } -} - -fn get_count(count: u64, return_type: Option) -> Result { - match return_type { - Some(typ) => match typ { - FieldType::UInt => Ok(Field::UInt(count)), - FieldType::U128 => Ok(Field::U128(count as u128)), - FieldType::Int => Ok(Field::Int(count as i64)), - FieldType::Int8 => Ok(Field::Int8(count as i8)), - FieldType::I128 => Ok(Field::I128(count as i128)), - FieldType::Float => Ok(Field::Float(OrderedFloat::from(count as f64))), - FieldType::Decimal => Ok(Field::Decimal(calculate_err_type!( - Decimal::from_f64(count as f64), - Count, - FieldType::Decimal - ))), - FieldType::Duration => Ok(Field::Int(count as i64)), - FieldType::Boolean - | FieldType::String - | FieldType::Text - | FieldType::Date - | FieldType::Timestamp - | FieldType::Binary - | FieldType::Json - | FieldType::Point => Err(PipelineError::InvalidReturnType(format!( - "Not supported return type {typ} for {Count}" - ))), - }, - None => Err(PipelineError::InvalidReturnType(format!( - "Not supported None return type for {Count}" - ))), - } -} diff --git a/dozer-sql/src/aggregation/factory.rs b/dozer-sql/src/aggregation/factory.rs deleted file mode 100644 index c91d333b0b..0000000000 --- a/dozer-sql/src/aggregation/factory.rs +++ /dev/null @@ -1,148 +0,0 @@ -use crate::planner::projection::CommonPlanner; -use crate::projection::processor::ProjectionProcessor; -use crate::{aggregation::processor::AggregationProcessor, errors::PipelineError}; -use dozer_core::event::EventHub; -use dozer_core::{ - node::{PortHandle, Processor, ProcessorFactory}, - DEFAULT_PORT_HANDLE, -}; -use dozer_sql_expression::sqlparser::ast::{Expr, SelectItem}; -use dozer_types::errors::internal::BoxedError; -use dozer_types::models::udf_config::UdfConfig; -use dozer_types::parking_lot::Mutex; -use dozer_types::tonic::async_trait; -use dozer_types::types::Schema; -use std::collections::HashMap; -use std::sync::Arc; -use tokio::runtime::Runtime; - -#[derive(Debug)] -pub struct AggregationProcessorFactory { - id: String, - projection: Vec, - group_by: Vec, - having: Option, - enable_probabilistic_optimizations: bool, - udfs: Vec, - runtime: Arc, - - /// Type name can only be determined after schema propagation. - type_name: Mutex>, -} - -impl AggregationProcessorFactory { - pub fn new( - id: String, - projection: Vec, - group_by: Vec, - having: Option, - enable_probabilistic_optimizations: bool, - udfs: Vec, - runtime: Arc, - ) -> Self { - Self { - id, - projection, - group_by, - having, - enable_probabilistic_optimizations, - udfs, - runtime, - type_name: Mutex::new(None), - } - } - - async fn get_planner(&self, input_schema: Schema) -> Result { - let mut projection_planner = - CommonPlanner::new(input_schema, self.udfs.as_slice(), self.runtime.clone()); - projection_planner - .plan( - self.projection.clone(), - self.group_by.clone(), - self.having.clone(), - ) - .await?; - Ok(projection_planner) - } -} - -#[async_trait] -impl ProcessorFactory for AggregationProcessorFactory { - fn type_name(&self) -> String { - self.type_name - .lock() - .as_deref() - .unwrap_or("Aggregation") - .to_string() - } - fn get_input_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - fn get_output_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - async fn get_output_schema( - &self, - _output_port: &PortHandle, - input_schemas: &HashMap, - ) -> Result { - let input_schema = input_schemas - .get(&DEFAULT_PORT_HANDLE) - .ok_or(PipelineError::InvalidPortHandle(DEFAULT_PORT_HANDLE))?; - - let planner = self.get_planner(input_schema.clone()).await?; - - *self.type_name.lock() = Some( - if is_projection(&planner) { - "Projection" - } else { - "Aggregation" - } - .to_string(), - ); - - Ok(planner.post_projection_schema) - } - - async fn build( - &self, - input_schemas: HashMap, - _output_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - let input_schema = input_schemas - .get(&DEFAULT_PORT_HANDLE) - .ok_or(PipelineError::InvalidPortHandle(DEFAULT_PORT_HANDLE))?; - - let planner = self.get_planner(input_schema.clone()).await?; - - let processor: Box = if is_projection(&planner) { - Box::new(ProjectionProcessor::new( - input_schema.clone(), - planner.projection_output, - )?) - } else { - Box::new(AggregationProcessor::new( - self.id.clone(), - planner.groupby, - planner.aggregation_output, - planner.projection_output, - planner.having, - input_schema.clone(), - planner.post_aggregation_schema, - self.enable_probabilistic_optimizations, - )?) - }; - Ok(processor) - } - - fn id(&self) -> String { - self.id.clone() - } -} - -fn is_projection(planner: &CommonPlanner) -> bool { - planner.aggregation_output.is_empty() && planner.groupby.is_empty() -} diff --git a/dozer-sql/src/aggregation/max.rs b/dozer-sql/src/aggregation/max.rs deleted file mode 100644 index cc67f9418c..0000000000 --- a/dozer-sql/src/aggregation/max.rs +++ /dev/null @@ -1,70 +0,0 @@ -use crate::aggregation::aggregator::Aggregator; -use crate::errors::PipelineError; -use dozer_sql_expression::aggregate::AggregateFunctionType::Max; -use dozer_types::types::{Field, FieldType}; - -use super::aggregator::OrderedAggregatorState; - -#[derive(Debug, bincode::Encode, bincode::Decode)] -pub struct MaxAggregator { - current_state: Option, - return_type: Option, -} - -impl MaxAggregator { - pub fn new() -> Self { - Self { - current_state: None, - return_type: None, - } - } -} - -impl Aggregator for MaxAggregator { - fn init(&mut self, return_type: FieldType) { - self.return_type = Some(return_type); - self.current_state = OrderedAggregatorState::new(Max, return_type); - } - - fn update(&mut self, old: &[Field], new: &[Field]) -> Result { - self.delete(old)?; - self.insert(new) - } - - fn delete(&mut self, old: &[Field]) -> Result { - let state = self.get_state()?; - for field in old { - state.decr(field)?; - } - Ok(state.get_max()) - } - - fn insert(&mut self, new: &[Field]) -> Result { - let state = self.get_state()?; - for field in new { - state.incr(field)?; - } - Ok(state.get_max()) - } -} - -impl MaxAggregator { - fn get_state(&mut self) -> Result<&mut OrderedAggregatorState, PipelineError> { - self.current_state.as_mut().ok_or_else(|| { - match self - .return_type - .expect("MaxAggregator processor not initialized") - { - typ @ (FieldType::Boolean - | FieldType::String - | FieldType::Text - | FieldType::Binary - | FieldType::Json - | FieldType::Point) => PipelineError::InvalidReturnType(format!( - "Not supported return type {typ} for {Max}" - )), - _ => panic!("MaxAggregator processor not correctly initialized"), - } - }) - } -} diff --git a/dozer-sql/src/aggregation/max_append_only.rs b/dozer-sql/src/aggregation/max_append_only.rs deleted file mode 100644 index d586ff5eae..0000000000 --- a/dozer-sql/src/aggregation/max_append_only.rs +++ /dev/null @@ -1,189 +0,0 @@ -use crate::aggregation::aggregator::Aggregator; - -use crate::calculate_err_field; -use crate::errors::{PipelineError, UnsupportedSqlError}; -use dozer_sql_expression::aggregate::AggregateFunctionType::MaxAppendOnly; - -use dozer_types::chrono::{DateTime, FixedOffset, NaiveDate, Utc}; -use dozer_types::ordered_float::OrderedFloat; -use dozer_types::rust_decimal::Decimal; - -use dozer_types::types::{DozerDuration, Field, FieldType, TimeUnit}; - -#[derive(Debug, bincode::Encode, bincode::Decode)] -pub struct MaxAppendOnlyAggregator { - current_state: Field, - return_type: Option, -} - -impl MaxAppendOnlyAggregator { - pub fn new() -> Self { - Self { - current_state: Field::Null, - return_type: None, - } - } - - pub fn update_state(&mut self, field: Field) { - self.current_state = field; - } -} - -impl Aggregator for MaxAppendOnlyAggregator { - fn init(&mut self, return_type: FieldType) { - self.return_type = Some(return_type); - } - - fn update(&mut self, _old: &[Field], _new: &[Field]) -> Result { - Err(PipelineError::UnsupportedSqlError( - UnsupportedSqlError::GenericError("Append only".to_string()), - )) - } - - fn delete(&mut self, _old: &[Field]) -> Result { - Err(PipelineError::UnsupportedSqlError( - UnsupportedSqlError::GenericError("Append only".to_string()), - )) - } - - fn insert(&mut self, new: &[Field]) -> Result { - let cur_max = self.current_state.clone(); - - for val in new { - if val == &Field::Null { - continue; - } - match self.return_type { - Some(typ) => match typ { - FieldType::UInt => { - let new_val = calculate_err_field!(val.to_uint(), MaxAppendOnly, val); - let max_val = match cur_max { - Field::Null => u64::MIN, - _ => calculate_err_field!(cur_max.to_uint(), MaxAppendOnly, val), - }; - if new_val > max_val { - self.update_state(Field::UInt(new_val)); - } - } - FieldType::U128 => { - let new_val = calculate_err_field!(val.to_u128(), MaxAppendOnly, val); - let max_val = match cur_max { - Field::Null => u128::MIN, - _ => calculate_err_field!(cur_max.to_u128(), MaxAppendOnly, val), - }; - - if new_val > max_val { - self.update_state(Field::U128(new_val)); - } - } - FieldType::Int => { - let new_val = calculate_err_field!(val.to_int(), MaxAppendOnly, val); - let max_val = match cur_max { - Field::Null => i64::MIN, - _ => calculate_err_field!(cur_max.to_int(), MaxAppendOnly, val), - }; - - if new_val > max_val { - self.update_state(Field::Int(new_val)); - } - } - FieldType::Int8 => { - let new_val = calculate_err_field!(val.to_int8(), MaxAppendOnly, val); - let max_val = match cur_max { - Field::Null => i8::MIN, - _ => calculate_err_field!(cur_max.to_int8(), MaxAppendOnly, val), - }; - - if new_val > max_val { - self.update_state(Field::Int8(new_val)); - } - } - FieldType::I128 => { - let new_val = calculate_err_field!(val.to_i128(), MaxAppendOnly, val); - let max_val = match cur_max { - Field::Null => i128::MIN, - _ => calculate_err_field!(cur_max.to_i128(), MaxAppendOnly, val), - }; - - if new_val > max_val { - self.update_state(Field::I128(new_val)); - } - } - FieldType::Float => { - let new_val = calculate_err_field!(val.to_float(), MaxAppendOnly, val); - let max_val = match cur_max { - Field::Null => f64::MIN, - _ => calculate_err_field!(cur_max.to_float(), MaxAppendOnly, val), - }; - - if new_val > max_val { - self.update_state(Field::Float(OrderedFloat(new_val))); - } - } - FieldType::Decimal => { - let new_val = calculate_err_field!(val.to_decimal(), MaxAppendOnly, val); - let max_val = match cur_max { - Field::Null => Decimal::MIN, - _ => calculate_err_field!(cur_max.to_decimal(), MaxAppendOnly, val), - }; - - if new_val > max_val { - self.update_state(Field::Decimal(new_val)); - } - } - FieldType::Timestamp => { - let new_val = calculate_err_field!(val.to_timestamp(), MaxAppendOnly, val); - let max_val = match cur_max { - Field::Null => DateTime::::from(DateTime::::MIN_UTC), - _ => calculate_err_field!(cur_max.to_timestamp(), MaxAppendOnly, val), - }; - - if new_val > max_val { - self.update_state(Field::Timestamp(new_val)); - } - } - FieldType::Date => { - let new_val = calculate_err_field!(val.to_date(), MaxAppendOnly, val); - let max_val = match cur_max { - Field::Null => NaiveDate::MIN, - _ => calculate_err_field!(cur_max.to_date(), MaxAppendOnly, val), - }; - - if new_val > max_val { - self.update_state(Field::Date(new_val)); - } - } - FieldType::Duration => { - let new_val = calculate_err_field!(val.to_duration(), MaxAppendOnly, val); - let max_val = match cur_max { - Field::Null => { - DozerDuration(std::time::Duration::ZERO, TimeUnit::Nanoseconds) - } - _ => calculate_err_field!(cur_max.to_duration(), MaxAppendOnly, val), - }; - - if new_val > max_val { - self.update_state(Field::Duration(new_val)); - } - } - FieldType::Boolean - | FieldType::String - | FieldType::Text - | FieldType::Binary - | FieldType::Json - | FieldType::Point => { - return Err(PipelineError::InvalidReturnType(format!( - "Not supported return type {typ} for {MaxAppendOnly}" - ))); - } - }, - None => { - return Err(PipelineError::InvalidReturnType(format!( - "Not supported None return type for {MaxAppendOnly}" - ))) - } - } - } - Ok(self.current_state.clone()) - } -} diff --git a/dozer-sql/src/aggregation/max_value.rs b/dozer-sql/src/aggregation/max_value.rs deleted file mode 100644 index 2e39568af5..0000000000 --- a/dozer-sql/src/aggregation/max_value.rs +++ /dev/null @@ -1,81 +0,0 @@ -use crate::aggregation::aggregator::{update_val_map, Aggregator}; -use crate::calculate_err; -use crate::errors::PipelineError; -use crate::errors::PipelineError::InvalidReturnType; -use dozer_sql_expression::aggregate::AggregateFunctionType::MaxValue; - -use dozer_types::types::{Field, FieldType}; -use std::collections::BTreeMap; - -#[derive(Debug, bincode::Encode, bincode::Decode)] -pub struct MaxValueAggregator { - current_state: BTreeMap, - return_state: BTreeMap>, - return_type: Option, -} - -impl MaxValueAggregator { - pub fn new() -> Self { - Self { - current_state: BTreeMap::new(), - return_state: BTreeMap::new(), - return_type: None, - } - } -} - -impl Aggregator for MaxValueAggregator { - fn init(&mut self, return_type: FieldType) { - self.return_type = Some(return_type); - } - - fn update(&mut self, old: &[Field], new: &[Field]) -> Result { - self.delete(old)?; - self.insert(new) - } - - fn delete(&mut self, old: &[Field]) -> Result { - update_val_map( - old, - 1_u64, - true, - &mut self.current_state, - &mut self.return_state, - )?; - get_max_value(&self.current_state, &self.return_state, self.return_type) - } - - fn insert(&mut self, new: &[Field]) -> Result { - update_val_map( - new, - 1_u64, - false, - &mut self.current_state, - &mut self.return_state, - )?; - get_max_value(&self.current_state, &self.return_state, self.return_type) - } -} - -fn get_max_value( - field_map: &BTreeMap, - return_map: &BTreeMap>, - return_type: Option, -) -> Result { - if field_map.is_empty() { - Ok(Field::Null) - } else { - let val = calculate_err!(field_map.keys().max(), MaxValue).clone(); - - match return_map.get(&val) { - Some(v) => match v.first() { - Some(v) => { - let value = v.clone(); - Ok(value) - } - None => Err(InvalidReturnType(format!("{:?}", return_type))), - }, - None => Err(InvalidReturnType(format!("{:?}", return_type))), - } - } -} diff --git a/dozer-sql/src/aggregation/min.rs b/dozer-sql/src/aggregation/min.rs deleted file mode 100644 index 767fb9bce9..0000000000 --- a/dozer-sql/src/aggregation/min.rs +++ /dev/null @@ -1,71 +0,0 @@ -use crate::aggregation::aggregator::Aggregator; -use crate::errors::PipelineError; -use dozer_sql_expression::aggregate::AggregateFunctionType::{self, Min}; - -use dozer_types::types::{Field, FieldType}; - -use super::aggregator::OrderedAggregatorState; - -#[derive(Debug, bincode::Encode, bincode::Decode)] -pub struct MinAggregator { - current_state: Option, - return_type: Option, -} - -impl MinAggregator { - pub fn new() -> Self { - Self { - current_state: None, - return_type: None, - } - } -} - -impl Aggregator for MinAggregator { - fn init(&mut self, return_type: FieldType) { - self.current_state = OrderedAggregatorState::new(AggregateFunctionType::Min, return_type); - self.return_type = Some(return_type); - } - - fn update(&mut self, old: &[Field], new: &[Field]) -> Result { - self.delete(old)?; - self.insert(new) - } - - fn delete(&mut self, old: &[Field]) -> Result { - let state = self.get_state()?; - for field in old { - state.decr(field)?; - } - Ok(state.get_min()) - } - - fn insert(&mut self, new: &[Field]) -> Result { - let state = self.get_state()?; - for field in new { - state.incr(field)?; - } - Ok(state.get_min()) - } -} - -impl MinAggregator { - fn get_state(&mut self) -> Result<&mut OrderedAggregatorState, PipelineError> { - self.current_state.as_mut().ok_or_else(|| { - match self - .return_type - .expect("MinAggregator processor not initialized") - { - typ @ (FieldType::Boolean - | FieldType::String - | FieldType::Text - | FieldType::Binary - | FieldType::Json - | FieldType::Point) => PipelineError::InvalidReturnType(format!( - "Not supported return type {typ} for {Min}" - )), - _ => panic!("MinAggregator processor not correctly initialized"), - } - }) - } -} diff --git a/dozer-sql/src/aggregation/min_append_only.rs b/dozer-sql/src/aggregation/min_append_only.rs deleted file mode 100644 index 6dbd471701..0000000000 --- a/dozer-sql/src/aggregation/min_append_only.rs +++ /dev/null @@ -1,189 +0,0 @@ -use crate::aggregation::aggregator::Aggregator; - -use crate::calculate_err_field; -use crate::errors::{PipelineError, UnsupportedSqlError}; -use dozer_sql_expression::aggregate::AggregateFunctionType::MinAppendOnly; - -use dozer_types::chrono::{DateTime, FixedOffset, NaiveDate, Utc}; -use dozer_types::ordered_float::OrderedFloat; -use dozer_types::rust_decimal::Decimal; - -use dozer_types::types::{DozerDuration, Field, FieldType, TimeUnit}; - -#[derive(Debug, bincode::Encode, bincode::Decode)] -pub struct MinAppendOnlyAggregator { - current_state: Field, - return_type: Option, -} - -impl MinAppendOnlyAggregator { - pub fn new() -> Self { - Self { - current_state: Field::Null, - return_type: None, - } - } - - pub fn update_state(&mut self, field: Field) { - self.current_state = field; - } -} - -impl Aggregator for MinAppendOnlyAggregator { - fn init(&mut self, return_type: FieldType) { - self.return_type = Some(return_type); - } - - fn update(&mut self, _old: &[Field], _new: &[Field]) -> Result { - Err(PipelineError::UnsupportedSqlError( - UnsupportedSqlError::GenericError("Append only".to_string()), - )) - } - - fn delete(&mut self, _old: &[Field]) -> Result { - Err(PipelineError::UnsupportedSqlError( - UnsupportedSqlError::GenericError("Append only".to_string()), - )) - } - - fn insert(&mut self, new: &[Field]) -> Result { - let cur_min = self.current_state.clone(); - - for val in new { - if val == &Field::Null { - continue; - } - match self.return_type { - Some(typ) => match typ { - FieldType::UInt => { - let new_val = calculate_err_field!(val.to_uint(), MinAppendOnly, val); - let min_val = match cur_min { - Field::Null => u64::MAX, - _ => calculate_err_field!(cur_min.to_uint(), MinAppendOnly, val), - }; - if new_val < min_val { - self.update_state(Field::UInt(new_val)); - } - } - FieldType::U128 => { - let new_val = calculate_err_field!(val.to_u128(), MinAppendOnly, val); - let min_val = match cur_min { - Field::Null => u128::MAX, - _ => calculate_err_field!(cur_min.to_u128(), MinAppendOnly, val), - }; - - if new_val < min_val { - self.update_state(Field::U128(new_val)); - } - } - FieldType::Int => { - let new_val = calculate_err_field!(val.to_int(), MinAppendOnly, val); - let min_val = match cur_min { - Field::Null => i64::MAX, - _ => calculate_err_field!(cur_min.to_int(), MinAppendOnly, val), - }; - - if new_val < min_val { - self.update_state(Field::Int(new_val)); - } - } - FieldType::Int8 => { - let new_val = calculate_err_field!(val.to_int8(), MinAppendOnly, val); - let min_val = match cur_min { - Field::Null => i8::MAX, - _ => calculate_err_field!(cur_min.to_int8(), MinAppendOnly, val), - }; - - if new_val < min_val { - self.update_state(Field::Int8(new_val)); - } - } - FieldType::I128 => { - let new_val = calculate_err_field!(val.to_i128(), MinAppendOnly, val); - let min_val = match cur_min { - Field::Null => i128::MAX, - _ => calculate_err_field!(cur_min.to_i128(), MinAppendOnly, val), - }; - - if new_val < min_val { - self.update_state(Field::I128(new_val)); - } - } - FieldType::Float => { - let new_val = calculate_err_field!(val.to_float(), MinAppendOnly, val); - let min_val = match cur_min { - Field::Null => f64::MAX, - _ => calculate_err_field!(cur_min.to_float(), MinAppendOnly, val), - }; - - if new_val < min_val { - self.update_state(Field::Float(OrderedFloat(new_val))); - } - } - FieldType::Decimal => { - let new_val = calculate_err_field!(val.to_decimal(), MinAppendOnly, val); - let min_val = match cur_min { - Field::Null => Decimal::MAX, - _ => calculate_err_field!(cur_min.to_decimal(), MinAppendOnly, val), - }; - - if new_val < min_val { - self.update_state(Field::Decimal(new_val)); - } - } - FieldType::Timestamp => { - let new_val = calculate_err_field!(val.to_timestamp(), MinAppendOnly, val); - let min_val = match cur_min { - Field::Null => DateTime::::from(DateTime::::MAX_UTC), - _ => calculate_err_field!(cur_min.to_timestamp(), MinAppendOnly, val), - }; - - if new_val < min_val { - self.update_state(Field::Timestamp(new_val)); - } - } - FieldType::Date => { - let new_val = calculate_err_field!(val.to_date(), MinAppendOnly, val); - let min_val = match cur_min { - Field::Null => NaiveDate::MAX, - _ => calculate_err_field!(cur_min.to_date(), MinAppendOnly, val), - }; - - if new_val < min_val { - self.update_state(Field::Date(new_val)); - } - } - FieldType::Duration => { - let new_val = calculate_err_field!(val.to_duration(), MinAppendOnly, val); - let min_val = match cur_min { - Field::Null => { - DozerDuration(std::time::Duration::MAX, TimeUnit::Nanoseconds) - } - _ => calculate_err_field!(cur_min.to_duration(), MinAppendOnly, val), - }; - - if new_val < min_val { - self.update_state(Field::Duration(new_val)); - } - } - FieldType::Boolean - | FieldType::String - | FieldType::Text - | FieldType::Binary - | FieldType::Json - | FieldType::Point => { - return Err(PipelineError::InvalidReturnType(format!( - "Not supported return type {typ} for {MinAppendOnly}" - ))); - } - }, - None => { - return Err(PipelineError::InvalidReturnType(format!( - "Not supported None return type for {MinAppendOnly}" - ))) - } - } - } - Ok(self.current_state.clone()) - } -} diff --git a/dozer-sql/src/aggregation/min_value.rs b/dozer-sql/src/aggregation/min_value.rs deleted file mode 100644 index 845635649e..0000000000 --- a/dozer-sql/src/aggregation/min_value.rs +++ /dev/null @@ -1,81 +0,0 @@ -use crate::aggregation::aggregator::{update_val_map, Aggregator}; -use crate::calculate_err; -use crate::errors::PipelineError; -use crate::errors::PipelineError::InvalidReturnType; -use dozer_sql_expression::aggregate::AggregateFunctionType::MinValue; - -use dozer_types::types::{Field, FieldType}; -use std::collections::BTreeMap; - -#[derive(Debug, bincode::Encode, bincode::Decode)] -pub struct MinValueAggregator { - current_state: BTreeMap, - return_state: BTreeMap>, - return_type: Option, -} - -impl MinValueAggregator { - pub fn new() -> Self { - Self { - current_state: BTreeMap::new(), - return_state: BTreeMap::new(), - return_type: None, - } - } -} - -impl Aggregator for MinValueAggregator { - fn init(&mut self, return_type: FieldType) { - self.return_type = Some(return_type); - } - - fn update(&mut self, old: &[Field], new: &[Field]) -> Result { - self.delete(old)?; - self.insert(new) - } - - fn delete(&mut self, old: &[Field]) -> Result { - update_val_map( - old, - 1_u64, - true, - &mut self.current_state, - &mut self.return_state, - )?; - get_min_value(&self.current_state, &self.return_state, self.return_type) - } - - fn insert(&mut self, new: &[Field]) -> Result { - update_val_map( - new, - 1_u64, - false, - &mut self.current_state, - &mut self.return_state, - )?; - get_min_value(&self.current_state, &self.return_state, self.return_type) - } -} - -fn get_min_value( - field_map: &BTreeMap, - return_map: &BTreeMap>, - return_type: Option, -) -> Result { - if field_map.is_empty() { - Ok(Field::Null) - } else { - let val = calculate_err!(field_map.keys().min(), MinValue).clone(); - - match return_map.get(&val) { - Some(v) => match v.first() { - Some(v) => { - let value = v.clone(); - Ok(value) - } - None => Err(InvalidReturnType(format!("{:?}", return_type))), - }, - None => Err(InvalidReturnType(format!("{:?}", return_type))), - } - } -} diff --git a/dozer-sql/src/aggregation/mod.rs b/dozer-sql/src/aggregation/mod.rs deleted file mode 100644 index 21bfebdfc8..0000000000 --- a/dozer-sql/src/aggregation/mod.rs +++ /dev/null @@ -1,14 +0,0 @@ -pub mod aggregator; -pub mod avg; -pub mod count; -pub mod factory; -pub mod max; -pub mod max_value; -pub mod min; -pub mod min_value; -pub mod processor; -pub mod sum; -mod tests; - -pub mod max_append_only; -pub mod min_append_only; diff --git a/dozer-sql/src/aggregation/processor.rs b/dozer-sql/src/aggregation/processor.rs deleted file mode 100644 index 4a9400c6b0..0000000000 --- a/dozer-sql/src/aggregation/processor.rs +++ /dev/null @@ -1,586 +0,0 @@ -#![allow(clippy::too_many_arguments)] - -use crate::aggregation::aggregator::Aggregator; -use crate::errors::PipelineError; -use crate::utils::record_hashtable_key::{get_record_hash, RecordKey}; -use dozer_core::channels::ProcessorChannelForwarder; -use dozer_core::node::Processor; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_sql_expression::execution::Expression; -use dozer_types::bincode; -use dozer_types::errors::internal::BoxedError; -use dozer_types::types::{Field, FieldType, Operation, Record, Schema, TableOperation}; -use std::collections::HashMap; - -use crate::aggregation::aggregator::{ - get_aggregator_from_aggregator_type, get_aggregator_type_from_aggregation_expression, - AggregatorEnum, AggregatorType, -}; -use dozer_core::epoch::Epoch; - -const DEFAULT_SEGMENT_KEY: &str = "DOZER_DEFAULT_SEGMENT_KEY"; - -#[derive(Debug, bincode::Encode, bincode::Decode)] -struct AggregationState { - count: usize, - states: Vec, - values: Option>, -} - -impl AggregationState { - pub fn new(types: &[AggregatorType], ret_types: &[FieldType]) -> Self { - let mut states: Vec = Vec::new(); - for (idx, typ) in types.iter().enumerate() { - let mut aggr = get_aggregator_from_aggregator_type(*typ); - aggr.init(ret_types[idx]); - states.push(aggr); - } - - Self { - count: 0, - states, - values: None, - } - } -} - -#[derive(Debug)] -pub struct AggregationProcessor { - _id: String, - dimensions: Vec, - measures: Vec>, - measures_types: Vec, - measures_return_types: Vec, - projections: Vec, - having: Option, - input_schema: Schema, - aggregation_schema: Schema, - states: HashMap, - default_segment_key: RecordKey, - having_eval_schema: Schema, - accurate_keys: bool, -} - -enum AggregatorOperation { - Insert, - Delete, - Update, -} - -impl AggregationProcessor { - pub fn new( - id: String, - dimensions: Vec, - measures: Vec, - projections: Vec, - having: Option, - input_schema: Schema, - aggregation_schema: Schema, - enable_probabilistic_optimizations: bool, - ) -> Result { - let mut aggr_types = Vec::new(); - let mut aggr_measures = Vec::new(); - let mut aggr_measures_ret_types = Vec::new(); - - for measure in measures { - let (aggr_measure, aggr_type) = - get_aggregator_type_from_aggregation_expression(&measure, &input_schema)?; - aggr_measures.push(aggr_measure); - aggr_types.push(aggr_type); - aggr_measures_ret_types.push(measure.get_type(&input_schema)?.return_type) - } - - let mut having_eval_schema_fields = input_schema.fields.clone(); - having_eval_schema_fields.extend(aggregation_schema.fields.clone()); - - let accurate_keys = !enable_probabilistic_optimizations; - - Ok(Self { - _id: id, - dimensions, - projections, - input_schema, - aggregation_schema, - states: Default::default(), - measures: aggr_measures, - having, - measures_types: aggr_types, - measures_return_types: aggr_measures_ret_types, - default_segment_key: { - let fields = vec![Field::String(DEFAULT_SEGMENT_KEY.into())]; - if accurate_keys { - RecordKey::Accurate(fields) - } else { - RecordKey::Hash(get_record_hash(fields.iter())) - } - }, - having_eval_schema: Schema { - fields: having_eval_schema_fields, - primary_index: vec![], - }, - accurate_keys, - }) - } - - fn calc_and_fill_measures( - curr_state: &mut AggregationState, - deleted_record: Option<&Record>, - inserted_record: Option<&Record>, - out_rec_delete: &mut Vec, - out_rec_insert: &mut Vec, - op: AggregatorOperation, - measures: &mut [Vec], - input_schema: &Schema, - ) -> Result, PipelineError> { - let mut new_fields: Vec = Vec::with_capacity(measures.len()); - - for (idx, measure) in measures.iter_mut().enumerate() { - let curr_aggr = &mut curr_state.states[idx]; - let curr_val_opt: Option<&Field> = curr_state.values.as_ref().map(|e| &e[idx]); - - let new_val = match op { - AggregatorOperation::Insert => { - let mut inserted_fields = Vec::with_capacity(measure.len()); - for m in measure { - inserted_fields.push(m.evaluate(inserted_record.unwrap(), input_schema)?); - } - if let Some(curr_val) = curr_val_opt { - out_rec_delete.push(curr_val.clone()); - } - curr_aggr.insert(&inserted_fields)? - } - AggregatorOperation::Delete => { - let mut deleted_fields = Vec::with_capacity(measure.len()); - for m in measure { - deleted_fields.push(m.evaluate(deleted_record.unwrap(), input_schema)?); - } - if let Some(curr_val) = curr_val_opt { - out_rec_delete.push(curr_val.clone()); - } - curr_aggr.delete(&deleted_fields)? - } - AggregatorOperation::Update => { - let mut deleted_fields = Vec::with_capacity(measure.len()); - for m in measure.iter_mut() { - deleted_fields.push(m.evaluate(deleted_record.unwrap(), input_schema)?); - } - let mut inserted_fields = Vec::with_capacity(measure.len()); - for m in measure { - inserted_fields.push(m.evaluate(inserted_record.unwrap(), input_schema)?); - } - if let Some(curr_val) = curr_val_opt { - out_rec_delete.push(curr_val.clone()); - } - curr_aggr.update(&deleted_fields, &inserted_fields)? - } - }; - out_rec_insert.push(new_val.clone()); - new_fields.push(new_val); - } - Ok(new_fields) - } - - fn agg_delete(&mut self, old: &mut Record) -> Result, PipelineError> { - let mut out_rec_delete: Vec = Vec::with_capacity(self.measures.len()); - let mut out_rec_insert: Vec = Vec::with_capacity(self.measures.len()); - - let key = if !self.dimensions.is_empty() { - Some(self.get_key(old)?) - } else { - None - }; - let key = key.as_ref().unwrap_or(&self.default_segment_key); - - let curr_state_opt = self.states.get_mut(key); - assert!( - curr_state_opt.is_some(), - "Unable to find aggregator state during DELETE operation" - ); - let curr_state = curr_state_opt.unwrap(); - - let new_values = Self::calc_and_fill_measures( - curr_state, - Some(old), - None, - &mut out_rec_delete, - &mut out_rec_insert, - AggregatorOperation::Delete, - &mut self.measures, - &self.input_schema, - )?; - - let (out_rec_delete_having_satisfied, out_rec_insert_having_satisfied) = - match &mut self.having { - None => (true, true), - Some(having) => ( - Self::having_is_satisfied( - &self.having_eval_schema, - old, - having, - &mut out_rec_delete, - )?, - Self::having_is_satisfied( - &self.having_eval_schema, - old, - having, - &mut out_rec_insert, - )?, - ), - }; - - let res = if curr_state.count == 1 { - self.states.remove(key); - if out_rec_delete_having_satisfied { - vec![Operation::Delete { - old: Self::build_projection( - old, - out_rec_delete, - &mut self.projections, - &self.aggregation_schema, - )?, - }] - } else { - vec![] - } - } else { - curr_state.count -= 1; - curr_state.values = Some(new_values); - - Self::generate_op_for_existing_segment( - out_rec_delete_having_satisfied, - out_rec_insert_having_satisfied, - out_rec_delete, - out_rec_insert, - old, - &mut self.projections, - &self.aggregation_schema, - )? - }; - - Ok(res) - } - - fn agg_insert(&mut self, new: &mut Record) -> Result, PipelineError> { - let mut out_rec_delete: Vec = Vec::with_capacity(self.measures.len()); - let mut out_rec_insert: Vec = Vec::with_capacity(self.measures.len()); - - let key = if !self.dimensions.is_empty() { - self.get_key(new)? - } else { - self.default_segment_key.clone() - }; - - let curr_state = self.states.entry(key).or_insert(AggregationState::new( - &self.measures_types, - &self.measures_return_types, - )); - - let new_values = Self::calc_and_fill_measures( - curr_state, - None, - Some(new), - &mut out_rec_delete, - &mut out_rec_insert, - AggregatorOperation::Insert, - &mut self.measures, - &self.input_schema, - )?; - - let (out_rec_delete_having_satisfied, out_rec_insert_having_satisfied) = - match &mut self.having { - None => (true, true), - Some(having) => ( - Self::having_is_satisfied( - &self.having_eval_schema, - new, - having, - &mut out_rec_delete, - )?, - Self::having_is_satisfied( - &self.having_eval_schema, - new, - having, - &mut out_rec_insert, - )?, - ), - }; - - let res = if curr_state.count == 0 { - if out_rec_insert_having_satisfied { - vec![Operation::Insert { - new: Self::build_projection( - new, - out_rec_insert, - &mut self.projections, - &self.aggregation_schema, - )?, - }] - } else { - vec![] - } - } else { - Self::generate_op_for_existing_segment( - out_rec_delete_having_satisfied, - out_rec_insert_having_satisfied, - out_rec_delete, - out_rec_insert, - new, - &mut self.projections, - &self.aggregation_schema, - )? - }; - - curr_state.count += 1; - curr_state.values = Some(new_values); - - Ok(res) - } - - fn generate_op_for_existing_segment( - out_rec_delete_having_satisfied: bool, - out_rec_insert_having_satisfied: bool, - out_rec_delete: Vec, - out_rec_insert: Vec, - rec: &mut Record, - projections: &mut Vec, - aggregation_schema: &Schema, - ) -> Result, PipelineError> { - Ok( - match ( - out_rec_delete_having_satisfied, - out_rec_insert_having_satisfied, - ) { - (false, true) => vec![Operation::Insert { - new: Self::build_projection( - rec, - out_rec_insert, - projections, - aggregation_schema, - )?, - }], - (true, false) => vec![Operation::Delete { - old: Self::build_projection( - rec, - out_rec_delete, - projections, - aggregation_schema, - )?, - }], - (true, true) => vec![Operation::Update { - new: Self::build_projection( - rec, - out_rec_insert, - projections, - aggregation_schema, - )?, - old: Self::build_projection( - rec, - out_rec_delete, - projections, - aggregation_schema, - )?, - }], - (false, false) => vec![], - }, - ) - } - - fn having_is_satisfied( - having_eval_schema: &Schema, - original_record: &mut Record, - having: &mut Expression, - out_rec: &mut Vec, - ) -> Result { - let original_record_len = original_record.values.len(); - Ok(match out_rec.len() { - 0 => false, - _ => { - original_record.values.extend(std::mem::take(out_rec)); - let r = having - .evaluate(original_record, having_eval_schema)? - .as_boolean() - .unwrap_or(false); - out_rec.extend( - original_record - .values - .drain(original_record_len..) - .collect::>(), - ); - r - } - }) - } - - fn agg_update( - &mut self, - old: &mut Record, - new: &mut Record, - key: RecordKey, - ) -> Result, PipelineError> { - let mut out_rec_delete: Vec = Vec::with_capacity(self.measures.len()); - let mut out_rec_insert: Vec = Vec::with_capacity(self.measures.len()); - - let curr_state_opt = self.states.get_mut(&key); - assert!( - curr_state_opt.is_some(), - "Unable to find aggregator state during UPDATE operation" - ); - let curr_state = curr_state_opt.unwrap(); - - let new_values = Self::calc_and_fill_measures( - curr_state, - Some(old), - Some(new), - &mut out_rec_delete, - &mut out_rec_insert, - AggregatorOperation::Update, - &mut self.measures, - &self.input_schema, - )?; - - let (out_rec_delete_having_satisfied, out_rec_insert_having_satisfied) = - match &mut self.having { - None => (true, true), - Some(having) => ( - Self::having_is_satisfied( - &self.having_eval_schema, - old, - having, - &mut out_rec_delete, - )?, - Self::having_is_satisfied( - &self.having_eval_schema, - new, - having, - &mut out_rec_insert, - )?, - ), - }; - - let res = match ( - out_rec_delete_having_satisfied, - out_rec_insert_having_satisfied, - ) { - (false, true) => vec![Operation::Insert { - new: Self::build_projection( - new, - out_rec_insert, - &mut self.projections, - &self.aggregation_schema, - )?, - }], - (true, false) => vec![Operation::Delete { - old: Self::build_projection( - old, - out_rec_delete, - &mut self.projections, - &self.aggregation_schema, - )?, - }], - (true, true) => vec![Operation::Update { - new: Self::build_projection( - new, - out_rec_insert, - &mut self.projections, - &self.aggregation_schema, - )?, - old: Self::build_projection( - old, - out_rec_delete, - &mut self.projections, - &self.aggregation_schema, - )?, - }], - (false, false) => vec![], - }; - - curr_state.values = Some(new_values); - Ok(res) - } - - pub fn build_projection( - original: &mut Record, - measures: Vec, - projections: &mut Vec, - aggregation_schema: &Schema, - ) -> Result { - let original_len = original.values.len(); - original.values.extend(measures); - let mut output = Vec::::with_capacity(projections.len()); - for exp in projections { - output.push(exp.evaluate(original, aggregation_schema)?); - } - original.values.drain(original_len..); - let mut output_record = Record::new(output); - - output_record.set_lifetime(original.get_lifetime()); - - Ok(output_record) - } - - pub fn aggregate(&mut self, mut op: Operation) -> Result, PipelineError> { - match op { - Operation::Insert { ref mut new } => Ok(self.agg_insert(new)?), - Operation::Delete { ref mut old } => Ok(self.agg_delete(old)?), - Operation::Update { - ref mut old, - ref mut new, - } => { - let (old_record_hash, new_record_hash) = if self.dimensions.is_empty() { - ( - self.default_segment_key.clone(), - self.default_segment_key.clone(), - ) - } else { - (self.get_key(old)?, self.get_key(new)?) - }; - - if old_record_hash == new_record_hash { - Ok(self.agg_update(old, new, old_record_hash)?) - } else { - let mut r = Vec::with_capacity(2); - r.extend(self.agg_delete(old)?); - r.extend(self.agg_insert(new)?); - Ok(r) - } - } - Operation::BatchInsert { new } => { - let mut result = vec![]; - for record in new { - result.extend(self.aggregate(Operation::Insert { new: record })?); - } - Ok(result) - } - } - } - - fn get_key(&mut self, record: &Record) -> Result { - let mut key = Vec::::with_capacity(self.dimensions.len()); - for dimension in self.dimensions.iter_mut() { - key.push(dimension.evaluate(record, &self.input_schema)?); - } - if self.accurate_keys { - Ok(RecordKey::Accurate(key)) - } else { - Ok(RecordKey::Hash(get_record_hash(key.iter()))) - } - } -} - -impl Processor for AggregationProcessor { - fn commit(&self, _epoch: &Epoch) -> Result<(), BoxedError> { - Ok(()) - } - - fn process( - &mut self, - op: TableOperation, - fw: &mut dyn ProcessorChannelForwarder, - ) -> Result<(), BoxedError> { - let ops = self.aggregate(op.op)?; - for output_op in ops { - fw.send(TableOperation::without_id(output_op, DEFAULT_PORT_HANDLE)); - } - Ok(()) - } -} diff --git a/dozer-sql/src/aggregation/sum.rs b/dozer-sql/src/aggregation/sum.rs deleted file mode 100644 index 03c16b7482..0000000000 --- a/dozer-sql/src/aggregation/sum.rs +++ /dev/null @@ -1,205 +0,0 @@ -use crate::aggregation::aggregator::Aggregator; -use crate::calculate_err_field; -use crate::errors::PipelineError; -use dozer_sql_expression::aggregate::AggregateFunctionType::Sum; -use dozer_sql_expression::num_traits::FromPrimitive; -use dozer_types::ordered_float::OrderedFloat; -use dozer_types::rust_decimal::Decimal; - -use dozer_types::types::{DozerDuration, Field, FieldType, TimeUnit}; - -#[derive(Debug, bincode::Encode, bincode::Decode)] -pub struct SumAggregator { - current_state: SumState, - return_type: Option, -} - -#[derive(Debug, bincode::Encode, bincode::Decode)] -pub struct SumState { - pub(crate) int_state: i64, - pub(crate) int8_state: i8, - pub(crate) i128_state: i128, - pub(crate) uint_state: u64, - pub(crate) u128_state: u128, - pub(crate) float_state: f64, - #[bincode(with_serde)] - pub(crate) decimal_state: Decimal, - pub(crate) duration_state: std::time::Duration, -} - -impl SumAggregator { - pub fn new() -> Self { - Self { - current_state: SumState { - int_state: 0_i64, - i128_state: 0_i128, - int8_state: 0_i8, - uint_state: 0_u64, - u128_state: 0_u128, - float_state: 0_f64, - decimal_state: Decimal::from_f64(0_f64).unwrap(), - duration_state: std::time::Duration::new(0, 0), - }, - return_type: None, - } - } -} - -impl Aggregator for SumAggregator { - fn init(&mut self, return_type: FieldType) { - self.return_type = Some(return_type); - } - - fn update(&mut self, old: &[Field], new: &[Field]) -> Result { - self.delete(old)?; - self.insert(new) - } - - fn delete(&mut self, old: &[Field]) -> Result { - get_sum(old, &mut self.current_state, self.return_type, true) - } - - fn insert(&mut self, new: &[Field]) -> Result { - get_sum(new, &mut self.current_state, self.return_type, false) - } -} - -pub fn get_sum( - fields: &[Field], - current_state: &mut SumState, - return_type: Option, - decr: bool, -) -> Result { - match return_type { - Some(typ) => match typ { - FieldType::UInt => { - if decr { - for field in fields { - let val = calculate_err_field!(field.to_uint(), Sum, field); - current_state.uint_state -= val; - } - } else { - for field in fields { - let val = calculate_err_field!(field.to_uint(), Sum, field); - current_state.uint_state += val; - } - } - Ok(Field::UInt(current_state.uint_state)) - } - FieldType::U128 => { - if decr { - for field in fields { - let val = calculate_err_field!(field.to_u128(), Sum, field); - current_state.u128_state -= val; - } - } else { - for field in fields { - let val = calculate_err_field!(field.to_u128(), Sum, field); - current_state.u128_state += val; - } - } - Ok(Field::U128(current_state.u128_state)) - } - FieldType::Int => { - if decr { - for field in fields { - let val = calculate_err_field!(field.to_int(), Sum, field); - current_state.int_state -= val; - } - } else { - for field in fields { - let val = calculate_err_field!(field.to_int(), Sum, field); - current_state.int_state += val; - } - } - Ok(Field::Int(current_state.int_state)) - } - FieldType::Int8 => { - if decr { - for field in fields { - let val = calculate_err_field!(field.to_int8(), Sum, field); - current_state.int8_state -= val; - } - } else { - for field in fields { - let val = calculate_err_field!(field.to_int8(), Sum, field); - current_state.int8_state += val; - } - } - Ok(Field::Int(current_state.int_state)) - } - FieldType::I128 => { - if decr { - for field in fields { - let val = calculate_err_field!(field.to_i128(), Sum, field); - current_state.i128_state -= val; - } - } else { - for field in fields { - let val = calculate_err_field!(field.to_i128(), Sum, field); - current_state.i128_state += val; - } - } - Ok(Field::I128(current_state.i128_state)) - } - FieldType::Float => { - if decr { - for field in fields { - let val = calculate_err_field!(field.to_float(), Sum, field); - current_state.float_state -= val; - } - } else { - for field in fields { - let val = calculate_err_field!(field.to_float(), Sum, field); - current_state.float_state += val; - } - } - Ok(Field::Float(OrderedFloat::from(current_state.float_state))) - } - FieldType::Decimal => { - if decr { - for field in fields { - let val = calculate_err_field!(field.to_decimal(), Sum, field); - current_state.decimal_state -= val; - } - } else { - for field in fields { - let val = calculate_err_field!(field.to_decimal(), Sum, field); - current_state.decimal_state += val; - } - } - Ok(Field::Decimal(current_state.decimal_state)) - } - FieldType::Duration => { - if decr { - for field in fields { - let val = calculate_err_field!(field.to_duration(), Sum, field); - current_state.duration_state -= val.0; - } - } else { - for field in fields { - let val = calculate_err_field!(field.to_duration(), Sum, field); - current_state.duration_state += val.0; - } - } - Ok(Field::Duration(DozerDuration( - current_state.duration_state, - TimeUnit::Nanoseconds, - ))) - } - FieldType::Boolean - | FieldType::String - | FieldType::Text - | FieldType::Date - | FieldType::Timestamp - | FieldType::Binary - | FieldType::Json - | FieldType::Point => Err(PipelineError::InvalidReturnType(format!( - "Not supported return type {typ} for {Sum}" - ))), - }, - None => Err(PipelineError::InvalidReturnType(format!( - "Not supported None return type for {Sum}" - ))), - } -} diff --git a/dozer-sql/src/aggregation/tests/aggregation_avg_tests.rs b/dozer-sql/src/aggregation/tests/aggregation_avg_tests.rs deleted file mode 100644 index 2bfad81230..0000000000 --- a/dozer-sql/src/aggregation/tests/aggregation_avg_tests.rs +++ /dev/null @@ -1,1023 +0,0 @@ -use crate::aggregation::tests::aggregation_tests_utils::{ - delete_exp, delete_field, get_decimal_div_field, get_decimal_field, get_duration_div_field, - get_duration_field, init_input_schema, init_processor, insert_exp, insert_field, update_exp, - update_field, FIELD_0_FLOAT, FIELD_100_FLOAT, FIELD_100_INT, FIELD_100_UINT, FIELD_200_FLOAT, - FIELD_200_INT, FIELD_200_UINT, FIELD_250_DIV_3_FLOAT, FIELD_350_DIV_3_FLOAT, FIELD_50_FLOAT, - FIELD_50_INT, FIELD_50_UINT, FIELD_75_FLOAT, FIELD_NULL, ITALY, SINGAPORE, -}; -use crate::output; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_types::types::FieldType::{Decimal, Duration, Float, Int, UInt}; -use std::collections::HashMap; - -#[test] -fn test_avg_aggregation_float() { - let schema = init_input_schema(Float, "AVG"); - let mut processor = init_processor( - "SELECT Country, AVG(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - AVG = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_FLOAT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - AVG = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_FLOAT, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - AVG = 100.0 - - Singapore, 50.0 - ------------- - AVG = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_50_FLOAT)]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - AVG = 83.333 - */ - inp = update_field(SINGAPORE, ITALY, FIELD_50_FLOAT, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, FIELD_50_FLOAT), - update_exp(ITALY, ITALY, FIELD_100_FLOAT, FIELD_250_DIV_3_FLOAT), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - AVG = 116.667 - */ - inp = update_field(ITALY, ITALY, FIELD_100_FLOAT, FIELD_200_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - FIELD_250_DIV_3_FLOAT, - FIELD_350_DIV_3_FLOAT, - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - AVG = 75.0 - */ - inp = delete_field(ITALY, FIELD_200_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - FIELD_350_DIV_3_FLOAT, - FIELD_75_FLOAT, - )]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - AVG = 100.0 - */ - inp = delete_field(ITALY, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_75_FLOAT, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - AVG = 0.0 - */ - inp = delete_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_100_FLOAT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_avg_aggregation_int() { - let schema = init_input_schema(Int, "AVG"); - let mut processor = init_processor( - "SELECT Country, AVG(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - AVG = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_INT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_decimal_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - AVG = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - AVG = 100.0 - - Singapore, 50.0 - ------------- - AVG = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_decimal_field(50))]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - AVG = 83.333 - */ - inp = update_field(SINGAPORE, ITALY, FIELD_50_INT, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, &get_decimal_field(50)), - update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_div_field(250, 3), - ), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - AVG = 116.667 - */ - inp = update_field(ITALY, ITALY, FIELD_100_INT, FIELD_200_INT); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_div_field(250, 3), - &get_decimal_div_field(350, 3), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - AVG = 75.0 - */ - inp = delete_field(ITALY, FIELD_200_INT); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_div_field(350, 3), - &get_decimal_field(75), - )]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - AVG = 100.0 - */ - inp = delete_field(ITALY, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(75), - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - AVG = 0.0 - */ - inp = delete_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_decimal_field(100))]; - assert_eq!(out, exp); -} - -#[test] -fn test_avg_aggregation_uint() { - let schema = init_input_schema(UInt, "AVG"); - let mut processor = init_processor( - "SELECT Country, AVG(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - AVG = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_UINT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_decimal_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - AVG = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_UINT); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - AVG = 100.0 - - Singapore, 50.0 - ------------- - AVG = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_UINT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_decimal_field(50))]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - AVG = 83.333 - */ - inp = update_field(SINGAPORE, ITALY, FIELD_50_UINT, FIELD_50_UINT); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, &get_decimal_field(50)), - update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_div_field(250, 3), - ), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - AVG = 116.667 - */ - inp = update_field(ITALY, ITALY, FIELD_100_UINT, FIELD_200_UINT); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_div_field(250, 3), - &get_decimal_div_field(350, 3), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - AVG = 75.0 - */ - inp = delete_field(ITALY, FIELD_200_UINT); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_div_field(350, 3), - &get_decimal_field(75), - )]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - AVG = 100.0 - */ - inp = delete_field(ITALY, FIELD_50_UINT); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(75), - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - AVG = 0.0 - */ - inp = delete_field(ITALY, FIELD_100_UINT); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_decimal_field(100))]; - assert_eq!(out, exp); -} - -#[test] -fn test_avg_aggregation_decimal() { - let schema = init_input_schema(Decimal, "AVG"); - let mut processor = init_processor( - "SELECT Country, AVG(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - AVG = 100.0 - */ - let mut inp = insert_field(ITALY, &get_decimal_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_decimal_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - AVG = 100.0 - */ - inp = insert_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - AVG = 100.0 - - Singapore, 50.0 - ------------- - AVG = 50.0 - */ - inp = insert_field(SINGAPORE, &get_decimal_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_decimal_field(50))]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - AVG = 83.333 - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_decimal_field(50), - &get_decimal_field(50), - ); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, &get_decimal_field(50)), - update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_div_field(250, 3), - ), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - AVG = 116.667 - */ - inp = update_field( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(200), - ); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_div_field(250, 3), - &get_decimal_div_field(350, 3), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - AVG = 75.0 - */ - inp = delete_field(ITALY, &get_decimal_field(200)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_div_field(350, 3), - &get_decimal_field(75), - )]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - AVG = 100.0 - */ - inp = delete_field(ITALY, &get_decimal_field(50)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(75), - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - AVG = 0.0 - */ - inp = delete_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_decimal_field(100))]; - assert_eq!(out, exp); -} - -#[test] -fn test_avg_aggregation_duration() { - let schema = init_input_schema(Duration, "AVG"); - let mut processor = init_processor( - "SELECT Country, AVG(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - AVG = 100.0 - */ - let mut inp = insert_field(ITALY, &get_duration_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_duration_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - AVG = 100.0 - */ - inp = insert_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_field(100), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - AVG = 100.0 - - Singapore, 50.0 - ------------- - AVG = 50.0 - */ - inp = insert_field(SINGAPORE, &get_duration_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_duration_field(50))]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - AVG = 83.333 - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_duration_field(50), - &get_duration_field(50), - ); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, &get_duration_field(50)), - update_exp( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_div_field(250, 3), - ), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - AVG = 116.667 - */ - inp = update_field( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_field(200), - ); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_div_field(250, 3), - &get_duration_div_field(350, 3), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - AVG = 75.0 - */ - inp = delete_field(ITALY, &get_duration_field(200)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_div_field(350, 3), - &get_duration_field(75), - )]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - AVG = 100.0 - */ - inp = delete_field(ITALY, &get_duration_field(50)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(75), - &get_duration_field(100), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - AVG = 0.0 - */ - inp = delete_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_duration_field(100))]; - assert_eq!(out, exp); -} - -#[test] -fn test_avg_aggregation_int_null() { - let schema = init_input_schema(Int, "AVG"); - let mut processor = init_processor( - "SELECT Country, AVG(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - AVG = 0 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_decimal_field(0))]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - AVG = 100 - */ - inp = insert_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(0), - &get_decimal_field(50), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - AVG = 0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_INT, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(50), - &get_decimal_field(0), - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - AVG = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(0), - &get_decimal_field(0), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - AVG = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_decimal_field(0))]; - assert_eq!(out, exp); -} - -#[test] -fn test_avg_aggregation_float_null() { - let schema = init_input_schema(Float, "AVG"); - let mut processor = init_processor( - "SELECT Country, AVG(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - AVG = 0 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_0_FLOAT)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - AVG = 50 - */ - inp = insert_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_0_FLOAT, FIELD_50_FLOAT)]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - AVG = 0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_FLOAT, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_50_FLOAT, FIELD_0_FLOAT)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - AVG = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_0_FLOAT, FIELD_0_FLOAT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - AVG = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_0_FLOAT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_avg_aggregation_decimal_null() { - let schema = init_input_schema(Decimal, "AVG"); - let mut processor = init_processor( - "SELECT Country, AVG(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - AVG = 0 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_decimal_field(0))]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - AVG = 50 - */ - inp = insert_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(0), - &get_decimal_field(50), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - AVG = 0 - */ - inp = update_field(ITALY, ITALY, &get_decimal_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(50), - &get_decimal_field(0), - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - AVG = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(0), - &get_decimal_field(0), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - AVG = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_decimal_field(0))]; - assert_eq!(out, exp); -} - -#[test] -fn test_avg_aggregation_duration_null() { - let schema = init_input_schema(Duration, "AVG"); - let mut processor = init_processor( - "SELECT Country, AVG(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - AVG = 0 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_duration_field(0))]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - AVG = 50 - */ - inp = insert_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(0), - &get_duration_field(50), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - AVG = 0 - */ - inp = update_field(ITALY, ITALY, &get_duration_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(50), - &get_duration_field(0), - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - AVG = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(0), - &get_duration_field(0), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - AVG = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_duration_field(0))]; - assert_eq!(out, exp); -} diff --git a/dozer-sql/src/aggregation/tests/aggregation_count_tests.rs b/dozer-sql/src/aggregation/tests/aggregation_count_tests.rs deleted file mode 100644 index 41a57f50e9..0000000000 --- a/dozer-sql/src/aggregation/tests/aggregation_count_tests.rs +++ /dev/null @@ -1,919 +0,0 @@ -use crate::aggregation::tests::aggregation_tests_utils::{ - delete_exp, delete_field, get_date_field, get_decimal_field, get_duration_field, get_ts_field, - init_input_schema, init_processor, insert_exp, insert_field, update_exp, update_field, DATE8, - FIELD_100_FLOAT, FIELD_100_INT, FIELD_1_INT, FIELD_200_FLOAT, FIELD_200_INT, FIELD_2_INT, - FIELD_3_INT, FIELD_50_FLOAT, FIELD_50_INT, FIELD_NULL, ITALY, SINGAPORE, -}; -use crate::output; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_types::types::FieldType::{Date, Decimal, Duration, Float, Int, Timestamp}; -use dozer_types::types::{Operation, Record}; -use std::collections::HashMap; - -#[test] -fn test_count_star() { - let schema = init_input_schema(Float, "COUNT"); - let mut processor = init_processor( - "SELECT COUNT(*) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - COUNT(*) = 2 - */ - output!(processor, insert_field(ITALY, FIELD_100_FLOAT)); - let out = output!(processor, insert_field(ITALY, FIELD_100_FLOAT)); - let old_record = Record::new(vec![FIELD_1_INT.clone()]); - let new_record = Record::new(vec![FIELD_2_INT.clone()]); - - let exp = vec![Operation::Update { - old: old_record, - new: new_record, - }]; - assert_eq!(out, exp); -} - -#[test] -fn test_count_aggregation_float() { - let schema = init_input_schema(Float, "COUNT"); - let mut processor = init_processor( - "SELECT Country, COUNT(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - COUNT = 1 - */ - let mut inp = insert_field(ITALY, FIELD_100_FLOAT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - COUNT = 2 - */ - inp = insert_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_1_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - COUNT = 2 - - Singapore, 50.0 - --------------- - COUNT = 1 - */ - inp = insert_field(SINGAPORE, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - COUNT = 3 - */ - inp = update_field(SINGAPORE, ITALY, FIELD_50_FLOAT, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, FIELD_1_INT), - update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_3_INT), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - COUNT = 3 - */ - inp = update_field(ITALY, ITALY, FIELD_100_FLOAT, FIELD_200_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_3_INT, FIELD_3_INT)]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - COUNT = 2 - */ - inp = delete_field(ITALY, FIELD_200_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_3_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - COUNT = 1 - */ - inp = delete_field(ITALY, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - COUNT = 0 - */ - inp = delete_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_count_aggregation_int() { - let schema = init_input_schema(Int, "COUNT"); - let mut processor = init_processor( - "SELECT Country, COUNT(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - COUNT = 1 - */ - let mut inp = insert_field(ITALY, FIELD_100_INT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - COUNT = 2 - */ - inp = insert_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_1_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - COUNT = 2 - - Singapore, 50.0 - --------------- - COUNT = 1 - */ - inp = insert_field(SINGAPORE, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - COUNT = 3 - */ - inp = update_field(SINGAPORE, ITALY, FIELD_50_INT, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, FIELD_1_INT), - update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_3_INT), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - COUNT = 3 - */ - inp = update_field(ITALY, ITALY, FIELD_100_INT, FIELD_200_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_3_INT, FIELD_3_INT)]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - COUNT = 2 - */ - inp = delete_field(ITALY, FIELD_200_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_3_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - COUNT = 1 - */ - inp = delete_field(ITALY, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - COUNT = 0 - */ - inp = delete_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_count_aggregation_decimal() { - let schema = init_input_schema(Decimal, "COUNT"); - let mut processor = init_processor( - "SELECT Country, COUNT(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - COUNT = 1 - */ - let mut inp = insert_field(ITALY, &get_decimal_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - COUNT = 2 - */ - inp = insert_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_1_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - COUNT = 2 - - Singapore, 50.0 - --------------- - COUNT = 1 - */ - inp = insert_field(SINGAPORE, &get_decimal_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - COUNT = 3 - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_decimal_field(50), - &get_decimal_field(50), - ); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, FIELD_1_INT), - update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_3_INT), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - COUNT = 3 - */ - inp = update_field( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(200), - ); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_3_INT, FIELD_3_INT)]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - COUNT = 2 - */ - inp = delete_field(ITALY, &get_decimal_field(200)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_3_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - COUNT = 1 - */ - inp = delete_field(ITALY, &get_decimal_field(50)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - COUNT = 0 - */ - inp = delete_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_count_aggregation_duration() { - let schema = init_input_schema(Duration, "COUNT"); - let mut processor = init_processor( - "SELECT Country, COUNT(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - COUNT = 1 - */ - let mut inp = insert_field(ITALY, &get_duration_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - COUNT = 2 - */ - inp = insert_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_1_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - COUNT = 2 - - Singapore, 50.0 - --------------- - COUNT = 1 - */ - inp = insert_field(SINGAPORE, &get_duration_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - COUNT = 3 - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_duration_field(50), - &get_duration_field(50), - ); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, FIELD_1_INT), - update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_3_INT), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - COUNT = 3 - */ - inp = update_field( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_field(200), - ); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_3_INT, FIELD_3_INT)]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - COUNT = 2 - */ - inp = delete_field(ITALY, &get_duration_field(200)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_3_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - COUNT = 1 - */ - inp = delete_field(ITALY, &get_duration_field(50)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - COUNT = 0 - */ - inp = delete_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_count_aggregation_int_null() { - let schema = init_input_schema(Int, "COUNT"); - let mut processor = init_processor( - "SELECT Country, COUNT(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - COUNT = 1 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - COUNT = 2 - */ - inp = insert_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_1_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - COUNT = 2 - */ - inp = update_field(ITALY, ITALY, FIELD_100_INT, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - COUNT = 1 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - COUNT = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_count_aggregation_float_null() { - let schema = init_input_schema(Float, "COUNT"); - let mut processor = init_processor( - "SELECT Country, COUNT(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - COUNT = 1 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - COUNT = 2 - */ - inp = insert_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_1_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - COUNT = 2 - */ - inp = update_field(ITALY, ITALY, FIELD_100_FLOAT, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - COUNT = 1 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - COUNT = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_count_aggregation_decimal_null() { - let schema = init_input_schema(Decimal, "COUNT"); - let mut processor = init_processor( - "SELECT Country, COUNT(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - COUNT = 1 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - COUNT = 2 - */ - inp = insert_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_1_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - COUNT = 2 - */ - inp = update_field(ITALY, ITALY, &get_decimal_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - COUNT = 1 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - COUNT = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_count_aggregation_timestamp_null() { - let schema = init_input_schema(Timestamp, "COUNT"); - let mut processor = init_processor( - "SELECT Country, COUNT(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - COUNT = 1 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - COUNT = 2 - */ - inp = insert_field(ITALY, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_1_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - COUNT = 2 - */ - inp = update_field(ITALY, ITALY, &get_ts_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - COUNT = 1 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - COUNT = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_count_aggregation_date_null() { - let schema = init_input_schema(Date, "COUNT"); - let mut processor = init_processor( - "SELECT Country, COUNT(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - COUNT = 1 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - COUNT = 2 - */ - inp = insert_field(ITALY, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_1_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - COUNT = 2 - */ - inp = update_field(ITALY, ITALY, &get_date_field(DATE8), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - COUNT = 1 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - COUNT = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_count_aggregation_duration_null() { - let schema = init_input_schema(Duration, "COUNT"); - let mut processor = init_processor( - "SELECT Country, COUNT(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - COUNT = 1 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - COUNT = 2 - */ - inp = insert_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_1_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - COUNT = 2 - */ - inp = update_field(ITALY, ITALY, &get_duration_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_2_INT)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - COUNT = 1 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_2_INT, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - COUNT = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); -} diff --git a/dozer-sql/src/aggregation/tests/aggregation_having_tests.rs b/dozer-sql/src/aggregation/tests/aggregation_having_tests.rs deleted file mode 100644 index 20d5ea801c..0000000000 --- a/dozer-sql/src/aggregation/tests/aggregation_having_tests.rs +++ /dev/null @@ -1,232 +0,0 @@ -use crate::aggregation::tests::aggregation_tests_utils::{ - delete_exp, delete_field, init_input_schema, init_processor, insert_exp, insert_field, - update_exp, update_field, FIELD_100_INT, FIELD_150_INT, FIELD_200_INT, FIELD_300_INT, - FIELD_400_INT, FIELD_500_INT, FIELD_50_INT, FIELD_600_INT, ITALY, SINGAPORE, -}; -use crate::output; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_types::types::FieldType::Int; -use std::collections::HashMap; - -#[test] -fn test_having_insert_delete_ops() { - let schema = init_input_schema(Int, "COUNT"); - let mut processor = init_processor( - "SELECT Country, SUM(Salary) \ - FROM Users \ - HAVING SUM(Salary) > 100 AND SUM(Salary) < 400", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Testing insert - - // 100 -> Nothing - let inp = insert_field(ITALY, FIELD_100_INT); - let out = output!(processor, inp); - let exp = vec![]; - assert_eq!(out, exp); - - // 200 -> Insert - let inp = insert_field(ITALY, FIELD_100_INT); - let out = output!(processor, inp); - let exp = vec![insert_exp(ITALY, FIELD_200_INT)]; - assert_eq!(out, exp); - - // 300 -> Update - let inp = insert_field(ITALY, FIELD_100_INT); - let out = output!(processor, inp); - let exp = vec![update_exp(ITALY, ITALY, FIELD_200_INT, FIELD_300_INT)]; - assert_eq!(out, exp); - - // 400 -> Delete - let inp = insert_field(ITALY, FIELD_100_INT); - let out = output!(processor, inp); - let exp = vec![delete_exp(ITALY, FIELD_300_INT)]; - assert_eq!(out, exp); - - // 500 -> Nothing - let inp = insert_field(ITALY, FIELD_100_INT); - let out = output!(processor, inp); - let exp = vec![]; - assert_eq!(out, exp); - - // Testing Delete - - // 400 -> Nothing - let inp = delete_field(ITALY, FIELD_100_INT); - let out = output!(processor, inp); - let exp = vec![]; - assert_eq!(out, exp); - - // 300 -> insert - let inp = delete_field(ITALY, FIELD_100_INT); - let out = output!(processor, inp); - let exp = vec![insert_exp(ITALY, FIELD_300_INT)]; - assert_eq!(out, exp); - - // 200 -> update - let inp = delete_field(ITALY, FIELD_100_INT); - let out = output!(processor, inp); - let exp = vec![update_exp(ITALY, ITALY, FIELD_300_INT, FIELD_200_INT)]; - assert_eq!(out, exp); - - // 100 -> delete - let inp = delete_field(ITALY, FIELD_100_INT); - let out = output!(processor, inp); - let exp = vec![delete_exp(ITALY, FIELD_200_INT)]; - assert_eq!(out, exp); - - // 0 -> delete - let inp = delete_field(ITALY, FIELD_100_INT); - let out = output!(processor, inp); - let exp = vec![]; - assert_eq!(out, exp); -} - -#[test] -fn test_having_update_ops() { - let schema = init_input_schema(Int, "COUNT"); - let mut processor = init_processor( - "SELECT Country, SUM(Salary) \ - FROM Users \ - HAVING SUM(Salary) > 300 AND SUM(Salary) < 600", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - let inp = insert_field(ITALY, FIELD_100_INT); - let _out = output!(processor, inp); - let inp = insert_field(ITALY, FIELD_100_INT); - let _out = output!(processor, inp); - - // 300 -> Nothing - let inp = update_field(ITALY, ITALY, FIELD_100_INT, FIELD_200_INT); - let out = output!(processor, inp); - let exp = vec![]; - assert_eq!(out, exp); - - // 400 -> Insert - let inp = update_field(ITALY, ITALY, FIELD_200_INT, FIELD_300_INT); - let out = output!(processor, inp); - let exp = vec![insert_exp(ITALY, FIELD_400_INT)]; - assert_eq!(out, exp); - - // 500 -> Insert - let inp = update_field(ITALY, ITALY, FIELD_300_INT, FIELD_400_INT); - let out = output!(processor, inp); - let exp = vec![update_exp(ITALY, ITALY, FIELD_400_INT, FIELD_500_INT)]; - assert_eq!(out, exp); - - // 600 -> Delete - let inp = update_field(ITALY, ITALY, FIELD_400_INT, FIELD_500_INT); - let out = output!(processor, inp); - let exp = vec![delete_exp(ITALY, FIELD_500_INT)]; - assert_eq!(out, exp); - - // 700 -> Nothing - let inp = update_field(ITALY, ITALY, FIELD_500_INT, FIELD_600_INT); - let out = output!(processor, inp); - let exp = vec![]; - assert_eq!(out, exp); - - // 600 -> Nothing - let inp = update_field(ITALY, ITALY, FIELD_600_INT, FIELD_500_INT); - let out = output!(processor, inp); - let exp = vec![]; - assert_eq!(out, exp); - - // 500 -> insert - let inp = update_field(ITALY, ITALY, FIELD_500_INT, FIELD_400_INT); - let out = output!(processor, inp); - let exp = vec![insert_exp(ITALY, FIELD_500_INT)]; - assert_eq!(out, exp); - - // 400 -> update - let inp = update_field(ITALY, ITALY, FIELD_400_INT, FIELD_300_INT); - let out = output!(processor, inp); - let exp = vec![update_exp(ITALY, ITALY, FIELD_500_INT, FIELD_400_INT)]; - assert_eq!(out, exp); - - // 300 -> Delete - let inp = update_field(ITALY, ITALY, FIELD_300_INT, FIELD_200_INT); - let out = output!(processor, inp); - let exp = vec![delete_exp(ITALY, FIELD_400_INT)]; - assert_eq!(out, exp); - - // 200 -> Delete - let inp = update_field(ITALY, ITALY, FIELD_200_INT, FIELD_100_INT); - let out = output!(processor, inp); - let exp = vec![]; - assert_eq!(out, exp); -} - -#[test] -fn test_having_update_multi_segment_insert_op() { - let schema = init_input_schema(Int, "COUNT"); - let mut processor = init_processor( - "SELECT Country, SUM(Salary) \ - FROM Users GROUP BY Country \ - HAVING SUM(Salary) > 100", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - let inp = insert_field(ITALY, FIELD_100_INT); - let _out = output!(processor, inp); - let inp = insert_field(SINGAPORE, FIELD_100_INT); - let _out = output!(processor, inp); - - // 700 -> Nothing - let inp = update_field(SINGAPORE, ITALY, FIELD_100_INT, FIELD_100_INT); - let out = output!(processor, inp); - let exp = vec![insert_exp(ITALY, FIELD_200_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_having_update_multi_segment_insert_delete_op() { - let schema = init_input_schema(Int, "COUNT"); - let mut processor = init_processor( - "SELECT Country, SUM(Salary) \ - FROM Users GROUP BY Country \ - HAVING SUM(Salary) > 100", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - let inp = insert_field(ITALY, FIELD_100_INT); - let _out = output!(processor, inp); - let inp = insert_field(SINGAPORE, FIELD_200_INT); - let _out = output!(processor, inp); - - let inp = update_field(SINGAPORE, ITALY, FIELD_200_INT, FIELD_200_INT); - let out = output!(processor, inp); - let exp = vec![ - delete_exp(SINGAPORE, FIELD_200_INT), - insert_exp(ITALY, FIELD_300_INT), - ]; - assert_eq!(out, exp); -} - -#[test] -fn test_having_update_multi_segment_delete_op() { - let schema = init_input_schema(Int, "COUNT"); - let mut processor = init_processor( - "SELECT Country, SUM(Salary) \ - FROM Users GROUP BY Country \ - HAVING SUM(Salary) > 100", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - let inp = insert_field(SINGAPORE, FIELD_50_INT); - let _out = output!(processor, inp); - let inp = insert_field(SINGAPORE, FIELD_100_INT); - let _out = output!(processor, inp); - - let inp = update_field(SINGAPORE, ITALY, FIELD_50_INT, FIELD_50_INT); - let out = output!(processor, inp); - let exp = vec![delete_exp(SINGAPORE, FIELD_150_INT)]; - assert_eq!(out, exp); -} diff --git a/dozer-sql/src/aggregation/tests/aggregation_max_append_only_tests.rs b/dozer-sql/src/aggregation/tests/aggregation_max_append_only_tests.rs deleted file mode 100644 index d26fbf2004..0000000000 --- a/dozer-sql/src/aggregation/tests/aggregation_max_append_only_tests.rs +++ /dev/null @@ -1,608 +0,0 @@ -use crate::aggregation::tests::aggregation_tests_utils::{ - get_date_field, get_decimal_field, get_duration_field, get_ts_field, init_input_schema, - init_processor, insert_exp, insert_field, update_exp, DATE4, DATE8, FIELD_100_FLOAT, - FIELD_100_INT, FIELD_100_UINT, FIELD_50_FLOAT, FIELD_50_INT, FIELD_50_UINT, FIELD_NULL, ITALY, - SINGAPORE, -}; -use crate::output; -use dozer_core::DEFAULT_PORT_HANDLE; - -use dozer_types::types::FieldType::{Date, Decimal, Duration, Float, Int, Timestamp, UInt}; -use std::collections::HashMap; - -#[test] -fn test_max_aggregation_float() { - let schema = init_input_schema(Float, "MAX_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MAX_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_FLOAT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_FLOAT, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - - Singapore, 50.0 - --------------- - MAX_APPEND_ONLY = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_50_FLOAT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_int() { - let schema = init_input_schema(Int, "MAX_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MAX_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_INT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_100_INT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_INT, FIELD_100_INT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - - Singapore, 50.0 - --------------- - MAX_APPEND_ONLY = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_50_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_uint() { - let schema = init_input_schema(UInt, "MAX_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MAX_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_UINT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_100_UINT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_UINT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_UINT, FIELD_100_UINT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - - Singapore, 50.0 - --------------- - MAX_APPEND_ONLY = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_UINT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_50_UINT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_decimal() { - let schema = init_input_schema(Decimal, "MAX_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MAX_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - */ - let mut inp = insert_field(ITALY, &get_decimal_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_decimal_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - */ - inp = insert_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - - Singapore, 50.0 - ------------- - MAX_APPEND_ONLY = 50.0 - */ - inp = insert_field(SINGAPORE, &get_decimal_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_decimal_field(50))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_duration() { - let schema = init_input_schema(Duration, "MAX_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MAX_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - */ - let mut inp = insert_field(ITALY, &get_duration_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_duration_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - */ - inp = insert_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_field(100), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - - Singapore, 50.0 - ------------- - MAX_APPEND_ONLY = 50.0 - */ - inp = insert_field(SINGAPORE, &get_duration_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_duration_field(50))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_timestamp() { - let schema = init_input_schema(Timestamp, "MAX_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MAX_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100 - ------------- - MAX_APPEND_ONLY = 100 - */ - let mut inp = insert_field(ITALY, &get_ts_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_ts_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100 - Italy, 100 - ------------- - MAX_APPEND_ONLY = 100 - */ - inp = insert_field(ITALY, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_ts_field(100), - &get_ts_field(100), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100 - Italy, 100 - ------------- - MAX_APPEND_ONLY = 100 - - Singapore, 50 - ------------- - MAX_APPEND_ONLY = 50 - */ - inp = insert_field(SINGAPORE, &get_ts_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_ts_field(50))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_date() { - let schema = init_input_schema(Date, "MAX_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MAX_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 2015-10-08 for segment Italy - /* - Italy, 2015-10-08 - ------------------ - MAX_APPEND_ONLY = 2015-10-08 - */ - let mut inp = insert_field(ITALY, &get_date_field(DATE8)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_date_field(DATE8))]; - assert_eq!(out, exp); - - // Insert another 2015-10-08 for segment Italy - /* - Italy, 2015-10-08 - Italy, 2015-10-08 - ----------------- - MAX_APPEND_ONLY = 2015-10-08 - */ - inp = insert_field(ITALY, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_date_field(DATE8), - &get_date_field(DATE8), - )]; - assert_eq!(out, exp); - - // Insert 2015-10-04 for segment Singapore - /* - Italy, 2015-10-08 - Italy, 2015-10-08 - ------------- - MAX_APPEND_ONLY = 2015-10-08 - - Singapore, 2015-10-04 - ------------- - MAX_APPEND_ONLY = 2015-10-04 - */ - inp = insert_field(SINGAPORE, &get_date_field(DATE4)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_date_field(DATE4))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_int_null() { - let schema = init_input_schema(Int, "MAX_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MAX_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX_APPEND_ONLY = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - MAX_APPEND_ONLY = 100 - */ - inp = insert_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_100_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_float_null() { - let schema = init_input_schema(Float, "MAX_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MAX_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX_APPEND_ONLY = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_100_FLOAT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_decimal_null() { - let schema = init_input_schema(Decimal, "MAX_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MAX_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX_APPEND_ONLY = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - */ - inp = insert_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - FIELD_NULL, - &get_decimal_field(100), - )]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_duration_null() { - let schema = init_input_schema(Duration, "MAX_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MAX_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX_APPEND_ONLY = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MAX_APPEND_ONLY = 100.0 - */ - inp = insert_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - FIELD_NULL, - &get_duration_field(100), - )]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_timestamp_null() { - let schema = init_input_schema(Timestamp, "MAX_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MAX_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX_APPEND_ONLY = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - MAX_APPEND_ONLY = 100 - */ - inp = insert_field(ITALY, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, &get_ts_field(100))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_date_null() { - let schema = init_input_schema(Date, "MAX_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MAX_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX_APPEND_ONLY = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 2015-10-08 for segment Italy - /* - Italy, NULL - Italy, 2015-10-08 - ------------- - MAX_APPEND_ONLY = 2015-10-08 - */ - inp = insert_field(ITALY, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, &get_date_field(DATE8))]; - assert_eq!(out, exp); -} diff --git a/dozer-sql/src/aggregation/tests/aggregation_max_tests.rs b/dozer-sql/src/aggregation/tests/aggregation_max_tests.rs deleted file mode 100644 index daccb788e3..0000000000 --- a/dozer-sql/src/aggregation/tests/aggregation_max_tests.rs +++ /dev/null @@ -1,1350 +0,0 @@ -use crate::aggregation::tests::aggregation_tests_utils::{ - delete_exp, delete_field, get_date_field, get_decimal_field, get_duration_field, get_ts_field, - init_input_schema, init_processor, insert_exp, insert_field, update_exp, update_field, DATE16, - DATE4, DATE8, FIELD_100_FLOAT, FIELD_100_INT, FIELD_100_UINT, FIELD_200_FLOAT, FIELD_200_INT, - FIELD_200_UINT, FIELD_50_FLOAT, FIELD_50_INT, FIELD_50_UINT, FIELD_NULL, ITALY, SINGAPORE, -}; -use crate::output; -use dozer_core::DEFAULT_PORT_HANDLE; - -use dozer_types::types::FieldType::{Date, Decimal, Duration, Float, Int, Timestamp, UInt}; -use std::collections::HashMap; - -#[test] -fn test_max_aggregation_float() { - let schema = init_input_schema(Float, "MAX"); - let mut processor = init_processor( - "SELECT Country, MAX(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MAX = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_FLOAT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_FLOAT, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX = 100.0 - - Singapore, 50.0 - --------------- - MAX = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_50_FLOAT)]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MAX = 100.0 - */ - inp = update_field(SINGAPORE, ITALY, FIELD_50_FLOAT, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, FIELD_50_FLOAT), - update_exp(ITALY, ITALY, FIELD_100_FLOAT, FIELD_100_FLOAT), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MAX = 200.0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_FLOAT, FIELD_200_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_FLOAT, FIELD_200_FLOAT)]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - MAX = 100.0 - */ - inp = delete_field(ITALY, FIELD_200_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_200_FLOAT, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - MAX = 100.0 - */ - inp = delete_field(ITALY, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_FLOAT, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX = Null - */ - inp = delete_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_100_FLOAT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_int() { - let schema = init_input_schema(Int, "MAX"); - let mut processor = init_processor( - "SELECT Country, MAX(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MAX = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_INT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_100_INT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_INT, FIELD_100_INT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX = 100.0 - - Singapore, 50.0 - --------------- - MAX = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_50_INT)]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MAX = 100.0 - */ - inp = update_field(SINGAPORE, ITALY, FIELD_50_INT, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, FIELD_50_INT), - update_exp(ITALY, ITALY, FIELD_100_INT, FIELD_100_INT), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MAX = 200.0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_INT, FIELD_200_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_INT, FIELD_200_INT)]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - MAX = 50.0 - */ - inp = delete_field(ITALY, FIELD_200_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_200_INT, FIELD_100_INT)]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - MAX = 100.0 - */ - inp = delete_field(ITALY, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_INT, FIELD_100_INT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX = Null - */ - inp = delete_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_100_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_uint() { - let schema = init_input_schema(UInt, "MAX"); - let mut processor = init_processor( - "SELECT Country, MAX(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MAX = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_UINT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_100_UINT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_UINT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_UINT, FIELD_100_UINT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX = 100.0 - - Singapore, 50.0 - --------------- - MAX = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_UINT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_50_UINT)]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MAX = 100.0 - */ - inp = update_field(SINGAPORE, ITALY, FIELD_50_UINT, FIELD_50_UINT); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, FIELD_50_UINT), - update_exp(ITALY, ITALY, FIELD_100_UINT, FIELD_100_UINT), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MAX = 200.0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_UINT, FIELD_200_UINT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_UINT, FIELD_200_UINT)]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - MAX = 50.0 - */ - inp = delete_field(ITALY, FIELD_200_UINT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_200_UINT, FIELD_100_UINT)]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - MAX = 100.0 - */ - inp = delete_field(ITALY, FIELD_50_UINT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_UINT, FIELD_100_UINT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX = Null - */ - inp = delete_field(ITALY, FIELD_100_UINT); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_100_UINT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_decimal() { - let schema = init_input_schema(Decimal, "MAX"); - let mut processor = init_processor( - "SELECT Country, MAX(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MAX = 100.0 - */ - let mut inp = insert_field(ITALY, &get_decimal_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_decimal_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX = 100.0 - */ - inp = insert_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX = 100.0 - - Singapore, 50.0 - ------------- - MAX = 50.0 - */ - inp = insert_field(SINGAPORE, &get_decimal_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_decimal_field(50))]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MAX = 100.0 - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_decimal_field(50), - &get_decimal_field(50), - ); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, &get_decimal_field(50)), - update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(100), - ), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MAX = 200.0 - */ - inp = update_field( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(200), - ); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(200), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - MAX = 100.0 - */ - inp = delete_field(ITALY, &get_decimal_field(200)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(200), - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - MAX = 100.0 - */ - inp = delete_field(ITALY, &get_decimal_field(50)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX = 0.0 - */ - inp = delete_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_decimal_field(100))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_duration() { - let schema = init_input_schema(Duration, "MAX"); - let mut processor = init_processor( - "SELECT Country, MAX(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MAX = 100.0 - */ - let mut inp = insert_field(ITALY, &get_duration_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_duration_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX = 100.0 - */ - inp = insert_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_field(100), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MAX = 100.0 - - Singapore, 50.0 - ------------- - MAX = 50.0 - */ - inp = insert_field(SINGAPORE, &get_duration_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_duration_field(50))]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MAX = 100.0 - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_duration_field(50), - &get_duration_field(50), - ); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, &get_duration_field(50)), - update_exp( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_field(100), - ), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MAX = 200.0 - */ - inp = update_field( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_field(200), - ); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_field(200), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - MAX = 100.0 - */ - inp = delete_field(ITALY, &get_duration_field(200)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(200), - &get_duration_field(100), - )]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - MAX = 100.0 - */ - inp = delete_field(ITALY, &get_duration_field(50)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_field(100), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX = 0.0 - */ - inp = delete_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_duration_field(100))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_timestamp() { - let schema = init_input_schema(Timestamp, "MAX"); - let mut processor = init_processor( - "SELECT Country, MAX(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100 - ------------- - MAX = 100 - */ - let mut inp = insert_field(ITALY, &get_ts_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_ts_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100 - Italy, 100 - ------------- - MAX = 100 - */ - inp = insert_field(ITALY, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_ts_field(100), - &get_ts_field(100), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100 - Italy, 100 - ------------- - MAX = 100 - - Singapore, 50 - ------------- - MAX = 50 - */ - inp = insert_field(SINGAPORE, &get_ts_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_ts_field(50))]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100 - Italy, 100 - Italy, 50 - ------------- - MAX = 100 - */ - inp = update_field(SINGAPORE, ITALY, &get_ts_field(50), &get_ts_field(50)); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, &get_ts_field(50)), - update_exp(ITALY, ITALY, &get_ts_field(100), &get_ts_field(100)), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200 - Italy, 100 - Italy, 50 - ------------- - MAX = 200 - */ - inp = update_field(ITALY, ITALY, &get_ts_field(100), &get_ts_field(200)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_ts_field(100), - &get_ts_field(200), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100 - Italy, 50 - ------------- - MAX = 100 - */ - inp = delete_field(ITALY, &get_ts_field(200)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_ts_field(200), - &get_ts_field(100), - )]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100 - ------------- - MAX = 100 - */ - inp = delete_field(ITALY, &get_ts_field(50)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_ts_field(100), - &get_ts_field(100), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX = 0 - */ - inp = delete_field(ITALY, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_ts_field(100))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_date() { - let schema = init_input_schema(Date, "MAX"); - let mut processor = init_processor( - "SELECT Country, MAX(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 2015-10-08 for segment Italy - /* - Italy, 2015-10-08 - ------------------ - MAX = 2015-10-08 - */ - let mut inp = insert_field(ITALY, &get_date_field(DATE8)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_date_field(DATE8))]; - assert_eq!(out, exp); - - // Insert another 2015-10-08 for segment Italy - /* - Italy, 2015-10-08 - Italy, 2015-10-08 - ----------------- - MAX = 2015-10-08 - */ - inp = insert_field(ITALY, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_date_field(DATE8), - &get_date_field(DATE8), - )]; - assert_eq!(out, exp); - - // Insert 2015-10-04 for segment Singapore - /* - Italy, 2015-10-08 - Italy, 2015-10-08 - ------------- - MAX = 2015-10-08 - - Singapore, 2015-10-04 - ------------- - MAX = 2015-10-04 - */ - inp = insert_field(SINGAPORE, &get_date_field(DATE4)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_date_field(DATE4))]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 2015-10-08 - Italy, 2015-10-08 - Italy, 2015-10-04 - ------------- - MAX = 2015-10-08 - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_date_field(DATE4), - &get_date_field(DATE4), - ); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, &get_date_field(DATE4)), - update_exp(ITALY, ITALY, &get_date_field(DATE8), &get_date_field(DATE8)), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 2015-10-16 - Italy, 2015-10-08 - Italy, 2015-10-04 - ------------- - MAX = 2015-10-16 - */ - inp = update_field( - ITALY, - ITALY, - &get_date_field(DATE8), - &get_date_field(DATE16), - ); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_date_field(DATE8), - &get_date_field(DATE16), - )]; - assert_eq!(out, exp); - - // Delete 1 record (2015-10-16) - /* - Italy, 2015-10-08 - Italy, 2015-10-04 - ------------- - MAX = 2015-10-08 - */ - inp = delete_field(ITALY, &get_date_field(DATE16)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_date_field(DATE16), - &get_date_field(DATE8), - )]; - assert_eq!(out, exp); - - // Delete another record (2015-10-04) - /* - Italy, 2015-10-08 - ------------- - MAX = 2015-10-08 - */ - inp = delete_field(ITALY, &get_date_field(DATE4)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_date_field(DATE8), - &get_date_field(DATE8), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX = 0 - */ - inp = delete_field(ITALY, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_date_field(DATE8))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_int_null() { - let schema = init_input_schema(Int, "MAX"); - let mut processor = init_processor( - "SELECT Country, MAX(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - MAX = 100 - */ - inp = insert_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_100_INT)]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MAX = 0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_INT, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_INT, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MAX = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_float_null() { - let schema = init_input_schema(Float, "MAX"); - let mut processor = init_processor( - "SELECT Country, MAX(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MAX = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MAX = 0.0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_FLOAT, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_FLOAT, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MAX = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_decimal_null() { - let schema = init_input_schema(Decimal, "MAX"); - let mut processor = init_processor( - "SELECT Country, MAX(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MAX = 100.0 - */ - inp = insert_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - FIELD_NULL, - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MAX = 0.0 - */ - inp = update_field(ITALY, ITALY, &get_decimal_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - FIELD_NULL, - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MAX = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_duration_null() { - let schema = init_input_schema(Duration, "MAX"); - let mut processor = init_processor( - "SELECT Country, MAX(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MAX = 100.0 - */ - inp = insert_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - FIELD_NULL, - &get_duration_field(100), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MAX = 0.0 - */ - inp = update_field(ITALY, ITALY, &get_duration_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(100), - FIELD_NULL, - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MAX = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_timestamp_null() { - let schema = init_input_schema(Timestamp, "MAX"); - let mut processor = init_processor( - "SELECT Country, MAX(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - MAX = 100 - */ - inp = insert_field(ITALY, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, &get_ts_field(100))]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MAX = 0 - */ - inp = update_field(ITALY, ITALY, &get_ts_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, &get_ts_field(100), FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MAX = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_date_null() { - let schema = init_input_schema(Date, "MAX"); - let mut processor = init_processor( - "SELECT Country, MAX(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 2015-10-08 for segment Italy - /* - Italy, NULL - Italy, 2015-10-08 - ------------- - MAX = 2015-10-08 - */ - inp = insert_field(ITALY, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, &get_date_field(DATE8))]; - assert_eq!(out, exp); - - // Update 2015-10-08 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MAX = 0 - */ - inp = update_field(ITALY, ITALY, &get_date_field(DATE8), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, &get_date_field(DATE8), FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MAX = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); -} diff --git a/dozer-sql/src/aggregation/tests/aggregation_max_value_tests.rs b/dozer-sql/src/aggregation/tests/aggregation_max_value_tests.rs deleted file mode 100644 index ea6f900865..0000000000 --- a/dozer-sql/src/aggregation/tests/aggregation_max_value_tests.rs +++ /dev/null @@ -1,1459 +0,0 @@ -use crate::aggregation::tests::aggregation_tests_utils::{ - delete_field, delete_val_exp, get_date_field, get_decimal_field, get_duration_field, - get_ts_field, init_input_schema, init_processor, init_val_input_schema, insert_field, - insert_val_exp, update_field, update_val_exp, DATE16, DATE4, DATE8, FIELD_100_FLOAT, - FIELD_100_INT, FIELD_100_UINT, FIELD_150_FLOAT, FIELD_150_INT, FIELD_150_UINT, FIELD_200_FLOAT, - FIELD_200_INT, FIELD_200_UINT, FIELD_NULL, ITALY, SINGAPORE, -}; -use crate::output; -use dozer_core::DEFAULT_PORT_HANDLE; - -use dozer_types::types::Field; -use dozer_types::types::FieldType::{Date, Decimal, Duration, Float, Int, Timestamp, UInt}; -use std::collections::HashMap; - -#[test] -fn test_max_aggregation_float() { - let schema = init_val_input_schema(Float, "MAX_VALUE"); - let mut processor = init_processor( - "SELECT MAX_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ----------------- - MAX_VALUE = Italy - */ - let mut inp = insert_field(ITALY, FIELD_100_FLOAT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - ----------------- - MAX_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - Singapore, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, FIELD_150_FLOAT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Singapore, 100.0 - Italy, 150.0 - --------------- - MAX_VALUE = Italy - */ - inp = update_field(SINGAPORE, ITALY, FIELD_150_FLOAT, FIELD_150_FLOAT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore value 100 -> 200 - /* - Italy, 100.0 - Singapore, 200.0 - Italy, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = update_field(SINGAPORE, SINGAPORE, FIELD_100_FLOAT, FIELD_200_FLOAT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 150.0 - --------------- - MAX_VALUE = Italy - */ - inp = delete_field(SINGAPORE, FIELD_200_FLOAT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Italy segment to Singapore - /* - Italy, 100.0 - Singapore, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = update_field(ITALY, SINGAPORE, FIELD_150_FLOAT, FIELD_150_FLOAT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete another record (150) - /* - Italy, 100.0 - ------------- - MAX_VALUE = Italy - */ - inp = delete_field(SINGAPORE, FIELD_150_FLOAT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX_VALUE = Null - */ - inp = delete_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![delete_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_int() { - let schema = init_val_input_schema(Int, "MAX_VALUE"); - let mut processor = init_processor( - "SELECT MAX_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ----------------- - MAX_VALUE = Italy - */ - let mut inp = insert_field(ITALY, FIELD_100_INT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - ----------------- - MAX_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - Singapore, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, FIELD_150_INT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Singapore, 100.0 - Italy, 150.0 - --------------- - MAX_VALUE = Italy - */ - inp = update_field(SINGAPORE, ITALY, FIELD_150_INT, FIELD_150_INT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore value 100 -> 200 - /* - Italy, 100.0 - Singapore, 200.0 - Italy, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = update_field(SINGAPORE, SINGAPORE, FIELD_100_INT, FIELD_200_INT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 150.0 - --------------- - MAX_VALUE = Italy - */ - inp = delete_field(SINGAPORE, FIELD_200_INT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Italy segment to Singapore - /* - Italy, 100.0 - Singapore, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = update_field(ITALY, SINGAPORE, FIELD_150_INT, FIELD_150_INT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete another record (150) - /* - Italy, 100.0 - ------------- - MAX_VALUE = Italy - */ - inp = delete_field(SINGAPORE, FIELD_150_INT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX_VALUE = Null - */ - inp = delete_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![delete_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_uint() { - let schema = init_val_input_schema(UInt, "MAX_VALUE"); - let mut processor = init_processor( - "SELECT MAX_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ----------------- - MAX_VALUE = Italy - */ - let mut inp = insert_field(ITALY, FIELD_100_UINT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - ----------------- - MAX_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, FIELD_100_UINT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - Singapore, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, FIELD_150_UINT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Singapore, 100.0 - Italy, 150.0 - --------------- - MAX_VALUE = Italy - */ - inp = update_field(SINGAPORE, ITALY, FIELD_150_UINT, FIELD_150_UINT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore value 100 -> 200 - /* - Italy, 100.0 - Singapore, 200.0 - Italy, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = update_field(SINGAPORE, SINGAPORE, FIELD_100_UINT, FIELD_200_UINT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 150.0 - --------------- - MAX_VALUE = Italy - */ - inp = delete_field(SINGAPORE, FIELD_200_UINT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Italy segment to Singapore - /* - Italy, 100.0 - Singapore, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = update_field(ITALY, SINGAPORE, FIELD_150_UINT, FIELD_150_UINT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete another record (150) - /* - Italy, 100.0 - ------------- - MAX_VALUE = Italy - */ - inp = delete_field(SINGAPORE, FIELD_150_UINT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX_VALUE = Null - */ - inp = delete_field(ITALY, FIELD_100_UINT); - out = output!(processor, inp); - exp = vec![delete_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_decimal() { - let schema = init_val_input_schema(Decimal, "MAX_VALUE"); - let mut processor = init_processor( - "SELECT MAX_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ----------------- - MAX_VALUE = Italy - */ - let mut inp = insert_field(ITALY, &get_decimal_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - ----------------- - MAX_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - Singapore, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, &get_decimal_field(150)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Singapore, 100.0 - Italy, 150.0 - --------------- - MAX_VALUE = Italy - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_decimal_field(150), - &get_decimal_field(150), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore value 100 -> 200 - /* - Italy, 100.0 - Singapore, 200.0 - Italy, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = update_field( - SINGAPORE, - SINGAPORE, - &get_decimal_field(100), - &get_decimal_field(200), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 150.0 - --------------- - MAX_VALUE = Italy - */ - inp = delete_field(SINGAPORE, &get_decimal_field(200)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Italy segment to Singapore - /* - Italy, 100.0 - Singapore, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = update_field( - ITALY, - SINGAPORE, - &get_decimal_field(150), - &get_decimal_field(150), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete another record (150) - /* - Italy, 100.0 - ------------- - MAX_VALUE = Italy - */ - inp = delete_field(SINGAPORE, &get_decimal_field(150)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX_VALUE = Null - */ - inp = delete_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![delete_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_duration() { - let schema = init_val_input_schema(Duration, "MAX_VALUE"); - let mut processor = init_processor( - "SELECT MAX_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ----------------- - MAX_VALUE = Italy - */ - let mut inp = insert_field(ITALY, &get_duration_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - ----------------- - MAX_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - Singapore, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, &get_duration_field(150)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Singapore, 100.0 - Italy, 150.0 - --------------- - MAX_VALUE = Italy - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_duration_field(150), - &get_duration_field(150), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore value 100 -> 200 - /* - Italy, 100.0 - Singapore, 200.0 - Italy, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = update_field( - SINGAPORE, - SINGAPORE, - &get_duration_field(100), - &get_duration_field(200), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 150.0 - --------------- - MAX_VALUE = Italy - */ - inp = delete_field(SINGAPORE, &get_duration_field(200)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Italy segment to Singapore - /* - Italy, 100.0 - Singapore, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = update_field( - ITALY, - SINGAPORE, - &get_duration_field(150), - &get_duration_field(150), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete another record (150) - /* - Italy, 100.0 - ------------- - MAX_VALUE = Italy - */ - inp = delete_field(SINGAPORE, &get_duration_field(150)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX_VALUE = Null - */ - inp = delete_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![delete_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_timestamp() { - let schema = init_val_input_schema(Timestamp, "MAX_VALUE"); - let mut processor = init_processor( - "SELECT MAX_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ----------------- - MAX_VALUE = Italy - */ - let mut inp = insert_field(ITALY, &get_ts_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - ----------------- - MAX_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - Singapore, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, &get_ts_field(150)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Singapore, 100.0 - Italy, 150.0 - --------------- - MAX_VALUE = Italy - */ - inp = update_field(SINGAPORE, ITALY, &get_ts_field(150), &get_ts_field(150)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore value 100 -> 200 - /* - Italy, 100.0 - Singapore, 200.0 - Italy, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = update_field(SINGAPORE, SINGAPORE, &get_ts_field(100), &get_ts_field(200)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 150.0 - --------------- - MAX_VALUE = Italy - */ - inp = delete_field(SINGAPORE, &get_ts_field(200)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Italy segment to Singapore - /* - Italy, 100.0 - Singapore, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = update_field(ITALY, SINGAPORE, &get_ts_field(150), &get_ts_field(150)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete another record (150) - /* - Italy, 100.0 - ------------- - MAX_VALUE = Italy - */ - inp = delete_field(SINGAPORE, &get_ts_field(150)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX_VALUE = Null - */ - inp = delete_field(ITALY, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![delete_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_date() { - let schema = init_val_input_schema(Date, "MAX_VALUE"); - let mut processor = init_processor( - "SELECT MAX_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ----------------- - MAX_VALUE = Italy - */ - let mut inp = insert_field(ITALY, &get_date_field(DATE4)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - ----------------- - MAX_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, &get_date_field(DATE4)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - Singapore, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Singapore, 100.0 - Italy, 150.0 - --------------- - MAX_VALUE = Italy - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_date_field(DATE8), - &get_date_field(DATE8), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore value 100 -> 200 - /* - Italy, 100.0 - Singapore, 200.0 - Italy, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = update_field( - SINGAPORE, - SINGAPORE, - &get_date_field(DATE4), - &get_date_field(DATE16), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 150.0 - --------------- - MAX_VALUE = Italy - */ - inp = delete_field(SINGAPORE, &get_date_field(DATE16)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Italy segment to Singapore - /* - Italy, 100.0 - Singapore, 150.0 - --------------- - MAX_VALUE = Singapore - */ - inp = update_field( - ITALY, - SINGAPORE, - &get_date_field(DATE8), - &get_date_field(DATE8), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete another record (150) - /* - Italy, 100.0 - ------------- - MAX_VALUE = Italy - */ - inp = delete_field(SINGAPORE, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX_VALUE = Null - */ - inp = delete_field(ITALY, &get_date_field(DATE4)); - out = output!(processor, inp); - exp = vec![delete_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_int_null() { - let schema = init_input_schema(Int, "MAX_VALUE"); - let mut processor = init_processor( - "SELECT MAX_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ---------------- - MAX_VALUE = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - MAX_VALUE = 100 - */ - inp = insert_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_val_exp( - FIELD_NULL, - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MAX_VALUE = 0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_INT, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - FIELD_NULL, - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MAX_VALUE = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp(FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX_VALUE = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_float_null() { - let schema = init_input_schema(Float, "MAX_VALUE"); - let mut processor = init_processor( - "SELECT MAX_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX_VALUE = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MAX_VALUE = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_val_exp( - FIELD_NULL, - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MAX_VALUE = 0.0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_FLOAT, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - FIELD_NULL, - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MAX_VALUE = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp(FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX_VALUE = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_decimal_null() { - let schema = init_input_schema(Decimal, "MAX_VALUE"); - let mut processor = init_processor( - "SELECT MAX_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX_VALUE = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MAX_VALUE = 100.0 - */ - inp = insert_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_val_exp( - FIELD_NULL, - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MAX_VALUE = 0.0 - */ - inp = update_field(ITALY, ITALY, &Field::String(ITALY.to_string()), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MAX_VALUE = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX_VALUE = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_duration_null() { - let schema = init_input_schema(Duration, "MAX_VALUE"); - let mut processor = init_processor( - "SELECT MAX_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX_VALUE = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MAX_VALUE = 100.0 - */ - inp = insert_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_val_exp( - FIELD_NULL, - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MAX_VALUE = 0.0 - */ - inp = update_field(ITALY, ITALY, &get_duration_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - FIELD_NULL, - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MAX_VALUE = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp(FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX_VALUE = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_timestamp_null() { - let schema = init_input_schema(Timestamp, "MAX_VALUE"); - let mut processor = init_processor( - "SELECT MAX_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX_VALUE = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - MAX_VALUE = 100 - */ - inp = insert_field(ITALY, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![update_val_exp( - FIELD_NULL, - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MAX_VALUE = 0 - */ - inp = update_field(ITALY, ITALY, &get_ts_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - FIELD_NULL, - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MAX_VALUE = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp(FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX_VALUE = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_max_aggregation_date_null() { - let schema = init_input_schema(Date, "MAX_VALUE"); - let mut processor = init_processor( - "SELECT MAX_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MAX_VALUE = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 2015-10-08 for segment Italy - /* - Italy, NULL - Italy, 2015-10-08 - ------------- - MAX_VALUE = 2015-10-08 - */ - inp = insert_field(ITALY, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![update_val_exp( - FIELD_NULL, - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update 2015-10-08 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MAX_VALUE = 0 - */ - inp = update_field(ITALY, ITALY, &get_date_field(DATE8), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - FIELD_NULL, - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MAX_VALUE = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp(FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MAX_VALUE = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); -} diff --git a/dozer-sql/src/aggregation/tests/aggregation_min_append_only_tests.rs b/dozer-sql/src/aggregation/tests/aggregation_min_append_only_tests.rs deleted file mode 100644 index 7b66d5cac2..0000000000 --- a/dozer-sql/src/aggregation/tests/aggregation_min_append_only_tests.rs +++ /dev/null @@ -1,608 +0,0 @@ -use crate::aggregation::tests::aggregation_tests_utils::{ - get_date_field, get_decimal_field, get_duration_field, get_ts_field, init_input_schema, - init_processor, insert_exp, insert_field, update_exp, DATE4, DATE8, FIELD_100_FLOAT, - FIELD_100_INT, FIELD_100_UINT, FIELD_50_FLOAT, FIELD_50_INT, FIELD_50_UINT, FIELD_NULL, ITALY, - SINGAPORE, -}; -use crate::output; -use dozer_core::DEFAULT_PORT_HANDLE; - -use dozer_types::types::FieldType::{Date, Decimal, Duration, Float, Int, Timestamp, UInt}; -use std::collections::HashMap; - -#[test] -fn test_min_aggregation_float() { - let schema = init_input_schema(Float, "MIN_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MIN_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_FLOAT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_FLOAT, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - - Singapore, 50.0 - --------------- - MIN_APPEND_ONLY = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_50_FLOAT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_int() { - let schema = init_input_schema(Int, "MIN_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MIN_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_INT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_100_INT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_INT, FIELD_100_INT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - - Singapore, 50.0 - --------------- - MIN_APPEND_ONLY = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_50_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_uint() { - let schema = init_input_schema(UInt, "MIN_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MIN_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_UINT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_100_UINT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_UINT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_UINT, FIELD_100_UINT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - - Singapore, 50.0 - --------------- - MIN_APPEND_ONLY = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_UINT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_50_UINT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_decimal() { - let schema = init_input_schema(Decimal, "MIN_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MIN_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - */ - let mut inp = insert_field(ITALY, &get_decimal_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_decimal_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - */ - inp = insert_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - - Singapore, 50.0 - ------------- - MIN_APPEND_ONLY = 50.0 - */ - inp = insert_field(SINGAPORE, &get_decimal_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_decimal_field(50))]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_duration() { - let schema = init_input_schema(Duration, "MIN_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MIN_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - */ - let mut inp = insert_field(ITALY, &get_duration_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_duration_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - */ - inp = insert_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_field(100), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - - Singapore, 50.0 - ------------- - MIN_APPEND_ONLY = 50.0 - */ - inp = insert_field(SINGAPORE, &get_duration_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_duration_field(50))]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_timestamp() { - let schema = init_input_schema(Timestamp, "MIN_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MIN_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100 - ------------- - MIN_APPEND_ONLY = 100 - */ - let mut inp = insert_field(ITALY, &get_ts_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_ts_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100 - Italy, 100 - ------------- - MIN_APPEND_ONLY = 100 - */ - inp = insert_field(ITALY, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_ts_field(100), - &get_ts_field(100), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100 - Italy, 100 - ------------- - MIN_APPEND_ONLY = 100 - - Singapore, 50 - ------------- - MIN_APPEND_ONLY = 50 - */ - inp = insert_field(SINGAPORE, &get_ts_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_ts_field(50))]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_date() { - let schema = init_input_schema(Date, "MIN_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MIN_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 2015-10-08 for segment Italy - /* - Italy, 2015-10-08 - ------------------ - MIN_APPEND_ONLY = 2015-10-08 - */ - let mut inp = insert_field(ITALY, &get_date_field(DATE8)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_date_field(DATE8))]; - assert_eq!(out, exp); - - // Insert another 2015-10-08 for segment Italy - /* - Italy, 2015-10-08 - Italy, 2015-10-08 - ----------------- - MIN_APPEND_ONLY = 2015-10-08 - */ - inp = insert_field(ITALY, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_date_field(DATE8), - &get_date_field(DATE8), - )]; - assert_eq!(out, exp); - - // Insert 2015-10-04 for segment Singapore - /* - Italy, 2015-10-08 - Italy, 2015-10-08 - ------------- - MIN_APPEND_ONLY = 2015-10-08 - - Singapore, 2015-10-04 - ------------- - MIN_APPEND_ONLY = 2015-10-04 - */ - inp = insert_field(SINGAPORE, &get_date_field(DATE4)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_date_field(DATE4))]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_int_null() { - let schema = init_input_schema(Int, "MIN_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MIN_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN_APPEND_ONLY = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - MIN_APPEND_ONLY = 100 - */ - inp = insert_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_100_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_float_null() { - let schema = init_input_schema(Float, "MIN_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MIN_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN_APPEND_ONLY = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_100_FLOAT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_decimal_null() { - let schema = init_input_schema(Decimal, "MIN_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MIN_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN_APPEND_ONLY = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - */ - inp = insert_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - FIELD_NULL, - &get_decimal_field(100), - )]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_duration_null() { - let schema = init_input_schema(Duration, "MIN_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MIN_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN_APPEND_ONLY = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MIN_APPEND_ONLY = 100.0 - */ - inp = insert_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - FIELD_NULL, - &get_duration_field(100), - )]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_timestamp_null() { - let schema = init_input_schema(Timestamp, "MIN_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MIN_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN_APPEND_ONLY = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - MIN_APPEND_ONLY = 100 - */ - inp = insert_field(ITALY, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, &get_ts_field(100))]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_date_null() { - let schema = init_input_schema(Date, "MIN_APPEND_ONLY"); - let mut processor = init_processor( - "SELECT Country, MIN_APPEND_ONLY(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN_APPEND_ONLY = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 2015-10-08 for segment Italy - /* - Italy, NULL - Italy, 2015-10-08 - ------------- - MIN_APPEND_ONLY = 2015-10-08 - */ - inp = insert_field(ITALY, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, &get_date_field(DATE8))]; - assert_eq!(out, exp); -} diff --git a/dozer-sql/src/aggregation/tests/aggregation_min_tests.rs b/dozer-sql/src/aggregation/tests/aggregation_min_tests.rs deleted file mode 100644 index d1ff5d2a16..0000000000 --- a/dozer-sql/src/aggregation/tests/aggregation_min_tests.rs +++ /dev/null @@ -1,1350 +0,0 @@ -use crate::aggregation::tests::aggregation_tests_utils::{ - delete_exp, delete_field, get_date_field, get_decimal_field, get_duration_field, get_ts_field, - init_input_schema, init_processor, insert_exp, insert_field, update_exp, update_field, DATE16, - DATE4, DATE8, FIELD_100_FLOAT, FIELD_100_INT, FIELD_100_UINT, FIELD_200_FLOAT, FIELD_200_INT, - FIELD_200_UINT, FIELD_50_FLOAT, FIELD_50_INT, FIELD_50_UINT, FIELD_NULL, ITALY, SINGAPORE, -}; -use crate::output; -use dozer_core::DEFAULT_PORT_HANDLE; - -use dozer_types::types::FieldType::{Date, Decimal, Duration, Float, Int, Timestamp, UInt}; -use std::collections::HashMap; - -#[test] -fn test_min_aggregation_float() { - let schema = init_input_schema(Float, "MIN"); - let mut processor = init_processor( - "SELECT Country, MIN(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MIN = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_FLOAT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_FLOAT, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN = 100.0 - - Singapore, 50.0 - --------------- - MIN = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_50_FLOAT)]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MIN = 50.0 - */ - inp = update_field(SINGAPORE, ITALY, FIELD_50_FLOAT, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, FIELD_50_FLOAT), - update_exp(ITALY, ITALY, FIELD_100_FLOAT, FIELD_50_FLOAT), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MIN = 50.0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_FLOAT, FIELD_200_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_50_FLOAT, FIELD_50_FLOAT)]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - MIN = 50.0 - */ - inp = delete_field(ITALY, FIELD_200_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_50_FLOAT, FIELD_50_FLOAT)]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - MIN = 100.0 - */ - inp = delete_field(ITALY, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_50_FLOAT, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN = Null - */ - inp = delete_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_100_FLOAT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_int() { - let schema = init_input_schema(Int, "MIN"); - let mut processor = init_processor( - "SELECT Country, MIN(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MIN = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_INT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_100_INT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_INT, FIELD_100_INT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN = 100.0 - - Singapore, 50.0 - --------------- - MIN = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_50_INT)]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MIN = 50.0 - */ - inp = update_field(SINGAPORE, ITALY, FIELD_50_INT, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, FIELD_50_INT), - update_exp(ITALY, ITALY, FIELD_100_INT, FIELD_50_INT), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MIN = 50.0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_INT, FIELD_200_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_50_INT, FIELD_50_INT)]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - MIN = 50.0 - */ - inp = delete_field(ITALY, FIELD_200_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_50_INT, FIELD_50_INT)]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - MIN = 100.0 - */ - inp = delete_field(ITALY, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_50_INT, FIELD_100_INT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN = Null - */ - inp = delete_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_100_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_uint() { - let schema = init_input_schema(UInt, "MIN"); - let mut processor = init_processor( - "SELECT Country, MIN(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MIN = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_UINT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_100_UINT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_UINT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_UINT, FIELD_100_UINT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN = 100.0 - - Singapore, 50.0 - --------------- - MIN = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_UINT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_50_UINT)]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MIN = 50.0 - */ - inp = update_field(SINGAPORE, ITALY, FIELD_50_UINT, FIELD_50_UINT); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, FIELD_50_UINT), - update_exp(ITALY, ITALY, FIELD_100_UINT, FIELD_50_UINT), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MIN = 50.0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_UINT, FIELD_200_UINT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_50_UINT, FIELD_50_UINT)]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - MIN = 50.0 - */ - inp = delete_field(ITALY, FIELD_200_UINT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_50_UINT, FIELD_50_UINT)]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - MIN = 100.0 - */ - inp = delete_field(ITALY, FIELD_50_UINT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_50_UINT, FIELD_100_UINT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN = Null - */ - inp = delete_field(ITALY, FIELD_100_UINT); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_100_UINT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_decimal() { - let schema = init_input_schema(Decimal, "MIN"); - let mut processor = init_processor( - "SELECT Country, MIN(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MIN = 100.0 - */ - let mut inp = insert_field(ITALY, &get_decimal_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_decimal_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN = 100.0 - */ - inp = insert_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN = 100.0 - - Singapore, 50.0 - ------------- - MIN = 50.0 - */ - inp = insert_field(SINGAPORE, &get_decimal_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_decimal_field(50))]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MIN = 50.0 - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_decimal_field(50), - &get_decimal_field(50), - ); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, &get_decimal_field(50)), - update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(50), - ), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MIN = 50.0 - */ - inp = update_field( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(200), - ); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(50), - &get_decimal_field(50), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - MIN = 50.0 - */ - inp = delete_field(ITALY, &get_decimal_field(200)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(50), - &get_decimal_field(50), - )]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - MIN = 100.0 - */ - inp = delete_field(ITALY, &get_decimal_field(50)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(50), - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN = 0.0 - */ - inp = delete_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_decimal_field(100))]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_duration() { - let schema = init_input_schema(Duration, "MIN"); - let mut processor = init_processor( - "SELECT Country, MIN(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - MIN = 100.0 - */ - let mut inp = insert_field(ITALY, &get_duration_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_duration_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN = 100.0 - */ - inp = insert_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_field(100), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - MIN = 100.0 - - Singapore, 50.0 - ------------- - MIN = 50.0 - */ - inp = insert_field(SINGAPORE, &get_duration_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_duration_field(50))]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MIN = 50.0 - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_duration_field(50), - &get_duration_field(50), - ); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, &get_duration_field(50)), - update_exp( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_field(50), - ), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - MIN = 50.0 - */ - inp = update_field( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_field(200), - ); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(50), - &get_duration_field(50), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - MIN = 50.0 - */ - inp = delete_field(ITALY, &get_duration_field(200)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(50), - &get_duration_field(50), - )]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - MIN = 100.0 - */ - inp = delete_field(ITALY, &get_duration_field(50)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(50), - &get_duration_field(100), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN = 0.0 - */ - inp = delete_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_duration_field(100))]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_timestamp() { - let schema = init_input_schema(Timestamp, "MIN"); - let mut processor = init_processor( - "SELECT Country, MIN(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100 - ------------- - MIN = 100 - */ - let mut inp = insert_field(ITALY, &get_ts_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_ts_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100 - Italy, 100 - ------------- - MIN = 100 - */ - inp = insert_field(ITALY, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_ts_field(100), - &get_ts_field(100), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100 - Italy, 100 - ------------- - MIN = 100 - - Singapore, 50 - ------------- - MIN = 50 - */ - inp = insert_field(SINGAPORE, &get_ts_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_ts_field(50))]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100 - Italy, 100 - Italy, 50 - ------------- - MIN = 50 - */ - inp = update_field(SINGAPORE, ITALY, &get_ts_field(50), &get_ts_field(50)); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, &get_ts_field(50)), - update_exp(ITALY, ITALY, &get_ts_field(100), &get_ts_field(50)), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200 - Italy, 100 - Italy, 50 - ------------- - MIN = 50 - */ - inp = update_field(ITALY, ITALY, &get_ts_field(100), &get_ts_field(200)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_ts_field(50), - &get_ts_field(50), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100 - Italy, 50 - ------------- - MIN = 50 - */ - inp = delete_field(ITALY, &get_ts_field(200)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_ts_field(50), - &get_ts_field(50), - )]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100 - ------------- - MIN = 100 - */ - inp = delete_field(ITALY, &get_ts_field(50)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_ts_field(50), - &get_ts_field(100), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN = 0 - */ - inp = delete_field(ITALY, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_ts_field(100))]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_date() { - let schema = init_input_schema(Date, "MIN"); - let mut processor = init_processor( - "SELECT Country, MIN(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 2015-10-08 for segment Italy - /* - Italy, 2015-10-08 - ------------------ - MIN = 2015-10-08 - */ - let mut inp = insert_field(ITALY, &get_date_field(DATE8)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_date_field(DATE8))]; - assert_eq!(out, exp); - - // Insert another 2015-10-08 for segment Italy - /* - Italy, 2015-10-08 - Italy, 2015-10-08 - ----------------- - MIN = 2015-10-08 - */ - inp = insert_field(ITALY, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_date_field(DATE8), - &get_date_field(DATE8), - )]; - assert_eq!(out, exp); - - // Insert 2015-10-04 for segment Singapore - /* - Italy, 2015-10-08 - Italy, 2015-10-08 - ------------- - MIN = 2015-10-08 - - Singapore, 2015-10-04 - ------------- - MIN = 2015-10-04 - */ - inp = insert_field(SINGAPORE, &get_date_field(DATE4)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_date_field(DATE4))]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 2015-10-08 - Italy, 2015-10-08 - Italy, 2015-10-04 - ------------- - MIN = 2015-10-04 - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_date_field(DATE4), - &get_date_field(DATE4), - ); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, &get_date_field(DATE4)), - update_exp(ITALY, ITALY, &get_date_field(DATE8), &get_date_field(DATE4)), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 2015-10-16 - Italy, 2015-10-08 - Italy, 2015-10-04 - ------------- - MIN = 2015-10-04 - */ - inp = update_field( - ITALY, - ITALY, - &get_date_field(DATE8), - &get_date_field(DATE16), - ); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_date_field(DATE4), - &get_date_field(DATE4), - )]; - assert_eq!(out, exp); - - // Delete 1 record (2015-10-16) - /* - Italy, 2015-10-08 - Italy, 2015-10-04 - ------------- - MIN = 2015-10-04 - */ - inp = delete_field(ITALY, &get_date_field(DATE16)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_date_field(DATE4), - &get_date_field(DATE4), - )]; - assert_eq!(out, exp); - - // Delete another record (2015-10-04) - /* - Italy, 2015-10-08 - ------------- - MIN = 2015-10-08 - */ - inp = delete_field(ITALY, &get_date_field(DATE4)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_date_field(DATE4), - &get_date_field(DATE8), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN = 0 - */ - inp = delete_field(ITALY, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_date_field(DATE8))]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_int_null() { - let schema = init_input_schema(Int, "MIN"); - let mut processor = init_processor( - "SELECT Country, MIN(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN = 0 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - MIN = 0 - */ - inp = insert_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_100_INT)]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MIN = 0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_INT, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_INT, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MIN = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_float_null() { - let schema = init_input_schema(Float, "MIN"); - let mut processor = init_processor( - "SELECT Country, MIN(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN = 0.0 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MIN = 0.0 - */ - inp = insert_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MIN = 0.0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_FLOAT, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_FLOAT, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MIN = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_decimal_null() { - let schema = init_input_schema(Decimal, "MIN"); - let mut processor = init_processor( - "SELECT Country, MIN(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MIN = NULL - */ - inp = insert_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - FIELD_NULL, - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MIN = 0.0 - */ - inp = update_field(ITALY, ITALY, &get_decimal_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - FIELD_NULL, - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MIN = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_duration_null() { - let schema = init_input_schema(Duration, "MIN"); - let mut processor = init_processor( - "SELECT Country, MIN(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MIN = NULL - */ - inp = insert_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - FIELD_NULL, - &get_duration_field(100), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MIN = 0.0 - */ - inp = update_field(ITALY, ITALY, &get_duration_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(100), - FIELD_NULL, - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MIN = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_timestamp_null() { - let schema = init_input_schema(Timestamp, "MIN"); - let mut processor = init_processor( - "SELECT Country, MIN(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN = 0 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - MIN = 100 - */ - inp = insert_field(ITALY, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, &get_ts_field(100))]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MIN = 0 - */ - inp = update_field(ITALY, ITALY, &get_ts_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, &get_ts_field(100), FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MIN = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_date_null() { - let schema = init_input_schema(Date, "MIN"); - let mut processor = init_processor( - "SELECT Country, MIN(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 2015-10-08 for segment Italy - /* - Italy, NULL - Italy, 2015-10-08 - ------------- - MIN = 0 - */ - inp = insert_field(ITALY, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, &get_date_field(DATE8))]; - assert_eq!(out, exp); - - // Update 2015-10-08 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MIN = NULL - */ - inp = update_field(ITALY, ITALY, &get_date_field(DATE8), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, &get_date_field(DATE8), FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MIN = NULL - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_NULL)]; - assert_eq!(out, exp); -} diff --git a/dozer-sql/src/aggregation/tests/aggregation_min_value_tests.rs b/dozer-sql/src/aggregation/tests/aggregation_min_value_tests.rs deleted file mode 100644 index 0ec34b08c0..0000000000 --- a/dozer-sql/src/aggregation/tests/aggregation_min_value_tests.rs +++ /dev/null @@ -1,1456 +0,0 @@ -use crate::aggregation::tests::aggregation_tests_utils::{ - delete_field, delete_val_exp, get_date_field, get_decimal_field, get_duration_field, - get_ts_field, init_input_schema, init_processor, init_val_input_schema, insert_field, - insert_val_exp, update_field, update_val_exp, DATE16, DATE4, DATE8, FIELD_100_FLOAT, - FIELD_100_INT, FIELD_100_UINT, FIELD_50_FLOAT, FIELD_50_INT, FIELD_50_UINT, FIELD_75_FLOAT, - FIELD_75_INT, FIELD_75_UINT, FIELD_NULL, ITALY, SINGAPORE, -}; -use crate::output; -use dozer_core::DEFAULT_PORT_HANDLE; - -use dozer_types::types::Field; -use dozer_types::types::FieldType::{Date, Decimal, Duration, Float, Int, Timestamp, UInt}; -use std::collections::HashMap; - -#[test] -fn test_min_aggregation_float() { - let schema = init_val_input_schema(Float, "MIN_VALUE"); - let mut processor = init_processor( - "SELECT MIN_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ----------------- - MIN_VALUE = Italy - */ - let mut inp = insert_field(ITALY, FIELD_100_FLOAT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - ----------------- - MIN_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Insert 75 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - Singapore, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, FIELD_75_FLOAT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Singapore, 100.0 - Italy, 75.0 - --------------- - MIN_VALUE = Italy - */ - inp = update_field(SINGAPORE, ITALY, FIELD_75_FLOAT, FIELD_75_FLOAT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore value 100 -> 50 - /* - Italy, 100.0 - Singapore, 50.0 - Italy, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = update_field(SINGAPORE, SINGAPORE, FIELD_100_FLOAT, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete 1 record (50) - /* - Italy, 100.0 - Italy, 75.0 - --------------- - MIN_VALUE = Italy - */ - inp = delete_field(SINGAPORE, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Italy segment to Singapore - /* - Italy, 100.0 - Singapore, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = update_field(ITALY, SINGAPORE, FIELD_75_FLOAT, FIELD_75_FLOAT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete another record (75) - /* - Italy, 100.0 - ------------- - MIN_VALUE = Italy - */ - inp = delete_field(SINGAPORE, FIELD_75_FLOAT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN_VALUE = Null - */ - inp = delete_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![delete_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_int() { - let schema = init_val_input_schema(Int, "MIN_VALUE"); - let mut processor = init_processor( - "SELECT MIN_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ----------------- - MIN_VALUE = Italy - */ - let mut inp = insert_field(ITALY, FIELD_100_INT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - ----------------- - MIN_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Insert 75 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - Singapore, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, FIELD_75_INT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Singapore, 100.0 - Italy, 75.0 - --------------- - MIN_VALUE = Italy - */ - inp = update_field(SINGAPORE, ITALY, FIELD_75_INT, FIELD_75_INT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore value 100 -> 50 - /* - Italy, 100.0 - Singapore, 50.0 - Italy, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = update_field(SINGAPORE, SINGAPORE, FIELD_100_INT, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete 1 record (50) - /* - Italy, 100.0 - Italy, 75.0 - --------------- - MIN_VALUE = Italy - */ - inp = delete_field(SINGAPORE, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Italy segment to Singapore - /* - Italy, 100.0 - Singapore, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = update_field(ITALY, SINGAPORE, FIELD_75_INT, FIELD_75_INT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete another record (75) - /* - Italy, 100.0 - ------------- - MIN_VALUE = Italy - */ - inp = delete_field(SINGAPORE, FIELD_75_INT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN_VALUE = Null - */ - inp = delete_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![delete_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_uint() { - let schema = init_val_input_schema(UInt, "MIN_VALUE"); - let mut processor = init_processor( - "SELECT MIN_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ----------------- - MIN_VALUE = Italy - */ - let mut inp = insert_field(ITALY, FIELD_100_UINT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - ----------------- - MIN_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, FIELD_100_UINT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Insert 75 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - Singapore, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, FIELD_75_UINT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Singapore, 100.0 - Italy, 75.0 - --------------- - MIN_VALUE = Italy - */ - inp = update_field(SINGAPORE, ITALY, FIELD_75_UINT, FIELD_75_UINT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore value 100 -> 50 - /* - Italy, 100.0 - Singapore, 50.0 - Italy, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = update_field(SINGAPORE, SINGAPORE, FIELD_100_UINT, FIELD_50_UINT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete 1 record (50) - /* - Italy, 100.0 - Italy, 75.0 - --------------- - MIN_VALUE = Italy - */ - inp = delete_field(SINGAPORE, FIELD_50_UINT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Italy segment to Singapore - /* - Italy, 100.0 - Singapore, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = update_field(ITALY, SINGAPORE, FIELD_75_UINT, FIELD_75_UINT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete another record (75) - /* - Italy, 100.0 - ------------- - MIN_VALUE = Italy - */ - inp = delete_field(SINGAPORE, FIELD_75_UINT); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN_VALUE = Null - */ - inp = delete_field(ITALY, FIELD_100_UINT); - out = output!(processor, inp); - exp = vec![delete_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_decimal() { - let schema = init_val_input_schema(Decimal, "MIN_VALUE"); - let mut processor = init_processor( - "SELECT MIN_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ----------------- - MIN_VALUE = Italy - */ - let mut inp = insert_field(ITALY, &get_decimal_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - ----------------- - MIN_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Insert 75 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - Singapore, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, &get_decimal_field(75)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Singapore, 100.0 - Italy, 75.0 - --------------- - MIN_VALUE = Italy - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_decimal_field(75), - &get_decimal_field(75), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore value 100 -> 50 - /* - Italy, 100.0 - Singapore, 50.0 - Italy, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = update_field( - SINGAPORE, - SINGAPORE, - &get_decimal_field(100), - &get_decimal_field(50), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete 1 record (50) - /* - Italy, 100.0 - Italy, 75.0 - --------------- - MIN_VALUE = Italy - */ - inp = delete_field(SINGAPORE, &get_decimal_field(50)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Italy segment to Singapore - /* - Italy, 100.0 - Singapore, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = update_field( - ITALY, - SINGAPORE, - &get_decimal_field(75), - &get_decimal_field(75), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete another record (75) - /* - Italy, 100.0 - ------------- - MIN_VALUE = Italy - */ - inp = delete_field(SINGAPORE, &get_decimal_field(75)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN_VALUE = Null - */ - inp = delete_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![delete_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_duration() { - let schema = init_val_input_schema(Duration, "MIN_VALUE"); - let mut processor = init_processor( - "SELECT MIN_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ----------------- - MIN_VALUE = Italy - */ - let mut inp = insert_field(ITALY, &get_duration_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - ----------------- - MIN_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Insert 75 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - Singapore, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, &get_duration_field(75)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Singapore, 100.0 - Italy, 75.0 - --------------- - MIN_VALUE = Italy - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_duration_field(75), - &get_duration_field(75), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore value 100 -> 50 - /* - Italy, 100.0 - Singapore, 50.0 - Italy, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = update_field( - SINGAPORE, - SINGAPORE, - &get_duration_field(100), - &get_duration_field(50), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete 1 record (50) - /* - Italy, 100.0 - Italy, 75.0 - --------------- - MIN_VALUE = Italy - */ - inp = delete_field(SINGAPORE, &get_duration_field(50)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Italy segment to Singapore - /* - Italy, 100.0 - Singapore, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = update_field( - ITALY, - SINGAPORE, - &get_duration_field(75), - &get_duration_field(75), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete another record (75) - /* - Italy, 100.0 - ------------- - MIN_VALUE = Italy - */ - inp = delete_field(SINGAPORE, &get_duration_field(75)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN_VALUE = Null - */ - inp = delete_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![delete_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_timestamp() { - let schema = init_val_input_schema(Timestamp, "MIN_VALUE"); - let mut processor = init_processor( - "SELECT MIN_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ----------------- - MIN_VALUE = Italy - */ - let mut inp = insert_field(ITALY, &get_ts_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - ----------------- - MIN_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Insert 75 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - Singapore, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, &get_ts_field(75)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Singapore, 100.0 - Italy, 75.0 - --------------- - MIN_VALUE = Italy - */ - inp = update_field(SINGAPORE, ITALY, &get_ts_field(75), &get_ts_field(75)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore value 100 -> 50 - /* - Italy, 100.0 - Singapore, 50.0 - Italy, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = update_field(SINGAPORE, SINGAPORE, &get_ts_field(100), &get_ts_field(50)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete 1 record (50) - /* - Italy, 100.0 - Italy, 75.0 - --------------- - MIN_VALUE = Italy - */ - inp = delete_field(SINGAPORE, &get_ts_field(50)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Italy segment to Singapore - /* - Italy, 100.0 - Singapore, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = update_field(ITALY, SINGAPORE, &get_ts_field(75), &get_ts_field(75)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete another record (75) - /* - Italy, 100.0 - ------------- - MIN_VALUE = Italy - */ - inp = delete_field(SINGAPORE, &get_ts_field(75)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN_VALUE = Null - */ - inp = delete_field(ITALY, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![delete_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_date() { - let schema = init_val_input_schema(Date, "MIN_VALUE"); - let mut processor = init_processor( - "SELECT MIN_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ----------------- - MIN_VALUE = Italy - */ - let mut inp = insert_field(ITALY, &get_date_field(DATE16)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - ----------------- - MIN_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, &get_date_field(DATE16)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Insert 75 for segment Singapore - /* - Italy, 100.0 - Singapore, 100.0 - Singapore, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = insert_field(SINGAPORE, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Singapore, 100.0 - Italy, 75.0 - --------------- - MIN_VALUE = Italy - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_date_field(DATE8), - &get_date_field(DATE8), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Singapore value 100 -> 50 - /* - Italy, 100.0 - Singapore, 50.0 - Italy, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = update_field( - SINGAPORE, - SINGAPORE, - &get_date_field(DATE16), - &get_date_field(DATE4), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete 1 record (50) - /* - Italy, 100.0 - Italy, 75.0 - --------------- - MIN_VALUE = Italy - */ - inp = delete_field(SINGAPORE, &get_date_field(DATE4)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update Italy segment to Singapore - /* - Italy, 100.0 - Singapore, 75.0 - --------------- - MIN_VALUE = Singapore - */ - inp = update_field( - ITALY, - SINGAPORE, - &get_date_field(DATE8), - &get_date_field(DATE8), - ); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - &Field::String(SINGAPORE.to_string()), - )]; - assert_eq!(out, exp); - - // Delete another record (75) - /* - Italy, 100.0 - ------------- - MIN_VALUE = Italy - */ - inp = delete_field(SINGAPORE, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(SINGAPORE.to_string()), - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN_VALUE = Null - */ - inp = delete_field(ITALY, &get_date_field(DATE16)); - out = output!(processor, inp); - exp = vec![delete_val_exp(&Field::String(ITALY.to_string()))]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_int_null() { - let schema = init_input_schema(Int, "MIN_VALUE"); - let mut processor = init_processor( - "SELECT MIN_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ---------------- - MIN_VALUE = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - MIN_VALUE = 100 - */ - inp = insert_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_val_exp( - FIELD_NULL, - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MIN_VALUE = 0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_INT, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - FIELD_NULL, - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MIN_VALUE = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp(FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN_VALUE = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_float_null() { - let schema = init_input_schema(Float, "MIN_VALUE"); - let mut processor = init_processor( - "SELECT MIN_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN_VALUE = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MIN_VALUE = 100.0 - */ - inp = insert_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_val_exp( - FIELD_NULL, - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MIN_VALUE = 0.0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_FLOAT, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - FIELD_NULL, - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MIN_VALUE = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp(FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN_VALUE = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_decimal_null() { - let schema = init_input_schema(Decimal, "MIN_VALUE"); - let mut processor = init_processor( - "SELECT MIN_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN_VALUE = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MIN_VALUE = 100.0 - */ - inp = insert_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_val_exp( - FIELD_NULL, - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MIN_VALUE = 0.0 - */ - inp = update_field(ITALY, ITALY, &get_decimal_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - FIELD_NULL, - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MIN_VALUE = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp(FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN_VALUE = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_duration_null() { - let schema = init_input_schema(Duration, "MIN_VALUE"); - let mut processor = init_processor( - "SELECT MIN_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN_VALUE = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100.0 - ------------- - MIN_VALUE = 100.0 - */ - inp = insert_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_val_exp( - FIELD_NULL, - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MIN_VALUE = 0.0 - */ - inp = update_field(ITALY, ITALY, &get_duration_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - FIELD_NULL, - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MIN_VALUE = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp(FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN_VALUE = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_timestamp_null() { - let schema = init_input_schema(Timestamp, "MIN_VALUE"); - let mut processor = init_processor( - "SELECT MIN_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN_VALUE = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - MIN_VALUE = 100 - */ - inp = insert_field(ITALY, &get_ts_field(100)); - out = output!(processor, inp); - exp = vec![update_val_exp( - FIELD_NULL, - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MIN_VALUE = 0 - */ - inp = update_field(ITALY, ITALY, &get_ts_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - FIELD_NULL, - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MIN_VALUE = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp(FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN_VALUE = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); -} - -#[test] -fn test_min_aggregation_date_null() { - let schema = init_input_schema(Date, "MIN_VALUE"); - let mut processor = init_processor( - "SELECT MIN_VALUE(Salary, Country) FROM Users", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - MIN_VALUE = NULL - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); - - // Insert 2015-10-08 for segment Italy - /* - Italy, NULL - Italy, 2015-10-08 - ------------- - MIN_VALUE = 2015-10-08 - */ - inp = insert_field(ITALY, &get_date_field(DATE8)); - out = output!(processor, inp); - exp = vec![update_val_exp( - FIELD_NULL, - &Field::String(ITALY.to_string()), - )]; - assert_eq!(out, exp); - - // Update 2015-10-08 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - MIN_VALUE = 0 - */ - inp = update_field(ITALY, ITALY, &get_date_field(DATE8), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp( - &Field::String(ITALY.to_string()), - FIELD_NULL, - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - MIN_VALUE = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_val_exp(FIELD_NULL, FIELD_NULL)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - MIN_VALUE = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_val_exp(FIELD_NULL)]; - assert_eq!(out, exp); -} diff --git a/dozer-sql/src/aggregation/tests/aggregation_null.rs b/dozer-sql/src/aggregation/tests/aggregation_null.rs deleted file mode 100644 index 4965cdc125..0000000000 --- a/dozer-sql/src/aggregation/tests/aggregation_null.rs +++ /dev/null @@ -1,86 +0,0 @@ -use crate::output; - -use crate::aggregation::tests::aggregation_tests_utils::{ - delete_exp, delete_field, init_input_schema, init_processor, insert_exp, insert_field, - FIELD_100_INT, FIELD_1_INT, ITALY, -}; - -use dozer_core::DEFAULT_PORT_HANDLE; - -use dozer_types::types::FieldType::Int; -use dozer_types::types::{Field, Operation, Record}; -use std::collections::HashMap; - -#[test] -fn test_sum_aggregation_null() { - let schema = init_input_schema(Int, "SUM"); - let mut processor = init_processor( - "SELECT Country, SUM(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - NULL, 100.0 - ------------- - SUM = 100.0 - */ - let record = Record::new(vec![ - Field::Int(0), - Field::Null, - FIELD_100_INT.clone(), - FIELD_100_INT.clone(), - ]); - let inp = Operation::Insert { new: record }; - let out = output!(processor, inp); - let exp_record = Record::new(vec![Field::Null, FIELD_100_INT.clone()]); - let exp = vec![Operation::Insert { new: exp_record }]; - assert_eq!(out, exp); -} - -#[test] -fn test_sum_aggregation_del_and_insert() { - let schema = init_input_schema(Int, "COUNT"); - let mut processor = init_processor( - "SELECT Country, COUNT(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - COUNT = 1 - */ - let mut inp = insert_field(ITALY, FIELD_100_INT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - COUNT = 0 - */ - inp = delete_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - COUNT = 1 - */ - let inp = insert_field(ITALY, FIELD_100_INT); - let out = output!(processor, inp); - let exp = vec![insert_exp(ITALY, FIELD_1_INT)]; - assert_eq!(out, exp); -} diff --git a/dozer-sql/src/aggregation/tests/aggregation_sum_tests.rs b/dozer-sql/src/aggregation/tests/aggregation_sum_tests.rs deleted file mode 100644 index 7bc07b01a4..0000000000 --- a/dozer-sql/src/aggregation/tests/aggregation_sum_tests.rs +++ /dev/null @@ -1,949 +0,0 @@ -use crate::aggregation::tests::aggregation_tests_utils::{ - delete_exp, delete_field, get_decimal_field, get_duration_field, init_input_schema, - init_processor, insert_exp, insert_field, update_exp, update_field, FIELD_0_FLOAT, FIELD_0_INT, - FIELD_100_FLOAT, FIELD_100_INT, FIELD_100_UINT, FIELD_150_FLOAT, FIELD_150_INT, FIELD_150_UINT, - FIELD_200_FLOAT, FIELD_200_INT, FIELD_200_UINT, FIELD_250_FLOAT, FIELD_250_INT, FIELD_250_UINT, - FIELD_350_FLOAT, FIELD_350_INT, FIELD_350_UINT, FIELD_50_FLOAT, FIELD_50_INT, FIELD_50_UINT, - FIELD_NULL, ITALY, SINGAPORE, -}; -use crate::output; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_types::types::FieldType::{Decimal, Duration, Float, Int, UInt}; -use std::collections::HashMap; - -#[test] -fn test_sum_aggregation_float() { - let schema = init_input_schema(Float, "SUM"); - let mut processor = init_processor( - "SELECT Country, SUM(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - SUM = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_FLOAT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - SUM = 200.0 - */ - inp = insert_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_FLOAT, FIELD_200_FLOAT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - SUM = 200.0 - - Singapore, 50.0 - --------------- - SUM = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_50_FLOAT)]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - SUM = 250.0 - */ - inp = update_field(SINGAPORE, ITALY, FIELD_50_FLOAT, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, FIELD_50_FLOAT), - update_exp(ITALY, ITALY, FIELD_200_FLOAT, FIELD_250_FLOAT), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - SUM = 350.0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_FLOAT, FIELD_200_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_250_FLOAT, FIELD_350_FLOAT)]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - SUM = 150.0 - */ - inp = delete_field(ITALY, FIELD_200_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_350_FLOAT, FIELD_150_FLOAT)]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - SUM = 100.0 - */ - inp = delete_field(ITALY, FIELD_50_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_150_FLOAT, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - SUM = 0.0 - */ - inp = delete_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_100_FLOAT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_sum_aggregation_int() { - let schema = init_input_schema(Int, "SUM"); - let mut processor = init_processor( - "SELECT Country, SUM(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - SUM = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_INT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_100_INT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - SUM = 200.0 - */ - inp = insert_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_INT, FIELD_200_INT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - SUM = 200.0 - - Singapore, 50.0 - --------------- - SUM = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_50_INT)]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - SUM = 250.0 - */ - inp = update_field(SINGAPORE, ITALY, FIELD_50_INT, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, FIELD_50_INT), - update_exp(ITALY, ITALY, FIELD_200_INT, FIELD_250_INT), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - SUM = 350.0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_INT, FIELD_200_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_250_INT, FIELD_350_INT)]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - SUM = 150.0 - */ - inp = delete_field(ITALY, FIELD_200_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_350_INT, FIELD_150_INT)]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - SUM = 100.0 - */ - inp = delete_field(ITALY, FIELD_50_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_150_INT, FIELD_100_INT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - SUM = 0.0 - */ - inp = delete_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_100_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_sum_aggregation_uint() { - let schema = init_input_schema(UInt, "SUM"); - let mut processor = init_processor( - "SELECT Country, SUM(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - SUM = 100.0 - */ - let mut inp = insert_field(ITALY, FIELD_100_UINT); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_100_UINT)]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - SUM = 200.0 - */ - inp = insert_field(ITALY, FIELD_100_UINT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_UINT, FIELD_200_UINT)]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - SUM = 200.0 - - Singapore, 50.0 - --------------- - SUM = 50.0 - */ - inp = insert_field(SINGAPORE, FIELD_50_UINT); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, FIELD_50_UINT)]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - SUM = 250.0 - */ - inp = update_field(SINGAPORE, ITALY, FIELD_50_UINT, FIELD_50_UINT); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, FIELD_50_UINT), - update_exp(ITALY, ITALY, FIELD_200_UINT, FIELD_250_UINT), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - SUM = 350.0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_UINT, FIELD_200_UINT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_250_UINT, FIELD_350_UINT)]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - SUM = 150.0 - */ - inp = delete_field(ITALY, FIELD_200_UINT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_350_UINT, FIELD_150_UINT)]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - SUM = 100.0 - */ - inp = delete_field(ITALY, FIELD_50_UINT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_150_UINT, FIELD_100_UINT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - SUM = 0.0 - */ - inp = delete_field(ITALY, FIELD_100_UINT); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_100_UINT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_sum_aggregation_decimal() { - let schema = init_input_schema(Decimal, "SUM"); - let mut processor = init_processor( - "SELECT Country, SUM(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - SUM = 100.0 - */ - let mut inp = insert_field(ITALY, &get_decimal_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_decimal_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - SUM = 200.0 - */ - inp = insert_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(200), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - SUM = 200.0 - - Singapore, 50.0 - --------------- - SUM = 50.0 - */ - inp = insert_field(SINGAPORE, &get_decimal_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_decimal_field(50))]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - SUM = 250.0 - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_decimal_field(50), - &get_decimal_field(50), - ); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, &get_decimal_field(50)), - update_exp( - ITALY, - ITALY, - &get_decimal_field(200), - &get_decimal_field(250), - ), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - SUM = 350.0 - */ - inp = update_field( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(200), - ); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(250), - &get_decimal_field(350), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - SUM = 150.0 - */ - inp = delete_field(ITALY, &get_decimal_field(200)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(350), - &get_decimal_field(150), - )]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - SUM = 100.0 - */ - inp = delete_field(ITALY, &get_decimal_field(50)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(150), - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - SUM = 0.0 - */ - inp = delete_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_decimal_field(100))]; - assert_eq!(out, exp); -} - -#[test] -fn test_sum_aggregation_duration() { - let schema = init_input_schema(Duration, "SUM"); - let mut processor = init_processor( - "SELECT Country, SUM(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert 100 for segment Italy - /* - Italy, 100.0 - ------------- - SUM = 100.0 - */ - let mut inp = insert_field(ITALY, &get_duration_field(100)); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_duration_field(100))]; - assert_eq!(out, exp); - - // Insert another 100 for segment Italy - /* - Italy, 100.0 - Italy, 100.0 - ------------- - SUM = 200.0 - */ - inp = insert_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_field(200), - )]; - assert_eq!(out, exp); - - // Insert 50 for segment Singapore - /* - Italy, 100.0 - Italy, 100.0 - ------------- - SUM = 200.0 - - Singapore, 50.0 - --------------- - SUM = 50.0 - */ - inp = insert_field(SINGAPORE, &get_duration_field(50)); - out = output!(processor, inp); - exp = vec![insert_exp(SINGAPORE, &get_duration_field(50))]; - assert_eq!(out, exp); - - // Update Singapore segment to Italy - /* - Italy, 100.0 - Italy, 100.0 - Italy, 50.0 - ------------- - SUM = 250.0 - */ - inp = update_field( - SINGAPORE, - ITALY, - &get_duration_field(50), - &get_duration_field(50), - ); - out = output!(processor, inp); - exp = vec![ - delete_exp(SINGAPORE, &get_duration_field(50)), - update_exp( - ITALY, - ITALY, - &get_duration_field(200), - &get_duration_field(250), - ), - ]; - assert_eq!(out, exp); - - // Update Italy value 100 -> 200 - /* - Italy, 200.0 - Italy, 100.0 - Italy, 50.0 - ------------- - SUM = 350.0 - */ - inp = update_field( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_field(200), - ); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(250), - &get_duration_field(350), - )]; - assert_eq!(out, exp); - - // Delete 1 record (200) - /* - Italy, 100.0 - Italy, 50.0 - ------------- - SUM = 150.0 - */ - inp = delete_field(ITALY, &get_duration_field(200)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(350), - &get_duration_field(150), - )]; - assert_eq!(out, exp); - - // Delete another record (50) - /* - Italy, 100.0 - ------------- - SUM = 100.0 - */ - inp = delete_field(ITALY, &get_duration_field(50)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(150), - &get_duration_field(100), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - SUM = 0.0 - */ - inp = delete_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_duration_field(100))]; - assert_eq!(out, exp); -} - -#[test] -fn test_sum_aggregation_int_null() { - let schema = init_input_schema(Int, "SUM"); - let mut processor = init_processor( - "SELECT Country, SUM(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - SUM = 0 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_0_INT)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - SUM = 100 - */ - inp = insert_field(ITALY, FIELD_100_INT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_0_INT, FIELD_100_INT)]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - SUM = 0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_INT, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_INT, FIELD_0_INT)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - SUM = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_0_INT, FIELD_0_INT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - SUM = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_0_INT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_sum_aggregation_float_null() { - let schema = init_input_schema(Float, "SUM"); - let mut processor = init_processor( - "SELECT Country, SUM(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - SUM = 0 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, FIELD_0_FLOAT)]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - SUM = 100 - */ - inp = insert_field(ITALY, FIELD_100_FLOAT); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_0_FLOAT, FIELD_100_FLOAT)]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - SUM = 0 - */ - inp = update_field(ITALY, ITALY, FIELD_100_FLOAT, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_100_FLOAT, FIELD_0_FLOAT)]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - SUM = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp(ITALY, ITALY, FIELD_0_FLOAT, FIELD_0_FLOAT)]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - SUM = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, FIELD_0_FLOAT)]; - assert_eq!(out, exp); -} - -#[test] -fn test_sum_aggregation_decimal_null() { - let schema = init_input_schema(Decimal, "SUM"); - let mut processor = init_processor( - "SELECT Country, SUM(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - SUM = 0 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_decimal_field(0))]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - SUM = 100 - */ - inp = insert_field(ITALY, &get_decimal_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(0), - &get_decimal_field(100), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - SUM = 0 - */ - inp = update_field(ITALY, ITALY, &get_decimal_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(100), - &get_decimal_field(0), - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - SUM = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_decimal_field(0), - &get_decimal_field(0), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - SUM = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_decimal_field(0))]; - assert_eq!(out, exp); -} - -#[test] -fn test_sum_aggregation_duration_null() { - let schema = init_input_schema(Duration, "SUM"); - let mut processor = init_processor( - "SELECT Country, SUM(Salary) \ - FROM Users \ - WHERE Salary >= 1 GROUP BY Country", - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - ) - .unwrap(); - - // Insert NULL for segment Italy - /* - Italy, NULL - ------------- - SUM = 0 - */ - let mut inp = insert_field(ITALY, FIELD_NULL); - let mut out = output!(processor, inp); - let mut exp = vec![insert_exp(ITALY, &get_duration_field(0))]; - assert_eq!(out, exp); - - // Insert 100 for segment Italy - /* - Italy, NULL - Italy, 100 - ------------- - SUM = 100 - */ - inp = insert_field(ITALY, &get_duration_field(100)); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(0), - &get_duration_field(100), - )]; - assert_eq!(out, exp); - - // Update 100 for segment Italy to NULL - /* - Italy, NULL - Italy, NULL - ------------- - SUM = 0 - */ - inp = update_field(ITALY, ITALY, &get_duration_field(100), FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(100), - &get_duration_field(0), - )]; - assert_eq!(out, exp); - - // Delete a record - /* - Italy, NULL - ------------- - SUM = 0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![update_exp( - ITALY, - ITALY, - &get_duration_field(0), - &get_duration_field(0), - )]; - assert_eq!(out, exp); - - // Delete last record - /* - ------------- - SUM = 0.0 - */ - inp = delete_field(ITALY, FIELD_NULL); - out = output!(processor, inp); - exp = vec![delete_exp(ITALY, &get_duration_field(0))]; - assert_eq!(out, exp); -} diff --git a/dozer-sql/src/aggregation/tests/aggregation_test_planner.rs b/dozer-sql/src/aggregation/tests/aggregation_test_planner.rs deleted file mode 100644 index cc2207ea91..0000000000 --- a/dozer-sql/src/aggregation/tests/aggregation_test_planner.rs +++ /dev/null @@ -1,115 +0,0 @@ -use crate::planner::projection::CommonPlanner; -use crate::tests::utils::get_select; -use crate::{aggregation::processor::AggregationProcessor, tests::utils::create_test_runtime}; -use dozer_types::types::{ - Field, FieldDefinition, FieldType, Operation, Record, Schema, SourceDefinition, -}; - -#[test] -fn test_planner_with_aggregator() { - let sql = "SELECT CONCAT(city,'/',country), CONCAT('Total: ', CAST(SUM(adults_count + children_count) AS STRING), ' people') as headcounts GROUP BY CONCAT(city,'/',country)"; - let schema = Schema::default() - .field( - FieldDefinition::new( - "household_name".to_string(), - FieldType::String, - false, - SourceDefinition::Table { - name: "households".to_string(), - connection: "test".to_string(), - }, - ), - false, - ) - .field( - FieldDefinition::new( - "city".to_string(), - FieldType::String, - false, - SourceDefinition::Table { - name: "households".to_string(), - connection: "test".to_string(), - }, - ), - false, - ) - .field( - FieldDefinition::new( - "country".to_string(), - FieldType::String, - false, - SourceDefinition::Table { - name: "households".to_string(), - connection: "test".to_string(), - }, - ), - false, - ) - .field( - FieldDefinition::new( - "adults_count".to_string(), - FieldType::Int, - false, - SourceDefinition::Table { - name: "households".to_string(), - connection: "test".to_string(), - }, - ), - false, - ) - .field( - FieldDefinition::new( - "children_count".to_string(), - FieldType::Int, - false, - SourceDefinition::Table { - name: "households".to_string(), - connection: "test".to_string(), - }, - ), - false, - ) - .clone(); - - let runtime = create_test_runtime(); - let mut projection_planner = CommonPlanner::new(schema.clone(), &[], runtime.clone()); - let statement = get_select(sql).unwrap(); - - runtime - .block_on(projection_planner.plan( - statement.projection, - statement.group_by, - statement.having, - )) - .unwrap(); - - let mut processor = AggregationProcessor::new( - "".to_string(), - projection_planner.groupby, - projection_planner.aggregation_output, - projection_planner.projection_output, - projection_planner.having, - schema, - projection_planner.post_aggregation_schema, - false, - ) - .unwrap(); - - let rec = Record::new(vec![ - Field::String("John Smith".to_string()), - Field::String("Johor".to_string()), - Field::String("Malaysia".to_string()), - Field::Int(2), - Field::Int(1), - ]); - let _r = processor.aggregate(Operation::Insert { new: rec }).unwrap(); - - let rec = Record::new(vec![ - Field::String("Todd Enton".to_string()), - Field::String("Johor".to_string()), - Field::String("Malaysia".to_string()), - Field::Int(2), - Field::Int(1), - ]); - let _r = processor.aggregate(Operation::Insert { new: rec }).unwrap(); -} diff --git a/dozer-sql/src/aggregation/tests/aggregation_tests_utils.rs b/dozer-sql/src/aggregation/tests/aggregation_tests_utils.rs deleted file mode 100644 index 4a81283f33..0000000000 --- a/dozer-sql/src/aggregation/tests/aggregation_tests_utils.rs +++ /dev/null @@ -1,310 +0,0 @@ -use dozer_core::{node::PortHandle, DEFAULT_PORT_HANDLE}; -use dozer_types::types::{ - DozerDuration, Field, FieldDefinition, FieldType, Operation, Record, Schema, SourceDefinition, - TimeUnit, DATE_FORMAT, -}; -use std::collections::HashMap; - -use crate::aggregation::processor::AggregationProcessor; -use crate::errors::PipelineError; -use crate::planner::projection::CommonPlanner; -use crate::tests::utils::{create_test_runtime, get_select}; -use dozer_types::arrow::datatypes::ArrowNativeTypeOp; -use dozer_types::chrono::{DateTime, NaiveDate, TimeZone, Utc}; -use dozer_types::ordered_float::OrderedFloat; -use dozer_types::rust_decimal::Decimal; -use std::ops::Div; - -pub(crate) fn init_processor( - sql: &str, - input_schemas: HashMap, -) -> Result { - let input_schema = input_schemas - .get(&DEFAULT_PORT_HANDLE) - .unwrap_or_else(|| panic!("Error getting Input Schema")); - - let runtime = create_test_runtime(); - let mut projection_planner = CommonPlanner::new(input_schema.clone(), &[], runtime.clone()); - let statement = get_select(sql).unwrap(); - - runtime - .block_on(projection_planner.plan( - statement.projection, - statement.group_by, - statement.having, - )) - .unwrap(); - - let processor = AggregationProcessor::new( - "".to_string(), - projection_planner.groupby, - projection_planner.aggregation_output, - projection_planner.projection_output, - projection_planner.having, - input_schema.clone(), - projection_planner.post_aggregation_schema, - false, - ) - .unwrap_or_else(|e| panic!("{}", e.to_string())); - - Ok(processor) -} - -pub(crate) fn init_input_schema(field_type: FieldType, aggregator_name: &str) -> Schema { - Schema::default() - .field( - FieldDefinition::new( - String::from("ID"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("Country"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("Salary"), - field_type, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - format!("{aggregator_name}(Salary)"), - field_type, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone() -} - -pub(crate) fn init_val_input_schema(field_type: FieldType, aggregator_name: &str) -> Schema { - Schema::default() - .field( - FieldDefinition::new( - String::from("ID"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("Country"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("Salary"), - field_type, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - format!("{aggregator_name}(Salary, Country)"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone() -} - -pub(crate) fn insert_field(country: &str, insert_field: &Field) -> Operation { - let rec = Record::new(vec![ - Field::Int(0), - Field::String(country.to_string()), - insert_field.clone(), - insert_field.clone(), - ]); - Operation::Insert { new: rec } -} - -pub(crate) fn delete_field(country: &str, deleted_field: &Field) -> Operation { - let rec = Record::new(vec![ - Field::Int(0), - Field::String(country.to_string()), - deleted_field.clone(), - deleted_field.clone(), - ]); - Operation::Delete { old: rec } -} - -pub(crate) fn update_field( - old_country: &str, - new_country: &str, - old: &Field, - new: &Field, -) -> Operation { - let old_rec = Record::new(vec![ - Field::Int(0), - Field::String(old_country.to_string()), - old.clone(), - old.clone(), - ]); - - let new_rec = Record::new(vec![ - Field::Int(0), - Field::String(new_country.to_string()), - new.clone(), - new.clone(), - ]); - Operation::Update { - old: old_rec, - new: new_rec, - } -} - -pub(crate) fn insert_val_exp(inserted_field: &Field) -> Operation { - let rec = Record::new(vec![inserted_field.clone()]); - Operation::Insert { new: rec } -} - -pub(crate) fn delete_val_exp(deleted_field: &Field) -> Operation { - let rec = Record::new(vec![deleted_field.clone()]); - Operation::Delete { old: rec } -} - -pub(crate) fn update_val_exp(old: &Field, new: &Field) -> Operation { - let old_rec = Record::new(vec![old.clone()]); - let new_rec = Record::new(vec![new.clone()]); - Operation::Update { - old: old_rec, - new: new_rec, - } -} - -pub(crate) fn insert_exp(country: &str, inserted_field: &Field) -> Operation { - let rec = Record::new(vec![ - Field::String(country.to_string()), - inserted_field.clone(), - ]); - Operation::Insert { new: rec } -} - -pub(crate) fn delete_exp(country: &str, deleted_field: &Field) -> Operation { - let rec = Record::new(vec![ - Field::String(country.to_string()), - deleted_field.clone(), - ]); - Operation::Delete { old: rec } -} - -pub(crate) fn update_exp( - old_country: &str, - new_country: &str, - old: &Field, - new: &Field, -) -> Operation { - let old_rec = Record::new(vec![Field::String(old_country.to_string()), old.clone()]); - let new_rec = Record::new(vec![Field::String(new_country.to_string()), new.clone()]); - Operation::Update { - old: old_rec, - new: new_rec, - } -} - -pub fn get_duration_field(val: u128) -> Field { - Field::Duration(DozerDuration( - std::time::Duration::from_nanos(val as u64), - TimeUnit::Nanoseconds, - )) -} - -pub fn get_duration_div_field(numerator: i128, denominator: i128) -> Field { - Field::Duration(DozerDuration( - std::time::Duration::from_nanos((numerator as u64).div_wrapping(denominator as u64)), - TimeUnit::Nanoseconds, - )) -} - -pub fn get_decimal_field(val: i64) -> Field { - Field::Decimal(Decimal::new(val, 0)) -} - -pub fn get_decimal_div_field(numerator: i64, denominator: i64) -> Field { - Field::Decimal(Decimal::new(numerator, 0).div(Decimal::new(denominator, 0))) -} - -pub fn get_ts_field(val: i64) -> Field { - Field::Timestamp(DateTime::from(Utc.timestamp_millis_opt(val).unwrap())) -} - -pub fn get_date_field(val: &str) -> Field { - Field::Date(NaiveDate::parse_from_str(val, DATE_FORMAT).unwrap()) -} - -#[macro_export] -macro_rules! output { - ($processor:expr, $inp:expr) => { - $processor - .aggregate($inp) - .unwrap_or_else(|_e| panic!("Error executing aggregate")) - }; -} - -pub const ITALY: &str = "Italy"; -pub const SINGAPORE: &str = "Singapore"; - -pub const DATE4: &str = "2015-10-04"; -pub const DATE8: &str = "2015-10-08"; -pub const DATE16: &str = "2015-10-16"; - -pub const FIELD_NULL: &Field = &Field::Null; - -pub const FIELD_0_FLOAT: &Field = &Field::Float(OrderedFloat(0.0)); -pub const FIELD_100_FLOAT: &Field = &Field::Float(OrderedFloat(100.0)); -pub const FIELD_150_FLOAT: &Field = &Field::Float(OrderedFloat(150.0)); -pub const FIELD_200_FLOAT: &Field = &Field::Float(OrderedFloat(200.0)); -pub const FIELD_250_FLOAT: &Field = &Field::Float(OrderedFloat(250.0)); -pub const FIELD_350_FLOAT: &Field = &Field::Float(OrderedFloat(350.0)); -pub const FIELD_75_FLOAT: &Field = &Field::Float(OrderedFloat(75.0)); -pub const FIELD_50_FLOAT: &Field = &Field::Float(OrderedFloat(50.0)); -pub const FIELD_250_DIV_3_FLOAT: &Field = &Field::Float(OrderedFloat(250.0 / 3.0)); -pub const FIELD_350_DIV_3_FLOAT: &Field = &Field::Float(OrderedFloat(350.0 / 3.0)); - -pub const FIELD_0_INT: &Field = &Field::Int(0); -pub const FIELD_1_INT: &Field = &Field::Int(1); -pub const FIELD_2_INT: &Field = &Field::Int(2); -pub const FIELD_3_INT: &Field = &Field::Int(3); -pub const FIELD_100_INT: &Field = &Field::Int(100); -pub const FIELD_150_INT: &Field = &Field::Int(150); -pub const FIELD_200_INT: &Field = &Field::Int(200); -pub const FIELD_250_INT: &Field = &Field::Int(250); -pub const FIELD_300_INT: &Field = &Field::Int(300); -pub const FIELD_350_INT: &Field = &Field::Int(350); -pub const FIELD_400_INT: &Field = &Field::Int(400); -pub const FIELD_500_INT: &Field = &Field::Int(500); -pub const FIELD_600_INT: &Field = &Field::Int(600); -pub const FIELD_50_INT: &Field = &Field::Int(50); -pub const FIELD_75_INT: &Field = &Field::Int(75); - -pub const FIELD_100_UINT: &Field = &Field::UInt(100); -pub const FIELD_150_UINT: &Field = &Field::UInt(150); -pub const FIELD_200_UINT: &Field = &Field::UInt(200); -pub const FIELD_250_UINT: &Field = &Field::UInt(250); -pub const FIELD_350_UINT: &Field = &Field::UInt(350); -pub const FIELD_50_UINT: &Field = &Field::UInt(50); -pub const FIELD_75_UINT: &Field = &Field::UInt(75); diff --git a/dozer-sql/src/aggregation/tests/mod.rs b/dozer-sql/src/aggregation/tests/mod.rs deleted file mode 100644 index 9d5eafa52c..0000000000 --- a/dozer-sql/src/aggregation/tests/mod.rs +++ /dev/null @@ -1,27 +0,0 @@ -#[cfg(test)] -mod aggregation_avg_tests; -#[cfg(test)] -mod aggregation_count_tests; -#[cfg(test)] -mod aggregation_having_tests; -#[cfg(test)] -mod aggregation_max_tests; -#[cfg(test)] -mod aggregation_max_value_tests; -#[cfg(test)] -mod aggregation_min_tests; -#[cfg(test)] -mod aggregation_min_value_tests; -#[cfg(test)] -mod aggregation_null; -#[cfg(test)] -mod aggregation_sum_tests; -#[cfg(test)] -mod aggregation_test_planner; -#[cfg(test)] -mod aggregation_tests_utils; - -#[cfg(test)] -mod aggregation_max_append_only_tests; -#[cfg(test)] -mod aggregation_min_append_only_tests; diff --git a/dozer-sql/src/builder/common.rs b/dozer-sql/src/builder/common.rs deleted file mode 100644 index 0d24256a15..0000000000 --- a/dozer-sql/src/builder/common.rs +++ /dev/null @@ -1,79 +0,0 @@ -use dozer_sql_expression::{ - builder::{ExpressionBuilder, NameOrAlias}, - sqlparser::ast::{ObjectName, TableFactor}, -}; - -use crate::errors::{PipelineError, ProductError}; - -use super::QueryContext; - -pub fn is_an_entry_point(name: &str, query_context: &QueryContext, pipeline_idx: usize) -> bool { - if query_context - .pipeline_map - .contains_key(&(pipeline_idx, name.to_owned())) - { - return false; - } - if query_context.processors_list.contains(&name.to_owned()) { - return false; - } - true -} - -pub fn is_a_pipeline_output( - name: &str, - query_context: &mut QueryContext, - pipeline_idx: usize, -) -> bool { - if query_context - .pipeline_map - .contains_key(&(pipeline_idx, name.to_owned())) - { - return true; - } - false -} - -pub fn get_name_or_alias(relation: &TableFactor) -> Result { - match relation { - TableFactor::Table { name, alias, .. } => { - let table_name = string_from_sql_object_name(name); - if let Some(table_alias) = alias { - let alias = table_alias.name.value.clone(); - return Ok(NameOrAlias(table_name, Some(alias))); - } - Ok(NameOrAlias(table_name, None)) - } - TableFactor::Derived { alias, .. } => { - if let Some(table_alias) = alias { - let alias = table_alias.name.value.clone(); - return Ok(NameOrAlias("dozer_derived".to_string(), Some(alias))); - } - Ok(NameOrAlias("dozer_derived".to_string(), None)) - } - TableFactor::TableFunction { .. } => Err(PipelineError::ProductError( - ProductError::UnsupportedTableFunction, - )), - TableFactor::UNNEST { .. } => { - Err(PipelineError::ProductError(ProductError::UnsupportedUnnest)) - } - TableFactor::NestedJoin { alias, .. } => { - if let Some(table_alias) = alias { - let alias = table_alias.name.value.clone(); - return Ok(NameOrAlias("dozer_nested".to_string(), Some(alias))); - } - Ok(NameOrAlias("dozer_nested".to_string(), None)) - } - TableFactor::Pivot { .. } => { - Err(PipelineError::ProductError(ProductError::UnsupportedPivot)) - } - } -} - -pub fn string_from_sql_object_name(name: &ObjectName) -> String { - name.0 - .iter() - .map(ExpressionBuilder::normalize_ident) - .collect::>() - .join(".") -} diff --git a/dozer-sql/src/builder/from.rs b/dozer-sql/src/builder/from.rs deleted file mode 100644 index 6fa98bea03..0000000000 --- a/dozer-sql/src/builder/from.rs +++ /dev/null @@ -1,122 +0,0 @@ -use dozer_core::{ - app::{AppPipeline, PipelineEntryPoint}, - DEFAULT_PORT_HANDLE, -}; -use dozer_sql_expression::sqlparser::ast::{TableFactor, TableWithJoins}; - -use crate::{ - builder::{get_from_source, QueryContext}, - errors::PipelineError, - product::table::factory::TableProcessorFactory, -}; - -use super::{ - common::{get_name_or_alias, is_an_entry_point}, - join::insert_join_to_pipeline, - table_operator::{insert_table_operator_processor_to_pipeline, is_table_operator}, - ConnectionInfo, -}; - -pub fn insert_from_to_pipeline( - from: TableWithJoins, - pipeline: &mut AppPipeline, - pipeline_idx: usize, - query_context: &mut QueryContext, -) -> Result { - if from.joins.is_empty() { - insert_table_to_pipeline(from.relation, pipeline, pipeline_idx, query_context) - } else { - insert_join_to_pipeline(from, pipeline, pipeline_idx, query_context) - } -} - -fn insert_table_to_pipeline( - relation: TableFactor, - pipeline: &mut AppPipeline, - pipeline_idx: usize, - query_context: &mut QueryContext, -) -> Result { - if let Some(operator) = is_table_operator(&relation)? { - let product_processor_name = - insert_from_processor_to_pipeline(query_context, relation, pipeline)?; - - let connection_info = insert_table_operator_processor_to_pipeline( - operator, - pipeline, - pipeline_idx, - query_context, - )?; - - pipeline.connect_nodes( - connection_info.output_node.0, - connection_info.output_node.1, - product_processor_name.clone(), - DEFAULT_PORT_HANDLE, - ); - - Ok(ConnectionInfo { - input_nodes: connection_info.input_nodes, - output_node: (product_processor_name, DEFAULT_PORT_HANDLE), - }) - } else { - insert_table_processor_to_pipeline(relation, pipeline, pipeline_idx, query_context) - } -} - -fn insert_table_processor_to_pipeline( - relation: TableFactor, - pipeline: &mut AppPipeline, - pipeline_idx: usize, - query_context: &mut QueryContext, -) -> Result { - let relation_name_or_alias = get_name_or_alias(&relation)?; - let product_input_name = get_from_source(relation, pipeline, query_context, pipeline_idx)?.0; - - let processor_name = format!( - "from:{}--{}", - product_input_name, - query_context.get_next_processor_id() - ); - if !query_context.processors_list.insert(processor_name.clone()) { - return Err(PipelineError::ProcessorAlreadyExists(processor_name)); - } - let product_processor_factory = - TableProcessorFactory::new(processor_name.clone(), relation_name_or_alias.clone()); - pipeline.add_processor(Box::new(product_processor_factory), processor_name.clone()); - - // is a node that is an entry point to the pipeline - let input_nodes = if is_an_entry_point(&product_input_name, query_context, pipeline_idx) { - let entry_point = PipelineEntryPoint::new(product_input_name.clone(), DEFAULT_PORT_HANDLE); - pipeline.add_entry_point(processor_name.clone(), entry_point); - query_context.used_sources.push(product_input_name); - vec![] - } - // is a node that is connected to another pipeline - else { - vec![( - product_input_name, - processor_name.clone(), - DEFAULT_PORT_HANDLE, - )] - }; - - Ok(ConnectionInfo { - input_nodes, - output_node: (processor_name, DEFAULT_PORT_HANDLE), - }) -} - -fn insert_from_processor_to_pipeline( - query_context: &mut QueryContext, - relation: TableFactor, - pipeline: &mut AppPipeline, -) -> Result { - let product_processor_name = format!("from--{}", query_context.get_next_processor_id()); - let product_processor = TableProcessorFactory::new( - product_processor_name.clone(), - get_name_or_alias(&relation)?, - ); - - pipeline.add_processor(Box::new(product_processor), product_processor_name.clone()); - Ok(product_processor_name) -} diff --git a/dozer-sql/src/builder/join.rs b/dozer-sql/src/builder/join.rs deleted file mode 100644 index 653010d927..0000000000 --- a/dozer-sql/src/builder/join.rs +++ /dev/null @@ -1,166 +0,0 @@ -use dozer_core::{ - app::{AppPipeline, PipelineEntryPoint}, - node::PortHandle, - DEFAULT_PORT_HANDLE, -}; -use dozer_sql_expression::sqlparser::ast::{TableFactor, TableWithJoins}; - -use crate::{ - builder::{get_from_source, QueryContext}, - errors::PipelineError, - product::join::factory::{JoinProcessorFactory, LEFT_JOIN_PORT, RIGHT_JOIN_PORT}, -}; - -use super::{ - common::{get_name_or_alias, is_an_entry_point}, - table_operator::{insert_table_operator_processor_to_pipeline, is_table_operator}, - ConnectionInfo, -}; - -#[derive(Clone, Debug)] -enum JoinSource { - Table(String), - Operator(ConnectionInfo), - Join(ConnectionInfo), -} - -pub fn insert_join_to_pipeline( - from: TableWithJoins, - pipeline: &mut AppPipeline, - pipeline_idx: usize, - query_context: &mut QueryContext, -) -> Result { - let mut input_nodes = vec![]; - - let left_table = from.relation; - let mut left_name_or_alias = Some(get_name_or_alias(&left_table)?); - let mut left_join_source = - insert_join_source_to_pipeline(left_table, pipeline, pipeline_idx, query_context)?; - - for join in from.joins { - let right_table = join.relation; - let right_name_or_alias = Some(get_name_or_alias(&right_table)?); - let right_join_source = insert_join_source_to_pipeline( - right_table.clone(), - pipeline, - pipeline_idx, - query_context, - )?; - - let join_processor_name = format!("join_{}", query_context.get_next_processor_id()); - if !query_context - .processors_list - .insert(join_processor_name.clone()) - { - return Err(PipelineError::ProcessorAlreadyExists(join_processor_name)); - } - let join_processor_factory = JoinProcessorFactory::new( - join_processor_name.clone(), - left_name_or_alias, - right_name_or_alias, - join.join_operator, - pipeline - .flags() - .enable_probabilistic_optimizations - .in_joins - .unwrap_or(false), - ); - pipeline.add_processor( - Box::new(join_processor_factory), - join_processor_name.clone(), - ); - - input_nodes.extend(modify_pipeline_graph( - left_join_source, - join_processor_name.clone(), - LEFT_JOIN_PORT, - pipeline, - pipeline_idx, - query_context, - )); - input_nodes.extend(modify_pipeline_graph( - right_join_source, - join_processor_name.clone(), - RIGHT_JOIN_PORT, - pipeline, - pipeline_idx, - query_context, - )); - - // TODO: refactor join source name and aliasing logic - left_name_or_alias = None; - left_join_source = JoinSource::Join(ConnectionInfo { - input_nodes: input_nodes.clone(), - output_node: (join_processor_name, DEFAULT_PORT_HANDLE), - }); - } - - match left_join_source { - JoinSource::Table(_) | JoinSource::Operator(_) => Err(PipelineError::InvalidJoin( - "No JOIN operator found".to_string(), - )), - JoinSource::Join(connection_info) => Ok(connection_info), - } -} - -// TODO: refactor this -fn insert_join_source_to_pipeline( - source: TableFactor, - pipeline: &mut AppPipeline, - pipeline_idx: usize, - query_context: &mut QueryContext, -) -> Result { - let join_source = if let Some(table_operator) = is_table_operator(&source)? { - let connection_info = insert_table_operator_processor_to_pipeline( - table_operator, - pipeline, - pipeline_idx, - query_context, - )?; - JoinSource::Operator(connection_info) - } else if is_nested_join(&source) { - return Err(PipelineError::InvalidJoin( - "Nested JOINs are not supported".to_string(), - )); - } else { - let name_or_alias = get_from_source(source, pipeline, query_context, pipeline_idx)?; - JoinSource::Table(name_or_alias.0) - }; - Ok(join_source) -} - -fn is_nested_join(left_table: &TableFactor) -> bool { - matches!(left_table, TableFactor::NestedJoin { .. }) -} - -/// Returns the input node if there is one -fn modify_pipeline_graph( - source: JoinSource, - id: String, - port: PortHandle, - pipeline: &mut AppPipeline, - pipeline_idx: usize, - query_context: &mut QueryContext, -) -> Option<(String, String, u16)> { - match source { - JoinSource::Table(source_table) => { - if is_an_entry_point(&source_table, query_context, pipeline_idx) { - let entry_point = PipelineEntryPoint::new(source_table.clone(), port); - pipeline.add_entry_point(id, entry_point); - query_context.used_sources.push(source_table); - None - } else { - Some((source_table, id, port)) - } - } - JoinSource::Operator(connection_info) | JoinSource::Join(connection_info) => { - pipeline.connect_nodes( - connection_info.output_node.0, - connection_info.output_node.1, - id, - port, - ); - None - } - } -} diff --git a/dozer-sql/src/builder/mod.rs b/dozer-sql/src/builder/mod.rs deleted file mode 100644 index 3ccf3ed356..0000000000 --- a/dozer-sql/src/builder/mod.rs +++ /dev/null @@ -1,570 +0,0 @@ -use crate::aggregation::factory::AggregationProcessorFactory; -use crate::builder::PipelineError::InvalidQuery; -use crate::errors::PipelineError; -use crate::selection::factory::SelectionProcessorFactory; -use dozer_core::app::AppPipeline; -use dozer_core::node::PortHandle; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_sql_expression::builder::{ExpressionBuilder, NameOrAlias}; -use dozer_sql_expression::sqlparser::ast::{SetOperator, SetQuantifier, TableFactor}; -use dozer_types::models::udf_config::UdfConfig; - -use dozer_sql_expression::sqlparser::{ - ast::{Query, Select, SetExpr, Statement}, - dialect::DozerDialect, - parser::Parser, -}; -use std::collections::HashMap; -use std::collections::HashSet; -use std::sync::Arc; -use tokio::runtime::Runtime; - -use super::errors::UnsupportedSqlError; - -use super::product::set::set_factory::SetProcessorFactory; - -#[derive(Debug, Clone)] -pub struct OutputNodeInfo { - // Name to connect in dag - pub node: String, - // Port to connect in dag - pub port: PortHandle, - // TODO add:indexes to the tables -} - -/// The struct contains some contexts during query to pipeline. -#[derive(Debug, Clone)] -pub struct QueryContext { - // Internal tables map, used to store the tables that are created by the queries - pipeline_map: HashMap<(usize, String), OutputNodeInfo>, - - // Output tables map that are marked with "INTO" used to store the tables, these can be exposed to sinks. - pub output_tables_map: HashMap, - - // Used Sources - pub used_sources: Vec, - - // Internal tables map, used to store the tables that are created by the queries - processors_list: HashSet, - - // Processors counter - processor_counter: usize, - - // Udf related configs - udfs: Vec, - - // The tokio runtime - runtime: Arc, -} - -impl QueryContext { - fn get_next_processor_id(&mut self) -> usize { - self.processor_counter += 1; - self.processor_counter - } - - pub fn new(udfs: Vec, runtime: Arc) -> Self { - QueryContext { - pipeline_map: Default::default(), - output_tables_map: Default::default(), - used_sources: Default::default(), - processors_list: Default::default(), - processor_counter: Default::default(), - udfs, - runtime, - } - } -} - -pub fn statement_to_pipeline( - sql: &str, - pipeline: &mut AppPipeline, - override_name: Option, - udfs: Vec, - runtime: Arc, -) -> Result { - let dialect = DozerDialect {}; - let mut ctx = QueryContext::new(udfs, runtime); - let is_top_select = true; - let ast = Parser::parse_sql(&dialect, sql) - .map_err(|err| PipelineError::InternalError(Box::new(err)))?; - let query_name = NameOrAlias(format!("query_{}", ctx.get_next_processor_id()), None); - - for (idx, statement) in ast.into_iter().enumerate() { - match statement { - Statement::Query(query) => { - query_to_pipeline( - TableInfo { - name: query_name.clone(), - override_name: override_name.clone(), - }, - *query, - pipeline, - &mut ctx, - idx, - is_top_select, - )?; - } - s => { - return Err(PipelineError::UnsupportedSqlError( - UnsupportedSqlError::GenericError(s.to_string()), - )) - } - } - } - - Ok(ctx) -} - -struct TableInfo { - name: NameOrAlias, - override_name: Option, -} - -fn query_to_pipeline( - table_info: TableInfo, - query: Query, - pipeline: &mut AppPipeline, - query_ctx: &mut QueryContext, - pipeline_idx: usize, - is_top_select: bool, -) -> Result<(), PipelineError> { - // return error if there is unsupported syntax - if !query.order_by.is_empty() { - return Err(PipelineError::UnsupportedSqlError( - UnsupportedSqlError::OrderByError, - )); - } - - if query.limit.is_some() || query.offset.is_some() { - return Err(PipelineError::UnsupportedSqlError( - UnsupportedSqlError::LimitOffsetError, - )); - } - - // Attach the first pipeline if there is with clause - if let Some(with) = query.with { - if with.recursive { - return Err(PipelineError::UnsupportedSqlError( - UnsupportedSqlError::Recursive, - )); - } - - for table in with.cte_tables { - if table.from.is_some() { - return Err(PipelineError::UnsupportedSqlError( - UnsupportedSqlError::CteFromError, - )); - } - let table_name = table.alias.name.to_string(); - if query_ctx - .pipeline_map - .contains_key(&(pipeline_idx, table_name.clone())) - { - return Err(InvalidQuery(format!( - "WITH query name {table_name:?} specified more than once" - ))); - } - query_to_pipeline( - TableInfo { - name: NameOrAlias(table_name.clone(), Some(table_name)), - override_name: None, - }, - *table.query, - pipeline, - query_ctx, - pipeline_idx, - false, //Inside a with clause, so not top select - )?; - } - }; - - match *query.body { - SetExpr::Select(select) => { - select_to_pipeline( - table_info, - *select, - pipeline, - query_ctx, - pipeline_idx, - is_top_select, - )?; - } - SetExpr::Query(query) => { - let query_name = format!("subquery_{}", query_ctx.get_next_processor_id()); - let mut ctx = QueryContext::new(query_ctx.udfs.clone(), query_ctx.runtime.clone()); - query_to_pipeline( - TableInfo { - name: NameOrAlias(query_name, None), - override_name: None, - }, - *query, - pipeline, - &mut ctx, - pipeline_idx, - false, //Inside a subquery, so not top select - )? - } - SetExpr::SetOperation { - op, - set_quantifier, - left, - right, - } => match op { - SetOperator::Union => { - set_to_pipeline( - table_info, - left, - right, - set_quantifier, - pipeline, - query_ctx, - pipeline_idx, - is_top_select, - )?; - } - _ => return Err(PipelineError::InvalidOperator(op.to_string())), - }, - _ => { - return Err(PipelineError::UnsupportedSqlError( - UnsupportedSqlError::GenericError("Unsupported query body structure".to_string()), - )) - } - }; - Ok(()) -} - -fn select_to_pipeline( - table_info: TableInfo, - select: Select, - pipeline: &mut AppPipeline, - query_ctx: &mut QueryContext, - pipeline_idx: usize, - is_top_select: bool, -) -> Result { - // FROM clause - let Some(from) = select.from.into_iter().next() else { - return Err(PipelineError::UnsupportedSqlError( - UnsupportedSqlError::FromCommaSyntax, - )); - }; - - let connection_info = from::insert_from_to_pipeline(from, pipeline, pipeline_idx, query_ctx)?; - - let input_nodes = connection_info.input_nodes; - let output_node = connection_info.output_node; - - let gen_agg_name = format!("agg--{}", query_ctx.get_next_processor_id()); - - let gen_selection_name = format!("select--{}", query_ctx.get_next_processor_id()); - let (gen_product_name, product_output_port) = output_node; - - for (source_name, processor_name, processor_port) in input_nodes { - if let Some(table_info) = query_ctx - .pipeline_map - .get(&(pipeline_idx, source_name.clone())) - { - pipeline.connect_nodes( - table_info.node.clone(), - table_info.port, - processor_name, - processor_port, - ); - // If not present in pipeline_map, insert into used_sources as this is coming from source - } else { - query_ctx.used_sources.push(source_name.clone()); - } - } - - let aggregation = AggregationProcessorFactory::new( - gen_agg_name.clone(), - select.projection, - select.group_by, - select.having, - pipeline - .flags() - .enable_probabilistic_optimizations - .in_aggregations - .unwrap_or(false), - query_ctx.udfs.clone(), - query_ctx.runtime.clone(), - ); - - pipeline.add_processor(Box::new(aggregation), gen_agg_name.clone()); - - // Where clause - if let Some(selection) = select.selection { - let selection = SelectionProcessorFactory::new( - gen_selection_name.clone(), - selection, - query_ctx.udfs.clone(), - query_ctx.runtime.clone(), - ); - - pipeline.add_processor(Box::new(selection), gen_selection_name.clone()); - - pipeline.connect_nodes( - gen_product_name, - product_output_port, - gen_selection_name.clone(), - DEFAULT_PORT_HANDLE, - ); - - pipeline.connect_nodes( - gen_selection_name, - DEFAULT_PORT_HANDLE, - gen_agg_name.clone(), - DEFAULT_PORT_HANDLE, - ); - } else { - pipeline.connect_nodes( - gen_product_name, - product_output_port, - gen_agg_name.clone(), - DEFAULT_PORT_HANDLE, - ); - } - - query_ctx.pipeline_map.insert( - (pipeline_idx, table_info.name.0.to_string()), - OutputNodeInfo { - node: gen_agg_name.clone(), - port: DEFAULT_PORT_HANDLE, - }, - ); - - let output_table_name = if let Some(into) = select.into { - Some(into.name.to_string()) - } else { - table_info.override_name.clone() - }; - - if is_top_select && output_table_name.is_none() { - return Err(PipelineError::MissingIntoClause); - } - - if let Some(table_name) = output_table_name { - if query_ctx.output_tables_map.contains_key(&table_name) { - return Err(PipelineError::DuplicateIntoClause(table_name)); - } - - query_ctx.output_tables_map.insert( - table_name, - OutputNodeInfo { - node: gen_agg_name.clone(), - port: DEFAULT_PORT_HANDLE, - }, - ); - } - - Ok(gen_agg_name) -} - -#[allow(clippy::too_many_arguments)] -fn set_to_pipeline( - table_info: TableInfo, - left_select: Box, - right_select: Box, - set_quantifier: SetQuantifier, - pipeline: &mut AppPipeline, - query_ctx: &mut QueryContext, - pipeline_idx: usize, - is_top_select: bool, -) -> Result { - let gen_left_set_name = format!("set_left_{}", query_ctx.get_next_processor_id()); - let left_table_info = TableInfo { - name: NameOrAlias(gen_left_set_name.clone(), None), - override_name: None, - }; - let gen_right_set_name = format!("set_right_{}", query_ctx.get_next_processor_id()); - let right_table_info = TableInfo { - name: NameOrAlias(gen_right_set_name.clone(), None), - override_name: None, - }; - - let _left_pipeline_name = match *left_select { - SetExpr::Select(select) => select_to_pipeline( - left_table_info, - *select, - pipeline, - query_ctx, - pipeline_idx, - is_top_select, - )?, - SetExpr::SetOperation { - op: _, - set_quantifier, - left, - right, - } => set_to_pipeline( - left_table_info, - left, - right, - set_quantifier, - pipeline, - query_ctx, - pipeline_idx, - is_top_select, - )?, - _ => { - return Err(PipelineError::InvalidQuery( - "Invalid UNION left Query".to_string(), - )) - } - }; - - let _right_pipeline_name = match *right_select { - SetExpr::Select(select) => select_to_pipeline( - right_table_info, - *select, - pipeline, - query_ctx, - pipeline_idx, - is_top_select, - )?, - SetExpr::SetOperation { - op: _, - set_quantifier, - left, - right, - } => set_to_pipeline( - right_table_info, - left, - right, - set_quantifier, - pipeline, - query_ctx, - pipeline_idx, - is_top_select, - )?, - _ => { - return Err(PipelineError::InvalidQuery( - "Invalid UNION right Query".to_string(), - )) - } - }; - - let mut gen_set_name = format!("set_{}", query_ctx.get_next_processor_id()); - - let left_pipeline_output_node = query_ctx - .pipeline_map - .get(&(pipeline_idx, gen_left_set_name)) - .ok_or_else(|| PipelineError::InvalidQuery("Invalid UNION left Query".to_string()))?; - - let right_pipeline_output_node = query_ctx - .pipeline_map - .get(&(pipeline_idx, gen_right_set_name)) - .ok_or_else(|| PipelineError::InvalidQuery("Invalid UNION right Query".to_string()))?; - - if table_info.override_name.is_some() { - gen_set_name = table_info.override_name.to_owned().unwrap(); - } - - let set_proc_fac = SetProcessorFactory::new( - gen_set_name.clone(), - set_quantifier, - pipeline - .flags() - .enable_probabilistic_optimizations - .in_sets - .unwrap_or(false), - ); - - pipeline.add_processor(Box::new(set_proc_fac), gen_set_name.clone()); - - pipeline.connect_nodes( - left_pipeline_output_node.node.clone(), - left_pipeline_output_node.port, - gen_set_name.clone(), - 0, - ); - - pipeline.connect_nodes( - right_pipeline_output_node.node.clone(), - right_pipeline_output_node.port, - gen_set_name.clone(), - 1, - ); - - for (_, table_name) in query_ctx.pipeline_map.keys() { - query_ctx.output_tables_map.remove_entry(table_name); - } - - query_ctx.pipeline_map.insert( - (pipeline_idx, table_info.name.0.to_string()), - OutputNodeInfo { - node: gen_set_name.clone(), - port: DEFAULT_PORT_HANDLE, - }, - ); - - Ok(gen_set_name) -} - -fn get_from_source( - relation: TableFactor, - pipeline: &mut AppPipeline, - query_ctx: &mut QueryContext, - pipeline_idx: usize, -) -> Result { - match relation { - TableFactor::Table { name, alias, .. } => { - let input_name = name - .0 - .iter() - .map(ExpressionBuilder::normalize_ident) - .collect::>() - .join("."); - let alias_name = alias - .as_ref() - .map(|a| ExpressionBuilder::fullname_from_ident(&[a.name.clone()])); - - Ok(NameOrAlias(input_name, alias_name)) - } - TableFactor::Derived { - lateral: _, - subquery, - alias, - } => { - let name = format!("derived_{}", query_ctx.get_next_processor_id()); - let alias_name = alias.as_ref().map(|alias_ident| { - ExpressionBuilder::fullname_from_ident(&[alias_ident.name.clone()]) - }); - let is_top_select = false; //inside FROM clause, so not top select - let name_or = NameOrAlias(name, alias_name); - query_to_pipeline( - TableInfo { - name: name_or.clone(), - override_name: None, - }, - *subquery, - pipeline, - query_ctx, - pipeline_idx, - is_top_select, - )?; - - Ok(name_or) - } - _ => Err(PipelineError::UnsupportedSqlError( - UnsupportedSqlError::JoinTable, - )), - } -} - -#[derive(Clone, Debug)] -struct ConnectionInfo { - input_nodes: Vec<(String, String, PortHandle)>, - output_node: (String, PortHandle), -} - -mod common; -mod from; -mod join; -mod table_operator; - -pub use common::string_from_sql_object_name; -pub use table_operator::{TableOperatorArg, TableOperatorDescriptor}; - -#[cfg(test)] -mod tests; diff --git a/dozer-sql/src/builder/table_operator.rs b/dozer-sql/src/builder/table_operator.rs deleted file mode 100644 index 3f1001e6c6..0000000000 --- a/dozer-sql/src/builder/table_operator.rs +++ /dev/null @@ -1,191 +0,0 @@ -use dozer_core::{ - app::{AppPipeline, PipelineEntryPoint}, - node::ProcessorFactory, - DEFAULT_PORT_HANDLE, -}; -use dozer_sql_expression::sqlparser::ast::{ - Expr, FunctionArg, FunctionArgExpr, ObjectName, TableFactor, -}; - -use crate::{ - errors::PipelineError, - table_operator::factory::{get_source_name, TableOperatorProcessorFactory}, - window::factory::WindowProcessorFactory, -}; - -use super::{ - common::{is_a_pipeline_output, is_an_entry_point, string_from_sql_object_name}, - ConnectionInfo, QueryContext, -}; - -#[derive(Clone, Debug)] -pub struct TableOperatorDescriptor { - pub name: String, - pub args: Vec, -} - -#[derive(Clone, Debug)] -pub enum TableOperatorArg { - Argument(FunctionArg), - Descriptor(TableOperatorDescriptor), -} - -pub fn is_table_operator( - relation: &TableFactor, -) -> Result, PipelineError> { - match relation { - TableFactor::Table { name, args, .. } => { - if args.is_none() { - return Ok(None); - } - let operator = get_table_operator_descriptor(name, args.as_deref())?; - - Ok(operator) - } - TableFactor::Derived { .. } => Ok(None), - TableFactor::TableFunction { .. } => Err(PipelineError::UnsupportedTableFunction), - TableFactor::UNNEST { .. } => Err(PipelineError::UnsupportedUnnest), - TableFactor::NestedJoin { .. } => Err(PipelineError::UnsupportedNestedJoin), - TableFactor::Pivot { .. } => Err(PipelineError::UnsupportedPivot), - } -} - -fn get_table_operator_descriptor( - name: &ObjectName, - args: Option<&[FunctionArg]>, -) -> Result, PipelineError> { - let mut operator_args = vec![]; - - if let Some(args) = args { - for arg in args { - let operator_arg = get_table_operator_arg(arg)?; - operator_args.push(operator_arg); - } - } - - Ok(Some(TableOperatorDescriptor { - name: string_from_sql_object_name(name), - args: operator_args, - })) -} - -fn get_table_operator_arg(arg: &FunctionArg) -> Result { - match arg { - FunctionArg::Named { name, arg: _ } => { - Err(PipelineError::UnsupportedTableOperator(name.to_string())) - } - FunctionArg::Unnamed(arg_expr) => match arg_expr { - FunctionArgExpr::Expr(Expr::Function(function)) => { - let operator_descriptor = - get_table_operator_descriptor(&function.name, Some(&function.args))?; - if let Some(descriptor) = operator_descriptor { - Ok(TableOperatorArg::Descriptor(descriptor)) - } else { - Err(PipelineError::UnsupportedTableOperator( - string_from_sql_object_name(&function.name), - )) - } - } - _ => Ok(TableOperatorArg::Argument(arg.clone())), - }, - } -} - -pub fn insert_table_operator_processor_to_pipeline( - operator: TableOperatorDescriptor, - pipeline: &mut AppPipeline, - pipeline_idx: usize, - query_context: &mut QueryContext, -) -> Result { - let (processor_name, processor): (_, Box) = - if operator.name.to_uppercase() == "TTL" { - let processor_name = generate_name("TOP", &operator, query_context); - let processor = Box::new(TableOperatorProcessorFactory::new( - processor_name.clone(), - operator.clone(), - query_context.udfs.to_owned(), - query_context.runtime.clone(), - )); - (processor_name, processor) - } else if operator.name.to_uppercase() == "TUMBLE" || operator.name.to_uppercase() == "HOP" - { - let processor_name = generate_name("WIN", &operator, query_context); - let processor = Box::new(WindowProcessorFactory::new( - processor_name.clone(), - operator.clone(), - )); - (processor_name, processor) - } else { - return Err(PipelineError::UnsupportedTableOperator( - operator.name.clone(), - )); - }; - - if !query_context.processors_list.insert(processor_name.clone()) { - return Err(PipelineError::ProcessorAlreadyExists(processor_name)); - } - - pipeline.add_processor(processor, processor_name.clone()); - - let Some(table) = operator.args.into_iter().next() else { - return Err(PipelineError::UnsupportedTableOperator( - operator.name.clone(), - )); - }; - - let source_name = match table { - TableOperatorArg::Argument(argument) => get_source_name(&operator.name, &argument)?, - TableOperatorArg::Descriptor(descriptor) => { - let connection_info = insert_table_operator_processor_to_pipeline( - descriptor, - pipeline, - pipeline_idx, - query_context, - )?; - connection_info.output_node.0 - } - }; - - let is_an_entry_point = is_an_entry_point(&source_name, query_context, pipeline_idx); - let is_a_pipeline_output = is_a_pipeline_output(&source_name, query_context, pipeline_idx); - - let input_nodes = if is_an_entry_point { - let entry_point = PipelineEntryPoint::new(source_name.clone(), DEFAULT_PORT_HANDLE); - pipeline.add_entry_point(processor_name.clone(), entry_point); - query_context.used_sources.push(source_name.clone()); - vec![] - } else if is_a_pipeline_output { - vec![( - source_name.clone(), - processor_name.clone(), - DEFAULT_PORT_HANDLE, - )] - } else { - pipeline.connect_nodes( - source_name, - DEFAULT_PORT_HANDLE, - processor_name.clone(), - DEFAULT_PORT_HANDLE, - ); - vec![] - }; - - Ok(ConnectionInfo { - input_nodes, - output_node: (processor_name, DEFAULT_PORT_HANDLE), - }) -} - -fn generate_name( - prefix: &str, - operator: &TableOperatorDescriptor, - query_context: &mut QueryContext, -) -> String { - let processor_name = format!( - "{0}_{1}_{2}", - prefix, - operator.name, - query_context.get_next_processor_id() - ); - processor_name -} diff --git a/dozer-sql/src/builder/tests.rs b/dozer-sql/src/builder/tests.rs deleted file mode 100644 index 0426c4dc84..0000000000 --- a/dozer-sql/src/builder/tests.rs +++ /dev/null @@ -1,211 +0,0 @@ -use super::statement_to_pipeline; -use crate::{errors::PipelineError, tests::utils::create_test_runtime}; -use dozer_core::app::AppPipeline; -#[test] -#[should_panic] -fn disallow_zero_outgoing_ndes() { - let sql = "select * from film"; - let runtime = create_test_runtime(); - statement_to_pipeline( - sql, - &mut AppPipeline::new_with_default_flags(), - None, - vec![], - runtime, - ) - .unwrap(); -} - -#[test] -fn test_duplicate_into_clause() { - let sql = "select * into table1 from film1 ; select * into table1 from film2"; - let runtime = create_test_runtime(); - let result = statement_to_pipeline( - sql, - &mut AppPipeline::new_with_default_flags(), - None, - vec![], - runtime, - ); - assert!(matches!( - result, - Err(PipelineError::DuplicateIntoClause(dup_table)) if dup_table == "table1" - )); -} - -#[test] -fn parse_sql_pipeline() { - let sql = r#" - SELECT - a.name as "Genre", - SUM(amount) as "Gross Revenue(in $)" - INTO gross_revenue_stats - FROM - ( - SELECT - c.name, - f.title, - p.amount - FROM film f - LEFT JOIN film_category fc - ON fc.film_id = f.film_id - LEFT JOIN category c - ON fc.category_id = c.category_id - LEFT JOIN inventory i - ON i.film_id = f.film_id - LEFT JOIN rental r - ON r.inventory_id = i.inventory_id - LEFT JOIN payment p - ON p.rental_id = r.rental_id - WHERE p.amount IS NOT NULL - ) a - GROUP BY name; - - SELECT - f.name, f.title, p.amount - INTO film_amounts - FROM film f - LEFT JOIN film_category fc; - - WITH tbl as (select id from a) - select id - into cte_table - from tbl; - - WITH tbl as (select id from a), - tbl2 as (select id from tbl) - select id - into nested_cte_table - from tbl2; - - WITH cte_table1 as (select id_dt1 from (select id_t1 from table_1) as derived_table_1), - cte_table2 as (select id_ct1 from cte_table1) - select id_ct2 - into nested_derived_table - from cte_table2; - - with tbl as (select id, ticker from stocks) - select tbl.id - into nested_stocks_table - from stocks join tbl on tbl.id = stocks.id; - "#; - - let runtime = create_test_runtime(); - let context = statement_to_pipeline( - sql, - &mut AppPipeline::new_with_default_flags(), - None, - vec![], - runtime, - ) - .unwrap(); - - // Should create as many output tables as into statements - let mut output_keys = context.output_tables_map.keys().collect::>(); - output_keys.sort(); - let mut expected_keys = vec![ - "gross_revenue_stats", - "film_amounts", - "cte_table", - "nested_cte_table", - "nested_derived_table", - "nested_stocks_table", - ]; - expected_keys.sort(); - assert_eq!(output_keys, expected_keys); -} - -#[test] -fn test_missing_into_in_simple_from_clause() { - let sql = r#"SELECT a FROM B "#; - let runtime = create_test_runtime(); - let result = statement_to_pipeline( - sql, - &mut AppPipeline::new_with_default_flags(), - None, - vec![], - runtime, - ); - //check if the result is an error - assert!(matches!(result, Err(PipelineError::MissingIntoClause))) -} - -#[test] -fn test_correct_into_clause() { - let sql = r#"SELECT a INTO C FROM B"#; - let runtime = create_test_runtime(); - let result = statement_to_pipeline( - sql, - &mut AppPipeline::new_with_default_flags(), - None, - vec![], - runtime, - ); - //check if the result is ok - assert!(result.is_ok()); -} - -#[test] -fn test_missing_into_in_nested_from_clause() { - let sql = r#"SELECT a FROM (SELECT a from b)"#; - let runtime = create_test_runtime(); - let result = statement_to_pipeline( - sql, - &mut AppPipeline::new_with_default_flags(), - None, - vec![], - runtime, - ); - //check if the result is an error - assert!(matches!(result, Err(PipelineError::MissingIntoClause))) -} - -#[test] -fn test_correct_into_in_nested_from() { - let sql = r#"SELECT a INTO c FROM (SELECT a from b)"#; - let runtime = create_test_runtime(); - let result = statement_to_pipeline( - sql, - &mut AppPipeline::new_with_default_flags(), - None, - vec![], - runtime, - ); - //check if the result is ok - assert!(result.is_ok()); -} - -#[test] -fn test_missing_into_in_with_clause() { - let sql = r#"WITH tbl as (select a from B) -select B -from tbl;"#; - let runtime = create_test_runtime(); - let result = statement_to_pipeline( - sql, - &mut AppPipeline::new_with_default_flags(), - None, - vec![], - runtime, - ); - //check if the result is an error - assert!(matches!(result, Err(PipelineError::MissingIntoClause))) -} - -#[test] -fn test_correct_into_in_with_clause() { - let sql = r#"WITH tbl as (select a from B) -select B -into C -from tbl;"#; - let runtime = create_test_runtime(); - let result = statement_to_pipeline( - sql, - &mut AppPipeline::new_with_default_flags(), - None, - vec![], - runtime, - ); - //check if the result is ok - assert!(result.is_ok()); -} diff --git a/dozer-sql/src/errors.rs b/dozer-sql/src/errors.rs deleted file mode 100644 index 0182b85594..0000000000 --- a/dozer-sql/src/errors.rs +++ /dev/null @@ -1,317 +0,0 @@ -#![allow(clippy::enum_variant_names)] - -use dozer_core::node::PortHandle; -use dozer_types::chrono::RoundingError; -use dozer_types::errors::internal::BoxedError; -use dozer_types::errors::types::{DeserializationError, TypeError}; - -use dozer_types::thiserror; -use dozer_types::thiserror::Error; -use dozer_types::types::{Field, FieldType}; -use std::fmt::{Display, Formatter}; - -#[derive(Debug, Clone)] -pub struct FieldTypes { - types: Vec, -} - -impl FieldTypes { - pub fn new(types: Vec) -> Self { - Self { types } - } -} - -impl Display for FieldTypes { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - let str_list: Vec = self.types.iter().map(|e| e.to_string()).collect(); - f.write_str(str_list.join(", ").as_str()) - } -} - -#[derive(Error, Debug)] -pub enum PipelineError { - #[error("Invalid operand type for function: {0}()")] - InvalidOperandType(String), - #[error("Invalid return type: {0}")] - InvalidReturnType(String), - #[error("Invalid function: {0}")] - InvalidFunction(String), - #[error("Invalid operator: {0}")] - InvalidOperator(String), - #[error("Invalid value: {0}")] - InvalidValue(String), - #[error("Invalid query: {0}")] - InvalidQuery(String), - #[error("Invalid argument for function {0}(): argument: {1}, index: {2}")] - InvalidFunctionArgument(String, Field, usize), - #[error("Not enough arguments for function {0}()")] - NotEnoughArguments(String), - #[error("Missing INTO clause for top-level SELECT statement")] - MissingIntoClause, - #[error("Duplicate INTO table name found: {0:?}")] - DuplicateIntoClause(String), - - // Error forwarding - #[error("Internal type error: {0}")] - InternalTypeError(#[from] TypeError), - #[error("Internal error: {0}")] - InternalError(#[from] BoxedError), - - #[error("Expression error: {0}")] - Expression(#[from] dozer_sql_expression::error::Error), - - #[error("Unsupported sql: {0}")] - UnsupportedSqlError(#[from] UnsupportedSqlError), - - #[error("Join: {0}")] - JoinError(#[from] JoinError), - - #[error("Product: {0}")] - ProductError(#[from] ProductError), - - #[error("Set: {0}")] - SetError(#[from] SetError), - - #[error("Window: {0}")] - WindowError(#[from] WindowError), - - #[error("Table Function is not supported")] - UnsupportedTableFunction, - - #[error("UNNEST not supported")] - UnsupportedUnnest, - - #[error("Nested Join is not supported")] - UnsupportedNestedJoin, - - #[error("Pivot is not supported")] - UnsupportedPivot, - - #[error("Table Operator: {0} is not supported")] - UnsupportedTableOperator(String), - - #[error("Invalid JOIN: {0}")] - InvalidJoin(String), - - #[error("The JOIN clause is not supported. In this version only INNER, LEFT and RIGHT OUTER JOINs are supported")] - UnsupportedJoinType, - - #[error( - "Unsupported JOIN constraint, only ON is allowed as the JOIN constraint using \'=\' and \'AND\' operators" - )] - UnsupportedJoinConstraintType, - - #[error("Unsupported JOIN constraint {0} only comparison of fields with \'=\' and \'AND\' operators are allowed in the JOIN ON constraint")] - UnsupportedJoinConstraint(String), - - #[error("Invalid JOIN constraint on: {0}")] - InvalidJoinConstraint(String), - - #[error( - "Unsupported JOIN constraint operator {0}, only \'=\' and \'AND\' operators are allowed in the JOIN ON constraint" - )] - UnsupportedJoinConstraintOperator(String), - - #[error("Currently JOIN supports two level of namespacing. For example, `source.field_name` is valid, but `connection.source.field_name` is not.")] - NameSpaceTooLong(String), - - #[error("Window: {0}")] - TableOperatorError(#[from] TableOperatorError), - - #[error("Invalid port handle: {0}")] - InvalidPortHandle(PortHandle), - - #[error("Duplicated Processor name: {0}")] - ProcessorAlreadyExists(String), -} - -#[derive(Error, Debug)] -pub enum UnsupportedSqlError { - #[error("Recursive CTE is not supported. Please refer to the documentation(https://getdozer.io/docs/reference/sql/introduction) for more information. ")] - Recursive, - #[error("Currently this syntax is not supported for CTEs")] - CteFromError, - #[error("Currently only SELECT operations are allowed")] - SelectOnlyError, - #[error("Unsupported syntax in FROM clause")] - JoinTable, - - #[error("FROM clause doesn't support \"Comma Syntax\"")] - FromCommaSyntax, - #[error("ORDER BY is not supported in SQL. You could achieve the same by using the ORDER BY operator in the cache and APIs")] - OrderByError, - #[error("Limit and Offset are not supported in SQL. You could achieve the same by using the LIMIT and OFFSET operators in the cache and APIs")] - LimitOffsetError, - #[error("Select statements should specify INTO for creating output tables")] - IntoError, - - #[error("Unsupported SQL statement {0}")] - GenericError(String), -} - -#[derive(Error, Debug)] -pub enum SetError { - #[error("Invalid input schemas have been populated")] - InvalidInputSchemas, - #[error("Database unavailable for SET")] - DatabaseUnavailable, - #[error("History unavailable for SET source [{0}]")] - HistoryUnavailable(u16), - #[error("Deserialization error: {0}")] - Deserialization(#[from] DeserializationError), -} - -#[derive(Error, Debug)] -pub enum JoinError { - #[error("Currently join supports two level of namespacing. For example, `connection1.field1` is valid, but `connection1.n1.field1` is not.")] - NameSpaceTooLong(String), - #[error("Invalid Join constraint on : {0}")] - InvalidJoinConstraint(String), - #[error("Ambigous field specified in join : {0}")] - AmbiguousField(String), - #[error("Invalid Field specified in join : {0}")] - InvalidFieldSpecified(String), - #[error("Unsupported Join constraint {0} only comparison of fields with \'=\' and \'AND\' operators are allowed in the JOIN ON constraint")] - UnsupportedJoinConstraint(String), - #[error( - "Unsupported Join constraint operator {0}, only \'=\' and \'AND\' operators are allowed in the JOIN ON constraint" - )] - UnsupportedJoinConstraintOperator(String), - #[error( - "Unsupported Join constraint, only ON is allowed as the JOIN constraint using \'=\' and \'AND\' operators" - )] - UnsupportedJoinConstraintType, - #[error("Unsupported Join type")] - UnsupportedJoinType, - - #[error("Overflow error computing the eviction time in the TTL reference field")] - EvictionTimeOverflow, - - #[error("Field type error computing the eviction time in the TTL reference field")] - EvictionTypeOverflow, - - #[error("Deserialization error: {0}")] - Deserialization(#[from] DeserializationError), -} - -#[derive(Error, Debug)] -pub enum ProductError { - #[error("Product Processor Database is not initialised properly")] - InvalidDatabase(), - - #[error("Error deleting a record coming from {0}\n{1}")] - DeleteError(String, #[source] BoxedError), - - #[error("Error inserting a record coming from {0}\n{1}")] - InsertError(String, #[source] BoxedError), - - #[error("Error updating a record from {0} cannot delete the old entry\n{1}")] - UpdateOldError(String, #[source] BoxedError), - - #[error("Error updating a record from {0} cannot insert the new entry\n{1}")] - UpdateNewError(String, #[source] BoxedError), - - #[error("Error in the FROM clause, Table Function is not supported")] - UnsupportedTableFunction, - - #[error("Error in the FROM clause, UNNEST is not supported")] - UnsupportedUnnest, - - #[error("Error in the FROM clause, Pivot is not supported")] - UnsupportedPivot, -} - -#[derive(Error, Debug)] -pub enum WindowError { - #[error("Error in the FROM clause, Invalid function {0:x?}")] - UnsupportedRelationFunction(String), - - #[error("Column name not specified in the window function")] - WindowMissingColumnArgument, - - #[error("Interval not specified in the window function")] - WindowMissingIntervalArgument, - - #[error("Hop size not specified in the window function")] - WindowMissingHopSizeArgument, - - #[error("Invalid time reference column {0} in the window function")] - WindowInvalidColumn(String), - - #[error("Invalid time interval '{0}' specified in the window function")] - WindowInvalidInterval(String), - - #[error("Invalid time hop '{0}' specified in the window function")] - WindowInvalidHop(String), - - #[error("Error in the FROM clause, Derived Table is not supported")] - UnsupportedDerivedTable, - - #[error("Error in the FROM clause, Table Function is not supported")] - UnsupportedTableFunction, - - #[error("Error in the FROM clause, UNNEST is not supported")] - UnsupportedUnnest, - - #[error("This type of Nested Join is not supported")] - UnsupportedNestedJoin, - - #[error("Invalid column specified in Tumble Windowing function.\nOnly Timestamp and Date types are supported")] - TumbleInvalidColumnType(), - #[error("Invalid column specified in Tumble Windowing function.")] - TumbleInvalidColumnIndex(), - #[error("Error in Tumble Windowing function:\n{0}")] - TumbleRoundingError(#[source] RoundingError), - - #[error("Invalid column specified in Hop Windowing function.\nOnly Timestamp and Date types are supported")] - HopInvalidColumnType(), - #[error("Invalid column specified in Hop Windowing function.")] - HopInvalidColumnIndex(), - #[error("Error in Hop Windowing function:\n{0}")] - HopRoundingError(#[source] RoundingError), - - #[error("Invalid WINDOW function")] - InvalidWindow(), - - #[error("For WINDOW functions and alias must be specified")] - AliasNotSpecified(), - - #[error("Source table not specified in the window function")] - WindowMissingSourceArgument, - - #[error("Error in the FROM clause, Derived Table is not supported as WINDOW source")] - UnsupportedDerived, - - #[error("Invalid source table {0} in the window function")] - WindowInvalidSource(String), - - #[error("WINDOW functions require alias")] - NoAlias, -} - -#[derive(Error, Debug)] -pub enum TableOperatorError { - #[error("Internal error: {0}")] - InternalError(#[from] BoxedError), - - #[error("Source Table not specified in the Table Operator {0}")] - MissingSourceArgument(String), - - #[error("Invalid source table {0} in the Table Operator {1}")] - InvalidSourceArgument(String, String), - - #[error("Interval is not specified in the Table Operator {0}")] - MissingIntervalArgument(String), - - #[error("Invalid time interval '{0}' specified in the Table Operator {1}")] - InvalidInterval(String, String), - - #[error("Invalid reference expression '{0}' specified in the Table Operator {1}")] - InvalidReference(String, String), - - #[error("Missing Argument in '{0}' ")] - MissingArgument(String), - - #[error("TTL input must evaluate to timestamp, but it evaluates to {0}")] - InvalidTtlInputType(Field), -} diff --git a/dozer-sql/src/expression/mod.rs b/dozer-sql/src/expression/mod.rs deleted file mode 100644 index 87c2771955..0000000000 --- a/dozer-sql/src/expression/mod.rs +++ /dev/null @@ -1,2 +0,0 @@ -#[cfg(test)] -mod tests; diff --git a/dozer-sql/src/expression/tests/case.rs b/dozer-sql/src/expression/tests/case.rs deleted file mode 100644 index e5e07c788b..0000000000 --- a/dozer-sql/src/expression/tests/case.rs +++ /dev/null @@ -1,106 +0,0 @@ -use crate::expression::tests::test_common::run_fct; -use dozer_types::types::{Field, FieldDefinition, FieldType, Schema, SourceDefinition}; - -#[test] -fn test_case() { - let f = run_fct( - "SELECT \ - CASE \ - WHEN age > 11 THEN 'The age is greater than 11' \ - WHEN age = 11 THEN 'The age is 11' \ - ELSE 'The age is under 11' \ - END AS age_text \ - FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("first_name"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("age"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("chloe".to_string()), Field::Int(11)], - ); - assert_eq!(f, Field::String("The age is 11".to_string())); -} - -#[test] -fn test_case_else() { - let f = run_fct( - "SELECT \ - CASE \ - WHEN age > 11 THEN 'The age is greater than 11' \ - WHEN age = 11 THEN 'The age is 11' \ - ELSE 'The age is under 11' \ - END AS age_text \ - FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("first_name"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("age"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("chloe".to_string()), Field::Int(0)], - ); - assert_eq!(f, Field::String("The age is under 11".to_string())); -} - -#[test] -fn test_case_no_else() { - let f = run_fct( - "SELECT \ - CASE \ - WHEN age > 11 THEN 'The age is greater than 11' \ - WHEN age = 11 THEN 'The age is 11' \ - END AS age_text \ - FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("first_name"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("age"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("chloe".to_string()), Field::Int(9)], - ); - assert_eq!(f, Field::Null); -} diff --git a/dozer-sql/src/expression/tests/cast.rs b/dozer-sql/src/expression/tests/cast.rs deleted file mode 100644 index 2bdfc701f1..0000000000 --- a/dozer-sql/src/expression/tests/cast.rs +++ /dev/null @@ -1,786 +0,0 @@ -use crate::expression::tests::test_common::*; -use dozer_types::types::SourceDefinition; -use dozer_types::{ - chrono::{DateTime, NaiveDate, TimeZone, Utc}, - ordered_float::OrderedFloat, - rust_decimal::Decimal, - types::{Field, FieldDefinition, FieldType, Schema}, -}; - -#[test] -fn test_uint() { - let f = run_fct( - "SELECT CAST(field AS UINT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Int(42)], - ); - assert_eq!(f, Field::UInt(42)); - - let f = run_fct( - "SELECT CAST(field AS UINT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("42".to_string())], - ); - assert_eq!(f, Field::UInt(42)); - - let f = run_fct( - "SELECT CAST(field AS UINT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::UInt, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::UInt(42)], - ); - assert_eq!(f, Field::UInt(42)); -} - -#[test] -fn test_u128() { - let f = run_fct( - "SELECT CAST(field AS U128) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Int(42)], - ); - assert_eq!(f, Field::U128(42)); - - let f = run_fct( - "SELECT CAST(field AS U128) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("42".to_string())], - ); - assert_eq!(f, Field::U128(42)); - - let f = run_fct( - "SELECT CAST(field AS U128) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::UInt, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::UInt(42)], - ); - assert_eq!(f, Field::U128(42)); -} - -#[test] -fn test_int() { - let f = run_fct( - "SELECT CAST(field AS INT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Int(42)], - ); - assert_eq!(f, Field::Int(42)); - - let f = run_fct( - "SELECT CAST(field AS INT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("42".to_string())], - ); - assert_eq!(f, Field::Int(42)); - - let f = run_fct( - "SELECT CAST(field AS INT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::UInt, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::UInt(42)], - ); - assert_eq!(f, Field::Int(42)); -} - -#[test] -fn test_i128() { - let f = run_fct( - "SELECT CAST(field AS I128) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Int(42)], - ); - assert_eq!(f, Field::I128(42)); - - let f = run_fct( - "SELECT CAST(field AS I128) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("42".to_string())], - ); - assert_eq!(f, Field::I128(42)); - - let f = run_fct( - "SELECT CAST(field AS I128) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::UInt, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::UInt(42)], - ); - assert_eq!(f, Field::I128(42)); -} - -#[test] -fn test_float() { - let f = run_fct( - "SELECT CAST(field AS FLOAT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Decimal, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Decimal(Decimal::new(42, 1))], - ); - assert_eq!( - f, - Field::Float(dozer_types::ordered_float::OrderedFloat(4.2)) - ); - - let f = run_fct( - "SELECT CAST(field AS FLOAT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Float(OrderedFloat(4.2))], - ); - assert_eq!(f, Field::Float(OrderedFloat(4.2))); - - let f = run_fct( - "SELECT CAST(field AS FLOAT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Int(4)], - ); - assert_eq!(f, Field::Float(OrderedFloat(4.0))); - - let f = run_fct( - "SELECT CAST(field AS FLOAT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("4.2".to_string())], - ); - assert_eq!(f, Field::Float(OrderedFloat(4.2))); - - let f = run_fct( - "SELECT CAST(field AS FLOAT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::UInt, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::UInt(4)], - ); - assert_eq!(f, Field::Float(OrderedFloat(4.0))); -} - -#[test] -fn test_boolean() { - let f = run_fct( - "SELECT CAST(field AS BOOLEAN) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Boolean, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Boolean(true)], - ); - assert_eq!(f, Field::Boolean(true)); - - let f = run_fct( - "SELECT CAST(field AS BOOLEAN) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Decimal, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Decimal(Decimal::new(0, 0))], - ); - assert_eq!(f, Field::Boolean(false)); - - let f = run_fct( - "SELECT CAST(field AS BOOLEAN) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Decimal, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Decimal(Decimal::new(1, 0))], - ); - assert_eq!(f, Field::Boolean(true)); - - let f = run_fct( - "SELECT CAST(field AS BOOLEAN) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Float(OrderedFloat(1.0))], - ); - assert_eq!(f, Field::Boolean(true)); - - let f = run_fct( - "SELECT CAST(field AS BOOLEAN) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Int(0)], - ); - assert_eq!(f, Field::Boolean(false)); - - let f = run_fct( - "SELECT CAST(field AS BOOLEAN) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::UInt, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::UInt(1)], - ); - assert_eq!(f, Field::Boolean(true)); -} - -#[test] -fn test_string() { - // let f = run_scalar_fct( - // "SELECT CAST(field AS STRING) FROM users", - // Schema::default() - // .field( - // FieldDefinition::new(String::from("field"), FieldType::Binary, false), - // false, - // ) - // .clone(), - // vec![Field::Binary(vec![])], - // ); - // assert_eq!(f, Field::String("".to_string())); - - let f = run_fct( - "SELECT CAST(field AS STRING) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Boolean, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Boolean(true)], - ); - assert_eq!(f, Field::String("TRUE".to_string())); - - let f = run_fct( - "SELECT CAST(field AS STRING) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Date, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Date(NaiveDate::from_ymd_opt(2022, 1, 1).unwrap())], - ); - assert_eq!(f, Field::String("2022-01-01".to_string())); - - let f = run_fct( - "SELECT CAST(field AS STRING) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Decimal, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Decimal(Decimal::new(42, 1))], - ); - assert_eq!(f, Field::String("4.2".to_string())); - - let f = run_fct( - "SELECT CAST(field AS STRING) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Float(OrderedFloat(4.2))], - ); - assert_eq!(f, Field::String("4.2".to_string())); - - let f = run_fct( - "SELECT CAST(field AS STRING) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Int(-42)], - ); - assert_eq!(f, Field::String("-42".to_string())); - - let f = run_fct( - "SELECT CAST(field AS STRING) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("Hello".to_string())], - ); - assert_eq!(f, Field::String("Hello".to_string())); - - let f = run_fct( - "SELECT CAST(field AS STRING) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Text, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Text("Hello".to_string())], - ); - assert_eq!(f, Field::String("Hello".to_string())); - - let f = run_fct( - "SELECT CAST(field AS STRING) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Timestamp(DateTime::from( - Utc.timestamp_millis_opt(42_000_000).unwrap(), - ))], - ); - assert_eq!(f, Field::String("1970-01-01T11:40:00+00:00".to_string())); - - let f = run_fct( - "SELECT CAST(field AS STRING) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::UInt, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::UInt(42)], - ); - assert_eq!(f, Field::String("42".to_string())); -} - -#[test] -fn test_text() { - // let f = run_scalar_fct( - // "SELECT CAST(field AS STRING) FROM users", - // Schema::default() - // .field( - // FieldDefinition::new(String::from("field"), FieldType::Binary, false), - // false, - // ) - // .clone(), - // vec![Field::Binary(vec![])], - // ); - // assert_eq!(f, Field::String("".to_string())); - - let f = run_fct( - "SELECT CAST(field AS TEXT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Boolean, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Boolean(true)], - ); - assert_eq!(f, Field::Text("TRUE".to_string())); - - let f = run_fct( - "SELECT CAST(field AS TEXT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Date, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Date(NaiveDate::from_ymd_opt(2022, 1, 1).unwrap())], - ); - assert_eq!(f, Field::Text("2022-01-01".to_string())); - - let f = run_fct( - "SELECT CAST(field AS TEXT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Decimal, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Decimal(Decimal::new(42, 1))], - ); - assert_eq!(f, Field::Text("4.2".to_string())); - - let f = run_fct( - "SELECT CAST(field AS TEXT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Float(OrderedFloat(4.2))], - ); - assert_eq!(f, Field::Text("4.2".to_string())); - - let f = run_fct( - "SELECT CAST(field AS TEXT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Int(-42)], - ); - assert_eq!(f, Field::Text("-42".to_string())); - - let f = run_fct( - "SELECT CAST(field AS TEXT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("Hello".to_string())], - ); - assert_eq!(f, Field::Text("Hello".to_string())); - - let f = run_fct( - "SELECT CAST(field AS TEXT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Text, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Text("Hello".to_string())], - ); - assert_eq!(f, Field::Text("Hello".to_string())); - - let f = run_fct( - "SELECT CAST(field AS TEXT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Timestamp(DateTime::from( - Utc.timestamp_millis_opt(42_000_000).unwrap(), - ))], - ); - assert_eq!(f, Field::Text("1970-01-01T11:40:00+00:00".to_string())); - - let f = run_fct( - "SELECT CAST(field AS TEXT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::UInt, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::UInt(42)], - ); - assert_eq!(f, Field::Text("42".to_string())); -} - -#[test] -fn test_date() { - let date = "2023-11-22"; - let date_time = DateTime::parse_from_rfc3339(&format!("{}T10:00:00+00:00", date)).unwrap(); - let f = run_fct( - "SELECT CAST(field AS DATE) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Timestamp(date_time)], - ); - - assert_eq!( - f, - Field::Date(NaiveDate::parse_from_str(date, "%Y-%m-%d").unwrap()) - ); -} diff --git a/dozer-sql/src/expression/tests/comparison.rs b/dozer-sql/src/expression/tests/comparison.rs deleted file mode 100644 index 235fef0a3e..0000000000 --- a/dozer-sql/src/expression/tests/comparison.rs +++ /dev/null @@ -1,175 +0,0 @@ -use dozer_types::chrono::{DateTime, NaiveDate}; -use dozer_types::types::DATE_FORMAT; -use dozer_types::types::{Field, Schema}; -use dozer_types::types::{FieldDefinition, FieldType, SourceDefinition}; - -use crate::expression::tests::test_common::run_fct; - -#[test] -fn test_comparison_logical_int() { - let record = vec![Field::Int(124)]; - let schema = Schema::default() - .field( - FieldDefinition::new( - String::from("id"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(); - - let f = run_fct( - "SELECT id FROM users WHERE id = '124'", - schema.clone(), - record.clone(), - ); - assert_eq!(f, Field::Int(124)); - - let f = run_fct( - "SELECT id FROM users WHERE id <= '124'", - schema.clone(), - record.clone(), - ); - assert_eq!(f, Field::Int(124)); - - let f = run_fct( - "SELECT id FROM users WHERE id >= '124'", - schema.clone(), - record.clone(), - ); - assert_eq!(f, Field::Int(124)); - - let f = run_fct( - "SELECT id = '124' FROM users", - schema.clone(), - record.clone(), - ); - assert_eq!(f, Field::Boolean(true)); - - let f = run_fct( - "SELECT id < '124' FROM users", - schema.clone(), - record.clone(), - ); - assert_eq!(f, Field::Boolean(false)); - - let f = run_fct( - "SELECT id > '124' FROM users", - schema.clone(), - record.clone(), - ); - assert_eq!(f, Field::Boolean(false)); - - let f = run_fct( - "SELECT id <= '124' FROM users", - schema.clone(), - record.clone(), - ); - assert_eq!(f, Field::Boolean(true)); - - let f = run_fct("SELECT id >= '124' FROM users", schema, record); - assert_eq!(f, Field::Boolean(true)); -} - -#[test] -fn test_comparison_logical_timestamp() { - let f = run_fct( - "SELECT time = '2020-01-01T00:00:00Z' FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("time"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Timestamp( - DateTime::parse_from_rfc3339("2020-01-01T00:00:00Z").unwrap(), - )], - ); - assert_eq!(f, Field::Boolean(true)); - - let f = run_fct( - "SELECT time < '2020-01-01T00:00:01Z' FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("time"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Timestamp( - DateTime::parse_from_rfc3339("2020-01-01T00:00:00Z").unwrap(), - )], - ); - assert_eq!(f, Field::Boolean(true)); -} - -#[test] -fn test_comparison_logical_date() { - let f = run_fct( - "SELECT date = '2020-01-01' FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("date"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Date( - NaiveDate::parse_from_str("2020-01-01", DATE_FORMAT).unwrap(), - )], - ); - assert_eq!(f, Field::Boolean(true)); - - let f = run_fct( - "SELECT date != '2020-01-01' FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("date"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Date( - NaiveDate::parse_from_str("2020-01-01", DATE_FORMAT).unwrap(), - )], - ); - assert_eq!(f, Field::Boolean(false)); - - let f = run_fct( - "SELECT date > '2020-01-01' FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("date"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Date( - NaiveDate::parse_from_str("2020-01-02", DATE_FORMAT).unwrap(), - )], - ); - assert_eq!(f, Field::Boolean(true)); -} diff --git a/dozer-sql/src/expression/tests/conditional.rs b/dozer-sql/src/expression/tests/conditional.rs deleted file mode 100644 index 8148e70735..0000000000 --- a/dozer-sql/src/expression/tests/conditional.rs +++ /dev/null @@ -1,96 +0,0 @@ -use crate::expression::tests::test_common::*; -use dozer_types::{ - ordered_float::OrderedFloat, - types::{Field, FieldDefinition, FieldType, Schema, SourceDefinition}, -}; - -#[test] -fn test_coalesce_logic() { - let f = run_fct( - "SELECT COALESCE(field, 2) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Null], - ); - assert_eq!(f, Field::Int(2)); - - let f = run_fct( - "SELECT COALESCE(field, CAST(2 AS FLOAT)) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Null], - ); - assert_eq!(f, Field::Float(OrderedFloat(2.0))); - - let f = run_fct( - "SELECT COALESCE(field, 'X') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Null], - ); - assert_eq!(f, Field::String("X".to_string())); - - let f = run_fct( - "SELECT COALESCE(field, 'X') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Null], - ); - assert_eq!(f, Field::String("X".to_string())); -} - -#[test] -fn test_coalesce_logic_null() { - let f = run_fct( - "SELECT COALESCE(field) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("field"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Null], - ); - assert_eq!(f, Field::Null); -} diff --git a/dozer-sql/src/expression/tests/datetime.rs b/dozer-sql/src/expression/tests/datetime.rs deleted file mode 100644 index 24d04d856b..0000000000 --- a/dozer-sql/src/expression/tests/datetime.rs +++ /dev/null @@ -1,182 +0,0 @@ -use crate::expression::tests::test_common::*; -use dozer_types::chrono::{DateTime, NaiveDate}; -use dozer_types::types::{ - DozerDuration, Field, FieldDefinition, FieldType, Schema, SourceDefinition, TimeUnit, -}; - -#[test] -fn test_extract_date() { - let date_fns: Vec<(&str, i64, i64)> = vec![ - ("dow", 6, 0), - ("day", 1, 2), - ("month", 1, 1), - ("year", 2023, 2023), - ("hour", 0, 0), - ("minute", 0, 12), - ("second", 0, 10), - ("millisecond", 1672531200000, 1672618330000), - ("microsecond", 1672531200000000, 1672618330000000), - ("nanoseconds", 1672531200000000000, 1672618330000000000), - ("quarter", 1, 1), - ("epoch", 1672531200, 1672618330), - ("week", 52, 1), - ("century", 21, 21), - ("decade", 203, 203), - ("doy", 1, 2), - ]; - let inputs = vec![ - Field::Date(NaiveDate::from_ymd_opt(2023, 1, 1).unwrap()), - Field::Timestamp(DateTime::parse_from_rfc3339("2023-01-02T00:12:10Z").unwrap()), - ]; - - for (part, val1, val2) in date_fns { - let mut results = vec![]; - for i in inputs.clone() { - let f = run_fct( - &format!("select extract({part} from date) from users"), - Schema::default() - .field( - FieldDefinition::new( - String::from("date"), - FieldType::Date, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![i.clone()], - ); - results.push(f.to_int().unwrap()); - } - assert_eq!(val1, results[0]); - assert_eq!(val2, results[1]); - } -} - -#[test] -fn test_timestamp_diff() { - let f = run_fct( - "SELECT ts1 - ts2 FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("ts1"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("ts2"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![ - Field::Timestamp(DateTime::parse_from_rfc3339("2023-01-02T00:12:11Z").unwrap()), - Field::Timestamp(DateTime::parse_from_rfc3339("2023-01-02T00:12:10Z").unwrap()), - ], - ); - assert_eq!( - f, - Field::Duration(DozerDuration( - std::time::Duration::from_secs(1), - TimeUnit::Nanoseconds - )) - ); -} - -#[test] -fn test_interval() { - let f = run_fct( - "SELECT ts1 - INTERVAL '1' SECOND FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("ts1"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Timestamp( - DateTime::parse_from_rfc3339("2023-01-02T00:12:11Z").unwrap(), - )], - ); - assert_eq!( - f, - Field::Timestamp(DateTime::parse_from_rfc3339("2023-01-02T00:12:10Z").unwrap()) - ); - - let f = run_fct( - "SELECT ts1 + INTERVAL '1' SECOND FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("ts1"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Timestamp( - DateTime::parse_from_rfc3339("2023-01-02T00:12:11Z").unwrap(), - )], - ); - assert_eq!( - f, - Field::Timestamp(DateTime::parse_from_rfc3339("2023-01-02T00:12:12Z").unwrap()) - ); - - let f = run_fct( - "SELECT INTERVAL '1' SECOND + ts1 FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("ts1"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Timestamp( - DateTime::parse_from_rfc3339("2023-01-02T00:12:11Z").unwrap(), - )], - ); - assert_eq!( - f, - Field::Timestamp(DateTime::parse_from_rfc3339("2023-01-02T00:12:12Z").unwrap()) - ); -} - -#[test] -fn test_now() { - let f = run_fct( - "SELECT NOW() FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("ts1"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![], - ); - assert!(f.to_timestamp().is_some()) -} diff --git a/dozer-sql/src/expression/tests/distance.rs b/dozer-sql/src/expression/tests/distance.rs deleted file mode 100644 index fa1a04ff49..0000000000 --- a/dozer-sql/src/expression/tests/distance.rs +++ /dev/null @@ -1,82 +0,0 @@ -use crate::expression::tests::test_common::*; -use dozer_types::ordered_float::OrderedFloat; -use dozer_types::types::{DozerPoint, Field, FieldDefinition, FieldType, Schema, SourceDefinition}; - -#[test] -fn test_distance_logical() { - let tests = vec![ - ("", 1113.0264976969), - ("GEODESIC", 1113.0264976969), - ("HAVERSINE", 1111.7814468418496), - ("VINCENTY", 1113.0264975564357), - ]; - - let schema = Schema::default() - .field( - FieldDefinition::new( - String::from("from"), - FieldType::Point, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("to"), - FieldType::Point, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(); - - let input = vec![ - Field::Point(DozerPoint::from((1.0, 1.0))), - Field::Point(DozerPoint::from((1.01, 1.0))), - ]; - - for (calculation_type, expected_result) in tests { - let sql = if calculation_type.is_empty() { - "SELECT DISTANCE(from, to) FROM LOCATIONS".to_string() - } else { - format!("SELECT DISTANCE(from, to, '{calculation_type}') FROM LOCATIONS") - }; - if let Field::Float(OrderedFloat(result)) = run_fct(&sql, schema.clone(), input.clone()) { - assert!((result - expected_result) < 0.000000001); - } else { - panic!("Expected float"); - } - } -} - -#[test] -fn test_distance_with_nullable_parameter() { - let f = run_fct( - "SELECT DISTANCE(from, to) FROM LOCATION", - Schema::default() - .field( - FieldDefinition::new( - String::from("from"), - FieldType::Point, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("to"), - FieldType::Point, - true, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Point(DozerPoint::from((0.0, 1.0))), Field::Null], - ); - - assert_eq!(f, Field::Null); -} diff --git a/dozer-sql/src/expression/tests/execution.rs b/dozer-sql/src/expression/tests/execution.rs deleted file mode 100644 index 42b697815b..0000000000 --- a/dozer-sql/src/expression/tests/execution.rs +++ /dev/null @@ -1,240 +0,0 @@ -use crate::projection::factory::ProjectionProcessorFactory; -use crate::tests::utils::{create_test_runtime, get_select}; -use dozer_core::node::ProcessorFactory; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_sql_expression::execution::Expression; -use dozer_sql_expression::operator::{BinaryOperatorType, UnaryOperatorType}; -use dozer_sql_expression::scalar::common::ScalarFunctionType; -use dozer_types::types::Record; -use dozer_types::types::{Field, FieldDefinition, FieldType, Schema, SourceDefinition}; - -#[test] -fn test_column_execution() { - use dozer_types::ordered_float::OrderedFloat; - - let schema = Schema::default() - .field( - FieldDefinition::new( - "int_field".to_string(), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "str_field".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "float_field".to_string(), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(); - - let record = Record::new(vec![ - Field::Int(1337), - Field::String("test".to_string()), - Field::Float(OrderedFloat(10.10)), - ]); - - // Column - let mut e = Expression::Column { index: 0 }; - assert_eq!( - e.evaluate(&record, &schema) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int(1337) - ); - - let mut e = Expression::Column { index: 1 }; - assert_eq!( - e.evaluate(&record, &schema) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::String("test".to_string()) - ); - - let mut e = Expression::Column { index: 2 }; - assert_eq!( - e.evaluate(&record, &schema) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Float(OrderedFloat(10.10)) - ); - - // Literal - let mut e = Expression::Literal(Field::Int(1337)); - assert_eq!( - e.evaluate(&record, &schema) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int(1337) - ); - - // UnaryOperator - let mut e = Expression::UnaryOperator { - operator: UnaryOperatorType::Not, - arg: Box::new(Expression::Literal(Field::Boolean(true))), - }; - assert_eq!( - e.evaluate(&record, &schema) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Boolean(false) - ); - - // BinaryOperator - let mut e = Expression::BinaryOperator { - left: Box::new(Expression::Literal(Field::Boolean(true))), - operator: BinaryOperatorType::And, - right: Box::new(Expression::Literal(Field::Boolean(false))), - }; - assert_eq!( - e.evaluate(&record, &schema) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Boolean(false), - ); - - // ScalarFunction - let mut e = Expression::ScalarFunction { - fun: ScalarFunctionType::Abs, - args: vec![Expression::Literal(Field::Int(-1))], - }; - assert_eq!( - e.evaluate(&record, &schema) - .unwrap_or_else(|e| panic!("{}", e.to_string())), - Field::Int(1) - ); -} - -#[test] -fn test_alias() { - let schema = Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::Text, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("ln"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(); - - let select = get_select("SELECT count(fn) AS alias1, ln as alias2 FROM t1").unwrap(); - let runtime = create_test_runtime(); - let processor_factory = ProjectionProcessorFactory::_new( - "projection_id".to_owned(), - select.projection, - vec![], - runtime.clone(), - ); - let r = runtime - .block_on(processor_factory.get_output_schema( - &DEFAULT_PORT_HANDLE, - &[(DEFAULT_PORT_HANDLE, schema)].into_iter().collect(), - )) - .unwrap(); - - assert_eq!( - r, - Schema::default() - .field( - FieldDefinition::new( - String::from("alias1"), - FieldType::Text, - false, - SourceDefinition::Dynamic - ), - false, - ) - .field( - FieldDefinition::new( - String::from("alias2"), - FieldType::String, - false, - SourceDefinition::Dynamic - ), - false, - ) - .clone() - ); -} - -#[test] -fn test_wildcard() { - let schema = Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::Text, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("ln"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(); - - let select = get_select("SELECT * FROM t1").unwrap(); - let runtime = create_test_runtime(); - let processor_factory = ProjectionProcessorFactory::_new( - "projection_id".to_owned(), - select.projection, - vec![], - runtime.clone(), - ); - let r = runtime - .block_on(processor_factory.get_output_schema( - &DEFAULT_PORT_HANDLE, - &[(DEFAULT_PORT_HANDLE, schema)].into_iter().collect(), - )) - .unwrap(); - - assert_eq!( - r, - Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::Text, - false, - SourceDefinition::Dynamic - ), - false, - ) - .field( - FieldDefinition::new( - String::from("ln"), - FieldType::String, - false, - SourceDefinition::Dynamic - ), - false, - ) - .clone() - ); -} diff --git a/dozer-sql/src/expression/tests/expression_builder_test.rs b/dozer-sql/src/expression/tests/expression_builder_test.rs deleted file mode 100644 index 5fd3e38771..0000000000 --- a/dozer-sql/src/expression/tests/expression_builder_test.rs +++ /dev/null @@ -1,512 +0,0 @@ -use crate::tests::utils::{create_test_runtime, get_select}; -use dozer_sql_expression::execution::Expression; -use dozer_sql_expression::operator::BinaryOperatorType; -use dozer_sql_expression::scalar::common::ScalarFunctionType; -use dozer_sql_expression::{builder::ExpressionBuilder, sqlparser::ast::SelectItem}; - -use dozer_sql_expression::aggregate::AggregateFunctionType; -use dozer_types::types::{Field, FieldDefinition, FieldType, Schema, SourceDefinition}; - -#[test] -fn test_simple_function() { - let sql = "SELECT CONCAT(a, b) FROM t0"; - let schema = Schema::default() - .field( - FieldDefinition::new( - "a".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "b".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .to_owned(); - - let runtime = create_test_runtime(); - let mut builder = ExpressionBuilder::new(schema.fields.len(), runtime.clone()); - let e = match &get_select(sql).unwrap().projection[0] { - SelectItem::UnnamedExpr(e) => runtime - .block_on(builder.build(true, e, &schema, &[])) - .unwrap(), - _ => panic!("Invalid expr"), - }; - - assert_eq!(builder.offset, schema.fields.len()); - assert_eq!(builder.aggregations, vec![]); - assert_eq!( - e, - Expression::ScalarFunction { - fun: ScalarFunctionType::Concat, - args: vec![ - Expression::Column { index: 0 }, - Expression::Column { index: 1 } - ] - } - ); -} - -#[test] -fn test_simple_aggr_function() { - let sql = "SELECT SUM(field0) FROM t0"; - let schema = Schema::default() - .field( - FieldDefinition::new( - "field0".to_string(), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .to_owned(); - - let runtime = create_test_runtime(); - let mut builder = ExpressionBuilder::new(schema.fields.len(), runtime.clone()); - let e = match &get_select(sql).unwrap().projection[0] { - SelectItem::UnnamedExpr(e) => runtime - .block_on(builder.build(true, e, &schema, &[])) - .unwrap(), - _ => panic!("Invalid expr"), - }; - - assert_eq!(builder.offset, schema.fields.len()); - assert_eq!( - builder.aggregations, - vec![Expression::AggregateFunction { - fun: AggregateFunctionType::Sum, - args: vec![Expression::Column { index: 0 }] - }] - ); - assert_eq!(e, Expression::Column { index: 1 }); -} - -#[test] -fn test_2_nested_aggr_function() { - let sql = "SELECT SUM(ROUND(field1, 2)) FROM t0"; - let schema = Schema::default() - .field( - FieldDefinition::new( - "field0".to_string(), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "field1".to_string(), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .to_owned(); - - let runtime = create_test_runtime(); - let mut builder = ExpressionBuilder::new(schema.fields.len(), runtime.clone()); - let e = match &get_select(sql).unwrap().projection[0] { - SelectItem::UnnamedExpr(e) => runtime - .block_on(builder.build(true, e, &schema, &[])) - .unwrap(), - _ => panic!("Invalid expr"), - }; - - assert_eq!(builder.offset, schema.fields.len()); - assert_eq!( - builder.aggregations, - vec![Expression::AggregateFunction { - fun: AggregateFunctionType::Sum, - args: vec![Expression::ScalarFunction { - fun: ScalarFunctionType::Round, - args: vec![ - Expression::Column { index: 1 }, - Expression::Literal(Field::Int(2)) - ] - }] - }] - ); - assert_eq!(e, Expression::Column { index: 2 }); -} - -#[test] -fn test_3_nested_aggr_function() { - let sql = "SELECT ROUND(SUM(ROUND(field1, 2))) FROM t0"; - let schema = Schema::default() - .field( - FieldDefinition::new( - "field0".to_string(), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "field1".to_string(), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .to_owned(); - - let runtime = create_test_runtime(); - let mut builder = ExpressionBuilder::new(schema.fields.len(), runtime.clone()); - let e = match &get_select(sql).unwrap().projection[0] { - SelectItem::UnnamedExpr(e) => runtime - .block_on(builder.build(true, e, &schema, &[])) - .unwrap(), - _ => panic!("Invalid expr"), - }; - - assert_eq!(builder.offset, schema.fields.len()); - assert_eq!( - builder.aggregations, - vec![Expression::AggregateFunction { - fun: AggregateFunctionType::Sum, - args: vec![Expression::ScalarFunction { - fun: ScalarFunctionType::Round, - args: vec![ - Expression::Column { index: 1 }, - Expression::Literal(Field::Int(2)) - ] - }] - }] - ); - assert_eq!( - e, - Expression::ScalarFunction { - fun: ScalarFunctionType::Round, - args: vec![Expression::Column { index: 2 }] - } - ); -} - -#[test] -fn test_3_nested_aggr_function_dup() { - let sql = "SELECT CONCAT(SUM(ROUND(field1, 2)), SUM(ROUND(field1, 2))) FROM t0"; - let schema = Schema::default() - .field( - FieldDefinition::new( - "field0".to_string(), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "field1".to_string(), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .to_owned(); - - let runtime = create_test_runtime(); - let mut builder = ExpressionBuilder::new(schema.fields.len(), runtime.clone()); - let e = match &get_select(sql).unwrap().projection[0] { - SelectItem::UnnamedExpr(e) => runtime - .block_on(builder.build(true, e, &schema, &[])) - .unwrap(), - _ => panic!("Invalid expr"), - }; - - assert_eq!(builder.offset, schema.fields.len()); - assert_eq!( - builder.aggregations, - vec![Expression::AggregateFunction { - fun: AggregateFunctionType::Sum, - args: vec![Expression::ScalarFunction { - fun: ScalarFunctionType::Round, - args: vec![ - Expression::Column { index: 1 }, - Expression::Literal(Field::Int(2)) - ] - }] - }] - ); - assert_eq!( - e, - Expression::ScalarFunction { - fun: ScalarFunctionType::Concat, - args: vec![ - Expression::Column { index: 2 }, - Expression::Column { index: 2 } - ] - } - ); -} - -#[test] -fn test_3_nested_aggr_function_and_sum() { - let sql = "SELECT ROUND(SUM(ROUND(field1, 2))) + SUM(field0) FROM t0"; - let schema = Schema::default() - .field( - FieldDefinition::new( - "field0".to_string(), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "field1".to_string(), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .to_owned(); - - let runtime = create_test_runtime(); - let mut builder = ExpressionBuilder::new(schema.fields.len(), runtime.clone()); - let e = match &get_select(sql).unwrap().projection[0] { - SelectItem::UnnamedExpr(e) => runtime - .block_on(builder.build(true, e, &schema, &[])) - .unwrap(), - _ => panic!("Invalid expr"), - }; - - assert_eq!(builder.offset, schema.fields.len()); - assert_eq!( - builder.aggregations, - vec![ - Expression::AggregateFunction { - fun: AggregateFunctionType::Sum, - args: vec![Expression::ScalarFunction { - fun: ScalarFunctionType::Round, - args: vec![ - Expression::Column { index: 1 }, - Expression::Literal(Field::Int(2)) - ] - }] - }, - Expression::AggregateFunction { - fun: AggregateFunctionType::Sum, - args: vec![Expression::Column { index: 0 }] - } - ] - ); - assert_eq!( - e, - Expression::BinaryOperator { - operator: BinaryOperatorType::Add, - left: Box::new(Expression::ScalarFunction { - fun: ScalarFunctionType::Round, - args: vec![Expression::Column { index: 2 }] - }), - right: Box::new(Expression::Column { index: 3 }) - } - ); -} - -#[test] -fn test_3_nested_aggr_function_and_sum_3() { - let sql = "SELECT (ROUND(SUM(ROUND(field1, 2))) + SUM(field0)) + field0 FROM t0"; - let schema = Schema::default() - .field( - FieldDefinition::new( - "field0".to_string(), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "field1".to_string(), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .to_owned(); - - let runtime = create_test_runtime(); - let mut builder = ExpressionBuilder::new(schema.fields.len(), runtime.clone()); - let e = match &get_select(sql).unwrap().projection[0] { - SelectItem::UnnamedExpr(e) => runtime - .block_on(builder.build(true, e, &schema, &[])) - .unwrap(), - _ => panic!("Invalid expr"), - }; - - assert_eq!(builder.offset, schema.fields.len()); - assert_eq!( - builder.aggregations, - vec![ - Expression::AggregateFunction { - fun: AggregateFunctionType::Sum, - args: vec![Expression::ScalarFunction { - fun: ScalarFunctionType::Round, - args: vec![ - Expression::Column { index: 1 }, - Expression::Literal(Field::Int(2)) - ] - }] - }, - Expression::AggregateFunction { - fun: AggregateFunctionType::Sum, - args: vec![Expression::Column { index: 0 }] - } - ] - ); - assert_eq!( - e, - Expression::BinaryOperator { - operator: BinaryOperatorType::Add, - left: Box::new(Expression::BinaryOperator { - operator: BinaryOperatorType::Add, - left: Box::new(Expression::ScalarFunction { - fun: ScalarFunctionType::Round, - args: vec![Expression::Column { index: 2 }] - }), - right: Box::new(Expression::Column { index: 3 }) - }), - right: Box::new(Expression::Column { index: 0 }) - } - ); -} - -#[test] -#[ignore] -#[should_panic] -fn test_wrong_nested_aggregations() { - let sql = "SELECT SUM(SUM(field0)) FROM t0"; - let schema = Schema::default() - .field( - FieldDefinition::new( - "field0".to_string(), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .to_owned(); - - let runtime = create_test_runtime(); - let mut builder = ExpressionBuilder::new(schema.fields.len(), runtime.clone()); - let _e = match &get_select(sql).unwrap().projection[0] { - SelectItem::UnnamedExpr(e) => runtime - .block_on(builder.build(true, e, &schema, &[])) - .unwrap(), - _ => panic!("Invalid expr"), - }; -} - -#[test] -fn test_name_resolution() { - let sql = "SELECT CONCAT(table0.a, connection1.table0.b, a) FROM t0"; - let schema = Schema::default() - .field( - FieldDefinition::new( - "a".to_string(), - FieldType::String, - false, - SourceDefinition::Table { - connection: "connection1".to_string(), - name: "table0".to_string(), - }, - ), - false, - ) - .field( - FieldDefinition::new( - "b".to_string(), - FieldType::String, - false, - SourceDefinition::Table { - connection: "connection1".to_string(), - name: "table0".to_string(), - }, - ), - false, - ) - .to_owned(); - - let runtime = create_test_runtime(); - let mut builder = ExpressionBuilder::new(schema.fields.len(), runtime.clone()); - let e = match &get_select(sql).unwrap().projection[0] { - SelectItem::UnnamedExpr(e) => runtime - .block_on(builder.build(true, e, &schema, &[])) - .unwrap(), - _ => panic!("Invalid expr"), - }; - - assert_eq!(builder.offset, schema.fields.len()); - assert_eq!(builder.aggregations, vec![]); - assert_eq!( - e, - Expression::ScalarFunction { - fun: ScalarFunctionType::Concat, - args: vec![ - Expression::Column { index: 0 }, - Expression::Column { index: 1 }, - Expression::Column { index: 0 } - ] - } - ); -} - -#[test] -fn test_alias_resolution() { - let sql = "SELECT CONCAT(alias.a, a) FROM t0"; - let schema = Schema::default() - .field( - FieldDefinition::new( - "a".to_string(), - FieldType::String, - false, - SourceDefinition::Alias { - name: "alias".to_string(), - }, - ), - false, - ) - .to_owned(); - - let runtime = create_test_runtime(); - let mut builder = ExpressionBuilder::new(schema.fields.len(), runtime.clone()); - let e = match &get_select(sql).unwrap().projection[0] { - SelectItem::UnnamedExpr(e) => runtime - .block_on(builder.build(true, e, &schema, &[])) - .unwrap(), - _ => panic!("Invalid expr"), - }; - - assert_eq!(builder.offset, schema.fields.len()); - assert_eq!(builder.aggregations, vec![]); - assert_eq!( - e, - Expression::ScalarFunction { - fun: ScalarFunctionType::Concat, - args: vec![ - Expression::Column { index: 0 }, - Expression::Column { index: 0 } - ] - } - ); -} diff --git a/dozer-sql/src/expression/tests/in_list.rs b/dozer-sql/src/expression/tests/in_list.rs deleted file mode 100644 index 0296f2210c..0000000000 --- a/dozer-sql/src/expression/tests/in_list.rs +++ /dev/null @@ -1,86 +0,0 @@ -use crate::expression::tests::test_common::run_fct; -use dozer_types::types::{Field, FieldDefinition, FieldType, Schema, SourceDefinition}; - -#[test] -fn test_in_list() { - let f = run_fct( - "SELECT 42 IN (1, 2, 3, 4, 5, 6, 7, 8, 9, 10)", - Schema::default(), - vec![], - ); - assert_eq!(f, Field::Boolean(false)); - - let f = run_fct( - "SELECT 42 IN (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 42)", - Schema::default(), - vec![], - ); - assert_eq!(f, Field::Boolean(true)); - - let schema = Schema::default() - .field( - FieldDefinition::new( - String::from("age"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(); - let f = run_fct( - "SELECT age IN (1, 2, 3, 4, 5, 6, 7, 8, 9, 10) FROM users", - schema.clone(), - vec![Field::Int(42)], - ); - assert_eq!(f, Field::Boolean(false)); - - let f = run_fct( - "SELECT age IN (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 42) FROM users", - schema, - vec![Field::Int(42)], - ); - assert_eq!(f, Field::Boolean(true)); -} - -#[test] -fn test_not_in_list() { - let f = run_fct( - "SELECT 42 NOT IN (1, 2, 3, 4, 5, 6, 7, 8, 9, 10)", - Schema::default(), - vec![], - ); - assert_eq!(f, Field::Boolean(true)); - - let f = run_fct( - "SELECT 42 NOT IN (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 42)", - Schema::default(), - vec![], - ); - assert_eq!(f, Field::Boolean(false)); - - let schema = Schema::default() - .field( - FieldDefinition::new( - String::from("age"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(); - let f = run_fct( - "SELECT age NOT IN (1, 2, 3, 4, 5, 6, 7, 8, 9, 10) FROM users", - schema.clone(), - vec![Field::Int(42)], - ); - assert_eq!(f, Field::Boolean(true)); - - let f = run_fct( - "SELECT age NOT IN (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 42) FROM users", - schema, - vec![Field::Int(42)], - ); - assert_eq!(f, Field::Boolean(false)); -} diff --git a/dozer-sql/src/expression/tests/json_functions.rs b/dozer-sql/src/expression/tests/json_functions.rs deleted file mode 100644 index 619b85aa00..0000000000 --- a/dozer-sql/src/expression/tests/json_functions.rs +++ /dev/null @@ -1,822 +0,0 @@ -use crate::expression::tests::test_common::run_fct; -use dozer_types::json_types::json; -use dozer_types::json_types::JsonValue; -use dozer_types::ordered_float::OrderedFloat; -use dozer_types::types::{Field, FieldDefinition, FieldType, Schema, SourceDefinition}; - -#[test] -fn test_json_value() { - let json_val = json!( - { - "info":{ - "type":1, - "address":{ - "town":"Bristol", - "county":"Avon", - "country":"England" - }, - "tags":["Sport", "Water polo"] - }, - "type":"Basic" - } - ); - - let f = run_fct( - "SELECT JSON_VALUE(jsonInfo,'$.info.address.town') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val)], - ); - - assert_eq!(f, Field::Json(String::from("Bristol").into())); -} - -#[test] -fn test_json_value_null() { - let json_val = json!( - { - "info":{ - "type":1, - "address":{ - "town":"Bristol", - "county":"Avon", - "country":"England" - }, - "tags":["Sport", "Water polo"] - }, - "type":"Basic" - } - ); - - let f = run_fct( - "SELECT JSON_VALUE(jsonInfo,'$.info.address.tags') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val)], - ); - - assert_eq!(f, Field::Json(JsonValue::NULL)); -} - -#[test] -fn test_json_query() { - let json_val = json!( - { - "info": { - "type": 1, - "address": { - "town": "Cheltenham", - "county": "Gloucestershire", - "country": "England" - }, - "tags": ["Sport", "Water polo"] - }, - "type": "Basic" - } - ); - - let f = run_fct( - "SELECT JSON_QUERY(jsonInfo,'$.info.address') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val)], - ); - - assert_eq!( - f, - Field::Json( - json!({"town": "Cheltenham", "county": "Gloucestershire", "country": "England"}) - ) - ); -} - -#[test] -fn test_json_query_null() { - let json_val = json!( - { - "info": { - "type": 1, - "address": { - "town": "Cheltenham", - "county": "Gloucestershire", - "country": "England" - }, - "tags": ["Sport", "Water polo"] - }, - "type": "Basic" - } - ); - - let f = run_fct( - "SELECT JSON_QUERY(jsonInfo,'$.type') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val)], - ); - assert_eq!(f, Field::Json(JsonValue::NULL)); -} - -#[test] -fn test_json_query_len_one_array() { - let json_val = json!( - { - "info": { - "type": 1, - "address": { - "town": "Cheltenham", - "county": "Gloucestershire", - "country": "England" - }, - "tags": ["Sport"] - }, - "type": "Basic" - } - ); - - let f = run_fct( - "SELECT JSON_QUERY(jsonInfo,'$.info.tags') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val)], - ); - assert_eq!(f, Field::Json(json!(["Sport"]))); -} - -#[test] -fn test_json_query_array() { - let json_val = json!( - { - "info": { - "type": 1, - "address": { - "town": "Cheltenham", - "county": "Gloucestershire", - "country": "England" - }, - "tags": ["Sport", "Water polo"] - }, - "type": "Basic" - } - ); - - let f = run_fct( - "SELECT JSON_QUERY(jsonInfo,'$.info.tags') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val)], - ); - - assert_eq!(f, Field::Json(json!(["Sport", "Water polo",]))); -} - -#[test] -fn test_json_query_default_path() { - let json_val = json!( - { - "Cities": [ - { - "Name": "Kabul", - "CountryCode": "AFG", - "District": "Kabol", - "Population": 1780000 - }, - { - "Name": "Qandahar", - "CountryCode": "AFG", - "District": "Qandahar", - "Population": 237500 - } - ] - } - ); - - let f = run_fct( - "SELECT JSON_QUERY(jsonInfo) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val.clone())], - ); - assert_eq!(f, Field::Json(json_val)); -} - -#[test] -fn test_json_query_all() { - let json_val = json!( - [ - {"digit": 30, "letter": "A"}, - {"digit": 31, "letter": "B"} - ] - ); - - let f = run_fct( - "SELECT JSON_QUERY(jsonInfo, '$..*') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val)], - ); - - assert_eq!( - f, - Field::Json(json!([ - { - "digit": 30, - "letter": "A" - }, - 30, - "A", - { - "digit": 31, - "letter": "B" - }, - 31, - "B" - ])) - ); -} - -#[test] -fn test_json_query_iter() { - let json_val = json!( - [ - {"digit": 30, "letter": "A"}, - {"digit": 31, "letter": "B"} - ] - ); - - let f = run_fct( - "SELECT JSON_QUERY(jsonInfo, '$[*].digit') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val)], - ); - - assert_eq!(f, Field::Json(json!([30, 31,]))); -} - -#[test] -fn test_json_cast() { - let f = run_fct( - "SELECT CAST(uint AS JSON) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("uint"), - FieldType::UInt, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::UInt(10_u64)], - ); - - assert_eq!(f, Field::Json(10_f64.into())); - - let f = run_fct( - "SELECT CAST(u128 AS JSON) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("u128"), - FieldType::U128, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::U128(10_u128)], - ); - - assert_eq!(f, Field::Json(10_f64.into())); - - let f = run_fct( - "SELECT CAST(int AS JSON) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("int"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Int(10_i64)], - ); - - assert_eq!(f, Field::Json(10_f64.into())); - - let f = run_fct( - "SELECT CAST(i128 AS JSON) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("i128"), - FieldType::I128, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::I128(10_i128)], - ); - - assert_eq!(f, Field::Json(10_f64.into())); - - let f = run_fct( - "SELECT CAST(float AS JSON) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("float"), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Float(OrderedFloat(10_f64))], - ); - - assert_eq!(f, Field::Json(10_f64.into())); - - let f = run_fct( - "SELECT CAST(str AS JSON) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("str"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("Dozer".to_string())], - ); - - assert_eq!(f, Field::Json("Dozer".into())); - - let f = run_fct( - "SELECT CAST(str AS JSON) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("str"), - FieldType::Text, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Text("Dozer".to_string())], - ); - - assert_eq!(f, Field::Json("Dozer".into())); - - let f = run_fct( - "SELECT CAST(bool AS JSON) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("bool"), - FieldType::Boolean, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Boolean(true)], - ); - - assert_eq!(f, Field::Json(true.into())); -} - -#[test] -fn test_json_value_cast() { - let json_val = json!( - [ - {"digit": 30, "letter": "A"}, - {"digit": 31, "letter": "B"} - ] - ); - - let f = run_fct( - "SELECT JSON_VALUE(jsonInfo, '$[0].digit') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val.clone())], - ); - - assert_eq!(f, Field::Json(30.into())); - - let f = run_fct( - "SELECT CAST(JSON_VALUE(jsonInfo, '$[0].digit') AS UINT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val.clone())], - ); - - assert_eq!(f, Field::UInt(30_u64)); - - let f = run_fct( - "SELECT CAST(JSON_VALUE(jsonInfo, '$[0].digit') AS INT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val.clone())], - ); - - assert_eq!(f, Field::Int(30_i64)); - - let f = run_fct( - "SELECT CAST(JSON_VALUE(jsonInfo, '$[0].digit') AS FLOAT) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val.clone())], - ); - - assert_eq!(f, Field::Float(OrderedFloat(30_f64))); - - let f = run_fct( - "SELECT CAST(JSON_VALUE(jsonInfo, '$[0].digit') AS STRING) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val)], - ); - - assert_eq!(f, Field::String("30".to_string())); - - let json_val = json!( - [ - {"bool": true}, - {"digit": 31, "letter": "B"} - ] - ); - - let f = run_fct( - "SELECT CAST(JSON_VALUE(jsonInfo, '$[0].bool') AS BOOLEAN) FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val)], - ); - - assert_eq!(f, Field::Boolean(true)); -} - -#[test] -fn test_json_value_diff_1() { - let json_val = json!( - { "x": [0,1], "y": "[0,1]", "z": "Monty" } - ); - - let mut f = run_fct( - "SELECT JSON_QUERY(jsonInfo,'$') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val.clone())], - ); - - assert_eq!(f, Field::Json(json_val.clone())); - - f = run_fct( - "SELECT JSON_VALUE(jsonInfo,'$') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val)], - ); - - assert_eq!(f, Field::Json(JsonValue::NULL)); -} - -#[test] -fn test_json_value_diff_2() { - let json_val = json!( - { "x": [0,1], "y": "[0,1]", "z": "Monty" } - ); - - let mut f = run_fct( - "SELECT JSON_QUERY(jsonInfo,'$.x') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val.clone())], - ); - - assert_eq!(f, Field::Json(json!([0, 1,]))); - - f = run_fct( - "SELECT JSON_VALUE(jsonInfo,'$.x') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val)], - ); - - assert_eq!(f, Field::Json(JsonValue::NULL)); -} - -#[test] -fn test_json_value_diff_3() { - let json_val = json!( - { "x": [0,1], "y": "[0,1]", "z": "Monty" } - ); - - let mut f = run_fct( - "SELECT JSON_QUERY(jsonInfo,'$.y') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val.clone())], - ); - - assert_eq!(f, Field::Json(JsonValue::NULL)); - - f = run_fct( - "SELECT JSON_VALUE(jsonInfo,'$.y') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val)], - ); - - assert_eq!(f, Field::Json("[0,1]".into())); -} - -#[test] -fn test_json_value_diff_4() { - let json_val = json!( - { "x": [0,1], "y": "[0,1]", "z": "Monty" } - ); - - let mut f = run_fct( - "SELECT JSON_QUERY(jsonInfo,'$.z') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val.clone())], - ); - - assert_eq!(f, Field::Json(JsonValue::NULL)); - - f = run_fct( - "SELECT JSON_VALUE(jsonInfo,'$.z') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val)], - ); - - assert_eq!(f, Field::Json("Monty".into())); -} - -#[test] -fn test_json_value_diff_5() { - let json_val = json!( - { "x": [0,1], "y": "[0,1]", "z": "Monty" } - ); - - let mut f = run_fct( - "SELECT JSON_QUERY(jsonInfo,'$.x[0]') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val.clone())], - ); - - assert_eq!(f, Field::Json(JsonValue::NULL)); - - f = run_fct( - "SELECT JSON_VALUE(jsonInfo,'$.x[0]') FROM users", - Schema::default() - .field( - FieldDefinition::new( - String::from("jsonInfo"), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Json(json_val)], - ); - - assert_eq!(f, Field::Json(0.into())); -} diff --git a/dozer-sql/src/expression/tests/mod.rs b/dozer-sql/src/expression/tests/mod.rs deleted file mode 100644 index 3fdb76ff53..0000000000 --- a/dozer-sql/src/expression/tests/mod.rs +++ /dev/null @@ -1,14 +0,0 @@ -mod case; -mod cast; -mod comparison; -mod conditional; -mod datetime; -mod distance; -mod execution; -mod expression_builder_test; -mod in_list; -mod json_functions; -mod number; -mod point; -mod string; -mod test_common; diff --git a/dozer-sql/src/expression/tests/models/onnx_modeling.py b/dozer-sql/src/expression/tests/models/onnx_modeling.py deleted file mode 100644 index 6fd8847f26..0000000000 --- a/dozer-sql/src/expression/tests/models/onnx_modeling.py +++ /dev/null @@ -1,22 +0,0 @@ -import torch - - -class Sum(torch.nn.Module): - def __init__(self): - super().__init__() - torch.nn.Sequential() - - def forward(self, x): - return torch.sum(x)[None] - - -model = torch.nn.Sequential() - -dummy_input = torch.randn(4) -model.add_module(name='sum_output', module=Sum()) - -print(model) -print(model(torch.ones(4))) - -# Exporting to ONNX format -torch.onnx.export(model, torch.ones(4), "sum.onnx") diff --git a/dozer-sql/src/expression/tests/number.rs b/dozer-sql/src/expression/tests/number.rs deleted file mode 100644 index abfeca822b..0000000000 --- a/dozer-sql/src/expression/tests/number.rs +++ /dev/null @@ -1,26 +0,0 @@ -use crate::expression::tests::test_common::*; -use dozer_types::types::{Field, FieldDefinition, FieldType, Schema, SourceDefinition}; -use proptest::prelude::*; -use std::ops::Neg; - -#[test] -fn test_abs_logic() { - proptest!(ProptestConfig::with_cases(1000), |(i_num in 0i64..100000000i64)| { - let f = run_fct( - "SELECT ABS(c) FROM USERS", - Schema::default() - .field( - FieldDefinition::new( - String::from("c"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Int(i_num.neg())], - ); - assert_eq!(f, Field::Int(i_num)); - }); -} diff --git a/dozer-sql/src/expression/tests/point.rs b/dozer-sql/src/expression/tests/point.rs deleted file mode 100644 index 407b3900b8..0000000000 --- a/dozer-sql/src/expression/tests/point.rs +++ /dev/null @@ -1,64 +0,0 @@ -use crate::expression::tests::test_common::*; -use dozer_types::ordered_float::OrderedFloat; -use dozer_types::types::{DozerPoint, Field, FieldDefinition, FieldType, Schema, SourceDefinition}; - -#[test] -fn test_point_logical() { - let f = run_fct( - "SELECT POINT(x, y) FROM LOCATION", - Schema::default() - .field( - FieldDefinition::new( - String::from("x"), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("y"), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![ - Field::Float(OrderedFloat(1.0)), - Field::Float(OrderedFloat(2.0)), - ], - ); - assert_eq!(f, Field::Point(DozerPoint::from((1.0, 2.0)))); -} - -#[test] -fn test_point_with_nullable_parameter() { - let f = run_fct( - "SELECT POINT(x, y) FROM LOCATION", - Schema::default() - .field( - FieldDefinition::new( - String::from("x"), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("y"), - FieldType::Float, - true, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Float(OrderedFloat(1.0)), Field::Null], - ); - assert_eq!(f, Field::Null); -} diff --git a/dozer-sql/src/expression/tests/string.rs b/dozer-sql/src/expression/tests/string.rs deleted file mode 100644 index 80f374a20b..0000000000 --- a/dozer-sql/src/expression/tests/string.rs +++ /dev/null @@ -1,481 +0,0 @@ -use crate::expression::tests::test_common::*; -use dozer_types::chrono::{DateTime, NaiveDate, TimeZone, Utc}; -use dozer_types::types::{Field, FieldDefinition, FieldType, Schema, SourceDefinition}; - -#[test] -fn test_concat_string() { - let f = run_fct( - "SELECT CONCAT(fn, ln, fn) FROM USERS", - Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("ln"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![ - Field::String("John".to_string()), - Field::String("Doe".to_string()), - ], - ); - assert_eq!(f, Field::String("JohnDoeJohn".to_string())); -} - -#[test] -fn test_concat_text() { - let f = run_fct( - "SELECT CONCAT(fn, ln, fn) FROM USERS", - Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::Text, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("ln"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![ - Field::Text("John".to_string()), - Field::String("Doe".to_string()), - ], - ); - assert_eq!(f, Field::Text("JohnDoeJohn".to_string())); -} - -#[test] -fn test_concat_text_empty() { - let f = run_fct( - "SELECT CONCAT() FROM USERS", - Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("ln"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![ - Field::String("John".to_string()), - Field::String("Doe".to_string()), - ], - ); - assert_eq!(f, Field::String("".to_string())); -} - -#[test] -#[should_panic] -fn test_concat_wrong_schema() { - let f = run_fct( - "SELECT CONCAT(fn, ln) FROM USERS", - Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("ln"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("John".to_string()), Field::Int(0)], - ); - assert_eq!(f, Field::String("JohnDoe".to_string())); -} - -#[test] -fn test_ucase_string() { - let f = run_fct( - "SELECT UCASE(fn) FROM USERS", - Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("John".to_string())], - ); - assert_eq!(f, Field::String("JOHN".to_string())); -} - -#[test] -fn test_ucase_text() { - let f = run_fct( - "SELECT UCASE(fn) FROM USERS", - Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::Text, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Text("John".to_string())], - ); - assert_eq!(f, Field::Text("JOHN".to_string())); -} - -#[test] -fn test_length() { - let f = run_fct( - "SELECT LENGTH(fn) FROM USERS", - Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("John".to_string())], - ); - assert_eq!(f, Field::UInt(4)); -} - -#[test] -fn test_trim_string() { - let f = run_fct( - "SELECT TRIM(fn) FROM USERS", - Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String(" John ".to_string())], - ); - assert_eq!(f, Field::String("John".to_string())); -} - -#[test] -fn test_trim_null() { - let f = run_fct( - "SELECT TRIM(fn) FROM USERS", - Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Null], - ); - assert_eq!(f, Field::String("".to_string())); -} - -#[test] -fn test_trim_text() { - let f = run_fct( - "SELECT TRIM(fn) FROM USERS", - Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::Text, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Text(" John ".to_string())], - ); - assert_eq!(f, Field::Text("John".to_string())); -} - -#[test] -fn test_trim_value() { - let f = run_fct( - "SELECT TRIM('_' FROM fn) FROM USERS", - Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("___John___".to_string())], - ); - assert_eq!(f, Field::String("John".to_string())); -} - -#[test] -fn test_btrim_value() { - let f = run_fct( - "SELECT TRIM(BOTH '_' FROM fn) FROM USERS", - Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("___John___".to_string())], - ); - assert_eq!(f, Field::String("John".to_string())); -} - -#[test] -fn test_ltrim_value() { - let f = run_fct( - "SELECT TRIM(LEADING '_' FROM fn) FROM USERS", - Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("___John___".to_string())], - ); - assert_eq!(f, Field::String("John___".to_string())); -} - -#[test] -fn test_ttrim_value() { - let f = run_fct( - "SELECT TRIM(TRAILING '_' FROM fn) FROM USERS", - Schema::default() - .field( - FieldDefinition::new( - String::from("fn"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("___John___".to_string())], - ); - assert_eq!(f, Field::String("___John".to_string())); -} - -#[test] -fn test_like_value() { - let f = run_fct( - "SELECT first_name FROM users WHERE first_name LIKE 'J%'", - Schema::default() - .field( - FieldDefinition::new( - String::from("first_name"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("John".to_string())], - ); - assert_eq!(f, Field::String("John".to_string())); -} - -#[test] -fn test_not_like_value() { - let f = run_fct( - "SELECT first_name FROM users WHERE first_name NOT LIKE 'A%'", - Schema::default() - .field( - FieldDefinition::new( - String::from("first_name"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("John".to_string())], - ); - assert_eq!(f, Field::String("John".to_string())); -} - -#[test] -fn test_like_escape() { - let f = run_fct( - "SELECT first_name FROM users WHERE first_name LIKE 'J$%'", - Schema::default() - .field( - FieldDefinition::new( - String::from("first_name"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::String("J%".to_string())], - ); - assert_eq!(f, Field::String("J%".to_string())); -} - -#[test] -fn test_to_char() { - let f = run_fct( - "SELECT TO_CHAR(ts, '%Y-%m-%d') FROM transactions", - Schema::default() - .field( - FieldDefinition::new( - String::from("ts"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Timestamp(DateTime::from( - Utc.timestamp_millis_opt(1672531200).unwrap(), - ))], - ); - assert_eq!(f, Field::String("1970-01-20".to_string())); - - let f = run_fct( - "SELECT TO_CHAR(ts, '%H:%M') FROM transactions", - Schema::default() - .field( - FieldDefinition::new( - String::from("ts"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Timestamp(DateTime::from( - Utc.timestamp_millis_opt(1672531200).unwrap(), - ))], - ); - assert_eq!(f, Field::String("08:35".to_string())); - - let f = run_fct( - "SELECT TO_CHAR(ts, '%H:%M') FROM transactions", - Schema::default() - .field( - FieldDefinition::new( - String::from("ts"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Null], - ); - assert_eq!(f, Field::Null); - - let f = run_fct( - "SELECT TO_CHAR(ts, '%Y-%m-%d') FROM transactions", - Schema::default() - .field( - FieldDefinition::new( - String::from("ts"), - FieldType::Date, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Date(NaiveDate::from_ymd_opt(2020, 1, 2).unwrap())], - ); - assert_eq!(f, Field::String("2020-01-02".to_string())); - - let f = run_fct( - "SELECT TO_CHAR(ts, '%H:%M') FROM transactions", - Schema::default() - .field( - FieldDefinition::new( - String::from("ts"), - FieldType::Date, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone(), - vec![Field::Date(NaiveDate::from_ymd_opt(2020, 1, 2).unwrap())], - ); - assert_eq!(f, Field::String("%H:%M".to_string())); -} diff --git a/dozer-sql/src/expression/tests/test_common.rs b/dozer-sql/src/expression/tests/test_common.rs deleted file mode 100644 index 951a2a08d5..0000000000 --- a/dozer-sql/src/expression/tests/test_common.rs +++ /dev/null @@ -1,62 +0,0 @@ -use crate::tests::utils::create_test_runtime; -use crate::{projection::factory::ProjectionProcessorFactory, tests::utils::get_select}; -use dozer_core::channels::ProcessorChannelForwarder; -use dozer_core::event::EventHub; -use dozer_core::node::ProcessorFactory; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_types::types::{Field, Schema, TableOperation}; -use dozer_types::types::{Operation, Record}; -use std::collections::HashMap; - -struct TestChannelForwarder { - operations: Vec, -} - -impl ProcessorChannelForwarder for TestChannelForwarder { - fn send(&mut self, op: TableOperation) { - self.operations.push(op); - } -} - -pub(crate) fn run_fct(sql: &str, schema: Schema, input: Vec) -> Field { - let select = get_select(sql).unwrap(); - let runtime = create_test_runtime(); - let processor_factory = ProjectionProcessorFactory::_new( - "projection_id".to_owned(), - select.projection, - vec![], - runtime.clone(), - ); - runtime - .block_on( - processor_factory.get_output_schema( - &DEFAULT_PORT_HANDLE, - &[(DEFAULT_PORT_HANDLE, schema.clone())] - .into_iter() - .collect(), - ), - ) - .unwrap(); - - let mut processor = runtime - .block_on(processor_factory.build( - HashMap::from([(DEFAULT_PORT_HANDLE, schema)]), - HashMap::new(), - EventHub::new(1), - )) - .unwrap(); - - let mut fw = TestChannelForwarder { operations: vec![] }; - let rec = Record::new(input); - - let op = Operation::Insert { new: rec }; - - processor - .process(TableOperation::without_id(op, DEFAULT_PORT_HANDLE), &mut fw) - .unwrap(); - - match &mut fw.operations[0].op { - Operation::Insert { new } => new.values.remove(0), - _ => panic!("Unable to find result value"), - } -} diff --git a/dozer-sql/src/lib.rs b/dozer-sql/src/lib.rs deleted file mode 100644 index 5ef05a1d2c..0000000000 --- a/dozer-sql/src/lib.rs +++ /dev/null @@ -1,16 +0,0 @@ -mod aggregation; -pub mod builder; -pub mod errors; -mod expression; -mod planner; -mod product; -mod projection; -mod selection; -mod table_operator; -mod utils; -mod window; - -pub use dozer_sql_expression::sqlparser; - -#[cfg(test)] -mod tests; diff --git a/dozer-sql/src/planner/mod.rs b/dozer-sql/src/planner/mod.rs deleted file mode 100644 index d63d7287cc..0000000000 --- a/dozer-sql/src/planner/mod.rs +++ /dev/null @@ -1,4 +0,0 @@ -pub mod projection; - -#[cfg(test)] -mod tests; diff --git a/dozer-sql/src/planner/projection.rs b/dozer-sql/src/planner/projection.rs deleted file mode 100644 index 6330adfbf1..0000000000 --- a/dozer-sql/src/planner/projection.rs +++ /dev/null @@ -1,248 +0,0 @@ -#![allow(dead_code)] -use std::sync::Arc; - -use crate::builder::string_from_sql_object_name; -use crate::errors::PipelineError; -use dozer_sql_expression::builder::ExpressionBuilder; -use dozer_sql_expression::execution::Expression; -use dozer_sql_expression::sqlparser::ast::{Expr, Ident, SelectItem}; -use dozer_types::models::udf_config::UdfConfig; -use dozer_types::types::{FieldDefinition, Schema}; -use tokio::runtime::Runtime; - -#[derive(Clone, Copy)] -pub enum PrimaryKeyAction { - Retain, - Drop, - Force, -} - -pub struct CommonPlanner<'a> { - input_schema: Schema, - pub post_aggregation_schema: Schema, - pub post_projection_schema: Schema, - // Vector of aggregations to be appended to the original record - pub aggregation_output: Vec, - pub having: Option, - pub groupby: Vec, - pub projection_output: Vec, - pub udfs: &'a [UdfConfig], - pub runtime: Arc, -} - -impl<'a> CommonPlanner<'_> { - fn append_to_schema( - expr: &Expression, - alias: Option, - input_schema: &Schema, - output_schema: &mut Schema, - ) -> Result<(), PipelineError> { - let expr_type = expr.get_type(input_schema)?; - output_schema.fields.push(FieldDefinition::new( - alias.unwrap_or_else(|| expr.to_string(input_schema)), - expr_type.return_type, - expr_type.nullable, - expr_type.source, - )); - - Ok(()) - } - - async fn add_select_item(&mut self, item: SelectItem) -> Result<(), PipelineError> { - let expr_items: Vec<(Expr, Option)> = match item { - SelectItem::UnnamedExpr(expr) => vec![(expr, None)], - SelectItem::ExprWithAlias { expr, alias } => vec![(expr, Some(alias.value))], - SelectItem::QualifiedWildcard(alias, _) => self - .input_schema - .fields - .iter() - .filter(|c| c.check_from(alias.to_string())) - .map(|col| { - ( - Expr::CompoundIdentifier(vec![ - Ident::new(string_from_sql_object_name(&alias)), - Ident::new(col.to_owned().name), - ]), - None, - ) - }) - .collect(), - SelectItem::Wildcard(_) => self - .input_schema - .fields - .iter() - .map(|col| (Expr::Identifier(Ident::new(col.to_owned().name)), None)) - .collect(), - }; - - for (expr, alias) in expr_items { - let mut builder = ExpressionBuilder::new( - self.input_schema.fields.len() + self.aggregation_output.len(), - self.runtime.clone(), - ); - let projection_expression = builder - .build(true, &expr, &self.input_schema, self.udfs) - .await?; - - for new_aggr in builder.aggregations { - Self::append_to_schema( - &new_aggr, - alias.clone(), - &self.input_schema, - &mut self.post_aggregation_schema, - )?; - self.aggregation_output.push(new_aggr); - } - - self.projection_output.push(projection_expression.clone()); - Self::append_to_schema( - &projection_expression, - alias, - &self.post_aggregation_schema, - &mut self.post_projection_schema, - )?; - } - - Ok(()) - } - - async fn add_join_item(&mut self, item: SelectItem) -> Result<(), PipelineError> { - let expr_items: Vec<(Expr, Option)> = match item { - SelectItem::UnnamedExpr(expr) => vec![(expr, None)], - SelectItem::ExprWithAlias { expr, alias } => vec![(expr, Some(alias.value))], - SelectItem::QualifiedWildcard(_, _) => panic!("not supported yet"), - SelectItem::Wildcard(_) => panic!("not supported yet"), - }; - - for (expr, alias) in expr_items { - let mut builder = ExpressionBuilder::new( - self.input_schema.fields.len() + self.aggregation_output.len(), - self.runtime.clone(), - ); - let projection_expression = builder - .build(true, &expr, &self.input_schema, self.udfs) - .await?; - - for new_aggr in builder.aggregations { - Self::append_to_schema( - &new_aggr, - alias.clone(), - &self.input_schema, - &mut self.post_aggregation_schema, - )?; - self.aggregation_output.push(new_aggr); - } - - self.projection_output.push(projection_expression.clone()); - Self::append_to_schema( - &projection_expression, - alias, - &self.post_aggregation_schema, - &mut self.post_projection_schema, - )?; - } - - Ok(()) - } - - async fn add_having_item(&mut self, expr: Expr) -> Result<(), PipelineError> { - let mut builder = ExpressionBuilder::from( - self.input_schema.fields.len(), - self.aggregation_output.clone(), - self.runtime.clone(), - ); - let having_expression = builder - .build(true, &expr, &self.input_schema, self.udfs) - .await?; - - let mut post_aggregation_schema = self.input_schema.clone(); - let mut aggregation_output = Vec::new(); - - for new_aggr in builder.aggregations { - Self::append_to_schema( - &new_aggr, - None, - &self.input_schema, - &mut post_aggregation_schema, - )?; - aggregation_output.push(new_aggr); - } - self.aggregation_output = aggregation_output; - self.post_aggregation_schema = post_aggregation_schema; - - self.having = Some(having_expression); - - Ok(()) - } - - async fn add_groupby_items(&mut self, expr_items: Vec) -> Result<(), PipelineError> { - let mut indexes = vec![]; - let mut set_pk = true; - for expr in expr_items { - let mut builder = ExpressionBuilder::new( - self.input_schema.fields.len() + self.aggregation_output.len(), - self.runtime.clone(), - ); - let groupby_expression = builder - .build(false, &expr, &self.input_schema, self.udfs) - .await?; - self.groupby.push(groupby_expression.clone()); - - if let Some(e) = self - .projection_output - .iter() - .enumerate() - .find(|e| e.1 == &groupby_expression) - { - indexes.push(e.0); - } else { - set_pk = false - } - } - - if set_pk { - indexes.sort(); - self.post_projection_schema.primary_index = indexes; - } - - Ok(()) - } - - pub async fn plan( - &mut self, - projection: Vec, - group_by: Vec, - having: Option, - ) -> Result<(), PipelineError> { - for expr in projection { - self.add_select_item(expr).await?; - } - if !group_by.is_empty() { - self.add_groupby_items(group_by).await?; - } - - if let Some(having) = having { - self.add_having_item(having).await?; - } - - Ok(()) - } - - pub fn new( - input_schema: Schema, - udfs: &'a [UdfConfig], - runtime: Arc, - ) -> CommonPlanner<'a> { - CommonPlanner { - input_schema: input_schema.clone(), - post_aggregation_schema: input_schema, - post_projection_schema: Schema::default(), - aggregation_output: Vec::new(), - having: None, - groupby: Vec::new(), - projection_output: Vec::new(), - udfs, - runtime, - } - } -} diff --git a/dozer-sql/src/planner/tests/mod.rs b/dozer-sql/src/planner/tests/mod.rs deleted file mode 100644 index 6f00d6133b..0000000000 --- a/dozer-sql/src/planner/tests/mod.rs +++ /dev/null @@ -1,3 +0,0 @@ -#[cfg(test)] -mod projection_tests; -mod schema_tests; diff --git a/dozer-sql/src/planner/tests/projection_tests.rs b/dozer-sql/src/planner/tests/projection_tests.rs deleted file mode 100644 index 4241957fce..0000000000 --- a/dozer-sql/src/planner/tests/projection_tests.rs +++ /dev/null @@ -1,131 +0,0 @@ -use dozer_sql_expression::aggregate::AggregateFunctionType; - -use crate::{planner::projection::CommonPlanner, tests::utils::create_test_runtime}; -use dozer_sql_expression::execution::Expression; -use dozer_sql_expression::operator::BinaryOperatorType; -use dozer_sql_expression::scalar::common::ScalarFunctionType; - -use crate::tests::utils::get_select; -use dozer_types::types::{Field, FieldDefinition, FieldType, Schema, SourceDefinition}; - -#[test] -fn test_basic_projection() { - let sql = - "SELECT ROUND(SUM(ROUND(a,2)),2), a as a2 FROM t0 GROUP BY b,a HAVING SUM(ROUND(a,2)) > SUM(b)"; - let schema = Schema::default() - .field( - FieldDefinition::new( - "a".to_string(), - FieldType::Int, - false, - SourceDefinition::Table { - name: "t0".to_string(), - connection: "c0".to_string(), - }, - ), - false, - ) - .field( - FieldDefinition::new( - "b".to_string(), - FieldType::Int, - false, - SourceDefinition::Table { - name: "t0".to_string(), - connection: "c0".to_string(), - }, - ), - false, - ) - .to_owned(); - - let runtime = create_test_runtime(); - let mut projection_planner = CommonPlanner::new(schema, &[], runtime.clone()); - let statement = get_select(sql).unwrap(); - - runtime - .block_on(projection_planner.plan( - statement.projection, - statement.group_by, - statement.having, - )) - .unwrap(); - - assert_eq!( - projection_planner.aggregation_output, - vec![ - Expression::AggregateFunction { - fun: AggregateFunctionType::Sum, - args: vec![Expression::ScalarFunction { - fun: ScalarFunctionType::Round, - args: vec![ - Expression::Column { index: 0 }, - Expression::Literal(Field::Int(2)) - ] - }] - }, - Expression::AggregateFunction { - fun: AggregateFunctionType::Sum, - args: vec![Expression::Column { index: 1 }] - } - ] - ); - - assert_eq!( - projection_planner.projection_output, - vec![ - Expression::ScalarFunction { - fun: ScalarFunctionType::Round, - args: vec![ - Expression::Column { index: 2 }, - Expression::Literal(Field::Int(2)) - ] - }, - Expression::Column { index: 0 } - ] - ); - - assert_eq!( - projection_planner.post_projection_schema, - Schema::default() - .field( - FieldDefinition::new( - "ROUND(SUM(ROUND(a,2)),2)".to_string(), - FieldType::Int, - true, - SourceDefinition::Dynamic - ), - false - ) - .field( - FieldDefinition::new( - "a2".to_string(), - FieldType::Int, - false, - SourceDefinition::Table { - name: "t0".to_string(), - connection: "c0".to_string(), - }, - ), - false, - ) - .to_owned() - ); - - assert_eq!( - projection_planner.groupby, - vec![ - Expression::Column { index: 1 }, - Expression::Column { index: 0 } - ] - ); - - assert_eq!( - projection_planner.having, - Some(Expression::BinaryOperator { - operator: BinaryOperatorType::Gt, - left: Box::new(Expression::Column { index: 2 }), - right: Box::new(Expression::Column { index: 3 }) - }) - ); -} diff --git a/dozer-sql/src/planner/tests/schema_tests.rs b/dozer-sql/src/planner/tests/schema_tests.rs deleted file mode 100644 index 1458080fa5..0000000000 --- a/dozer-sql/src/planner/tests/schema_tests.rs +++ /dev/null @@ -1,123 +0,0 @@ -use crate::tests::utils::get_select; -use crate::{planner::projection::CommonPlanner, tests::utils::create_test_runtime}; -use dozer_types::types::{FieldDefinition, FieldType, Schema, SourceDefinition}; - -#[test] -fn test_schema_index_partial_group_by() { - let sql = "SELECT COUNT(a), b FROM t0 GROUP BY b, c"; - let schema = Schema::default() - .field( - FieldDefinition::new( - "a".to_string(), - FieldType::Int, - false, - SourceDefinition::Table { - name: "t0".to_string(), - connection: "c0".to_string(), - }, - ), - false, - ) - .field( - FieldDefinition::new( - "b".to_string(), - FieldType::Int, - false, - SourceDefinition::Table { - name: "t0".to_string(), - connection: "c0".to_string(), - }, - ), - false, - ) - .field( - FieldDefinition::new( - "c".to_string(), - FieldType::Int, - false, - SourceDefinition::Table { - name: "t0".to_string(), - connection: "c0".to_string(), - }, - ), - false, - ) - .to_owned(); - - let runtime = create_test_runtime(); - let mut projection_planner = CommonPlanner::new(schema, &[], runtime.clone()); - let statement = get_select(sql).unwrap(); - - runtime - .block_on(projection_planner.plan( - statement.projection, - statement.group_by, - statement.having, - )) - .unwrap(); - - assert!(projection_planner - .post_projection_schema - .primary_index - .is_empty(),) -} - -#[test] -fn test_schema_index_full_group_by() { - let sql = "SELECT COUNT(a), c, b FROM t0 GROUP BY b, c"; - let schema = Schema::default() - .field( - FieldDefinition::new( - "a".to_string(), - FieldType::Int, - false, - SourceDefinition::Table { - name: "t0".to_string(), - connection: "c0".to_string(), - }, - ), - false, - ) - .field( - FieldDefinition::new( - "b".to_string(), - FieldType::Int, - false, - SourceDefinition::Table { - name: "t0".to_string(), - connection: "c0".to_string(), - }, - ), - false, - ) - .field( - FieldDefinition::new( - "c".to_string(), - FieldType::Int, - false, - SourceDefinition::Table { - name: "t0".to_string(), - connection: "c0".to_string(), - }, - ), - false, - ) - .to_owned(); - - let runtime = create_test_runtime(); - let mut projection_planner = CommonPlanner::new(schema, &[], runtime.clone()); - let statement = get_select(sql).unwrap(); - - runtime - .block_on(projection_planner.plan( - statement.projection, - statement.group_by, - statement.having, - )) - .unwrap(); - - assert_eq!( - projection_planner.post_projection_schema.primary_index, - vec![1, 2] - ); -} diff --git a/dozer-sql/src/product/join/factory.rs b/dozer-sql/src/product/join/factory.rs deleted file mode 100644 index 8a04d0ce23..0000000000 --- a/dozer-sql/src/product/join/factory.rs +++ /dev/null @@ -1,331 +0,0 @@ -use std::collections::HashMap; - -use dozer_core::{ - event::EventHub, - node::{PortHandle, Processor, ProcessorFactory}, - DEFAULT_PORT_HANDLE, -}; -use dozer_sql_expression::{ - builder::{ExpressionBuilder, NameOrAlias}, - sqlparser::ast::{ - BinaryOperator, Expr as SqlExpr, Ident, JoinConstraint as SqlJoinConstraint, - JoinOperator as SqlJoinOperator, - }, -}; - -use dozer_types::{ - errors::internal::BoxedError, - tonic::async_trait, - types::{FieldDefinition, Schema}, -}; - -use crate::errors::JoinError; -use crate::errors::PipelineError; -use dozer_sql_expression::builder::extend_schema_source_def; - -use super::{ - operator::{JoinOperator, JoinType}, - processor::ProductProcessor, -}; - -pub(crate) const LEFT_JOIN_PORT: PortHandle = 0; -pub(crate) const RIGHT_JOIN_PORT: PortHandle = 1; - -#[derive(Debug)] -pub struct JoinProcessorFactory { - id: String, - left: Option, - right: Option, - join_operator: SqlJoinOperator, - enable_probabilistic_optimizations: bool, -} - -impl JoinProcessorFactory { - pub fn new( - id: String, - left: Option, - right: Option, - join_operator: SqlJoinOperator, - enable_probabilistic_optimizations: bool, - ) -> Self { - Self { - id, - left, - right, - join_operator, - enable_probabilistic_optimizations, - } - } -} - -#[async_trait] -impl ProcessorFactory for JoinProcessorFactory { - fn id(&self) -> String { - self.id.clone() - } - - fn type_name(&self) -> String { - "Join".to_string() - } - fn get_input_ports(&self) -> Vec { - vec![LEFT_JOIN_PORT, RIGHT_JOIN_PORT] - } - - fn get_output_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - async fn get_output_schema( - &self, - _output_port: &PortHandle, - input_schemas: &HashMap, - ) -> Result { - let mut left_schema = input_schemas - .get(&LEFT_JOIN_PORT) - .ok_or(PipelineError::InternalError( - "Invalid Product".to_string().into(), - ))? - .clone(); - - if let Some(left_table_name) = &self.left { - left_schema = extend_schema_source_def(&left_schema, left_table_name); - } - - let mut right_schema = input_schemas - .get(&RIGHT_JOIN_PORT) - .ok_or(PipelineError::InternalError( - "Invalid Product".to_string().into(), - ))? - .clone(); - - if let Some(right_table_name) = &self.right { - right_schema = extend_schema_source_def(&right_schema, right_table_name); - } - - let output_schema = append_schema(&left_schema, &right_schema); - - Ok(output_schema) - } - - async fn build( - &self, - input_schemas: HashMap, - _output_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - let (join_type, join_constraint) = match &self.join_operator { - SqlJoinOperator::Inner(constraint) => (JoinType::Inner, constraint), - SqlJoinOperator::LeftOuter(constraint) => (JoinType::LeftOuter, constraint), - SqlJoinOperator::RightOuter(constraint) => (JoinType::RightOuter, constraint), - _ => return Err(PipelineError::JoinError(JoinError::UnsupportedJoinType).into()), - }; - - let expression = match join_constraint { - SqlJoinConstraint::On(expression) => expression, - _ => { - return Err( - PipelineError::JoinError(JoinError::UnsupportedJoinConstraintType).into(), - ) - } - }; - - let mut left_schema = input_schemas - .get(&LEFT_JOIN_PORT) - .ok_or(PipelineError::InternalError( - "Invalid Product".to_string().into(), - ))? - .clone(); - if let Some(left_table_name) = &self.left { - left_schema = extend_schema_source_def(&left_schema, left_table_name); - } - - let mut right_schema = input_schemas - .get(&RIGHT_JOIN_PORT) - .ok_or(PipelineError::InternalError( - "Invalid Product".to_string().into(), - ))? - .clone(); - if let Some(right_table_name) = &self.right { - right_schema = extend_schema_source_def(&right_schema, right_table_name); - } - - let (left_join_key_indexes, right_join_key_indexes) = - parse_join_constraint(expression, &left_schema, &right_schema)?; - - let join_operator = JoinOperator::new( - join_type, - (left_join_key_indexes, right_join_key_indexes), - (&left_schema, &right_schema), - self.enable_probabilistic_optimizations, - )?; - - Ok(Box::new(ProductProcessor::new( - self.id.clone(), - join_operator, - ))) - } -} - -fn append_schema(left_schema: &Schema, right_schema: &Schema) -> Schema { - let mut output_schema = Schema::default(); - - let left_len = left_schema.fields.len(); - - for field in left_schema.fields.iter() { - output_schema.fields.push(field.clone()); - } - - for field in right_schema.fields.iter() { - output_schema.fields.push(field.clone()); - } - - for primary_key in left_schema.clone().primary_index.into_iter() { - output_schema.primary_index.push(primary_key); - } - - for primary_key in right_schema.clone().primary_index.into_iter() { - output_schema.primary_index.push(primary_key + left_len); - } - - output_schema -} - -fn parse_join_constraint( - expression: &dozer_sql_expression::sqlparser::ast::Expr, - left_join_table: &Schema, - right_join_table: &Schema, -) -> Result<(Vec, Vec), JoinError> { - match expression { - SqlExpr::BinaryOp { - ref left, - op, - ref right, - } => match op { - BinaryOperator::And => { - let (mut left_keys, mut right_keys) = - parse_join_constraint(left, left_join_table, right_join_table)?; - - let (mut left_keys_from_right, mut right_keys_from_right) = - parse_join_constraint(right, left_join_table, right_join_table)?; - left_keys.append(&mut left_keys_from_right); - right_keys.append(&mut right_keys_from_right); - - Ok((left_keys, right_keys)) - } - BinaryOperator::Eq => { - let mut left_key_indexes = vec![]; - let mut right_key_indexes = vec![]; - - let (left_arr, right_arr) = - parse_join_eq_expression(left, left_join_table, right_join_table)?; - left_key_indexes.extend(left_arr); - right_key_indexes.extend(right_arr); - - let (left_arr, right_arr) = - parse_join_eq_expression(right, left_join_table, right_join_table)?; - left_key_indexes.extend(left_arr); - right_key_indexes.extend(right_arr); - - Ok((left_key_indexes, right_key_indexes)) - } - _ => Err(JoinError::UnsupportedJoinConstraintOperator(op.to_string())), - }, - _ => Err(JoinError::UnsupportedJoinConstraint(expression.to_string())), - } -} - -fn parse_join_eq_expression( - expr: &SqlExpr, - left_join_table: &Schema, - right_join_table: &Schema, -) -> Result<(Vec, Vec), JoinError> { - let mut left_key_indexes = vec![]; - let mut right_key_indexes = vec![]; - let (left_keys, right_keys) = match expr.clone() { - SqlExpr::Identifier(ident) => parse_identifier(&[ident], left_join_table, right_join_table), - SqlExpr::CompoundIdentifier(ident) => { - parse_identifier(&ident, left_join_table, right_join_table) - } - _ => { - return Err(JoinError::UnsupportedJoinConstraint( - expr.clone().to_string(), - )) - } - }?; - - match (left_keys, right_keys) { - (Some(left_key), None) => left_key_indexes.push(left_key), - (None, Some(right_key)) => right_key_indexes.push(right_key), - _ => return Err(JoinError::UnsupportedJoinConstraint("".to_string())), - } - - Ok((left_key_indexes, right_key_indexes)) -} - -fn parse_identifier( - ident: &[Ident], - left_join_schema: &Schema, - right_join_schema: &Schema, -) -> Result<(Option, Option), JoinError> { - let left_idx = get_field_index(ident, left_join_schema)?; - - let right_idx = get_field_index(ident, right_join_schema)?; - - match (left_idx, right_idx) { - (None, None) => Err(JoinError::InvalidFieldSpecified( - ExpressionBuilder::fullname_from_ident(ident), - )), - (None, Some(idx)) => Ok((None, Some(idx))), - (Some(idx), None) => Ok((Some(idx), None)), - (Some(_), Some(_)) => Err(JoinError::InvalidJoinConstraint( - ExpressionBuilder::fullname_from_ident(ident), - )), - } -} - -pub fn get_field_index(ident: &[Ident], schema: &Schema) -> Result, JoinError> { - let tables_matches = |table_ident: &Ident, fd: &FieldDefinition| -> bool { - match fd.source.clone() { - dozer_types::types::SourceDefinition::Table { - connection: _, - name, - } => name == table_ident.value, - dozer_types::types::SourceDefinition::Alias { name } => name == table_ident.value, - dozer_types::types::SourceDefinition::Dynamic => false, - } - }; - - let field_index = match ident.len() { - 1 => { - let field_index = schema - .fields - .iter() - .enumerate() - .find(|(_, f)| f.name == ident[0].value) - .map(|(idx, fd)| (idx, fd.clone())); - field_index - } - 2 => { - let table_name = ident.first().expect("table_name is expected"); - let field_name = ident.last().expect("field_name is expected"); - - let index = schema - .fields - .iter() - .enumerate() - .find(|(_, f)| tables_matches(table_name, f) && f.name == field_name.value) - .map(|(idx, fd)| (idx, fd.clone())); - index - } - _ => { - return Err(JoinError::NameSpaceTooLong( - ident - .iter() - .map(|a| a.value.clone()) - .collect::>() - .join("."), - )); - } - }; - field_index.map_or(Ok(None), |(i, _fd)| Ok(Some(i))) -} diff --git a/dozer-sql/src/product/join/mod.rs b/dozer-sql/src/product/join/mod.rs deleted file mode 100644 index e3e84169aa..0000000000 --- a/dozer-sql/src/product/join/mod.rs +++ /dev/null @@ -1,8 +0,0 @@ -use crate::errors::JoinError; - -pub mod factory; - -pub(crate) mod operator; -mod processor; - -type JoinResult = Result; diff --git a/dozer-sql/src/product/join/operator/mod.rs b/dozer-sql/src/product/join/operator/mod.rs deleted file mode 100644 index 1e8e3c6738..0000000000 --- a/dozer-sql/src/product/join/operator/mod.rs +++ /dev/null @@ -1,236 +0,0 @@ -use dozer_types::types::{Record, Schema, Timestamp}; - -use crate::errors::JoinError; - -use self::table::{JoinKey, JoinTable}; - -use super::JoinResult; - -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum JoinBranch { - Left, - Right, -} - -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum JoinType { - Inner, - LeftOuter, - RightOuter, -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum JoinAction { - Insert, - Delete, -} - -mod table; - -#[derive(Debug, Clone)] -pub struct JoinOperator { - join_type: JoinType, - - left: JoinTable, - right: JoinTable, -} - -impl JoinOperator { - pub fn new( - join_type: JoinType, - (left_join_key_indexes, right_join_key_indexes): (Vec, Vec), - (left_schema, right_schema): (&Schema, &Schema), - enable_probabilistic_optimizations: bool, - ) -> Result { - let accurate_keys = !enable_probabilistic_optimizations; - let left = JoinTable::new(left_schema, left_join_key_indexes, accurate_keys)?; - let right = JoinTable::new(right_schema, right_join_key_indexes, accurate_keys)?; - Ok(Self { - join_type, - left, - right, - }) - } - - fn inner_join( - &self, - action: JoinAction, - join_key: &JoinKey, - record: &Record, - record_branch: JoinBranch, - default_if_no_match: bool, - ) -> Vec<(JoinAction, Record)> { - let table = match record_branch { - JoinBranch::Left => &self.right, - JoinBranch::Right => &self.left, - }; - let join_records = create_join_records_fn(record, record_branch); - - table - .get_matching_records(join_key, default_if_no_match) - .map(|matching_record| (action, join_records(matching_record))) - .collect() - } - - fn outer_join( - &self, - action: JoinAction, - join_key: &JoinKey, - record: &Record, - record_branch: JoinBranch, - ) -> Vec<(JoinAction, Record)> { - let (table_to_match, table_of_record) = match record_branch { - JoinBranch::Left => (&self.right, &self.left), - JoinBranch::Right => (&self.left, &self.right), - }; - let join_records = create_join_records_fn(record, record_branch); - let default_join_records = - create_join_records_fn(table_of_record.default_record(), record_branch); - - // We need to query from the table where this record is from: - // - For JoinAction::Insert, did this join key exist before this insert? If not, we need to remove the default record. - // - For JoinAction::Delete, does this join key exist after this delete? If not, we need to insert the default record. - let need_to_act_on_default_record = match action { - JoinAction::Insert => { - // Because this record is already inserted, the join key didn't exist before this insert iif the matching count is now 1. - table_of_record - .get_matching_records(join_key, false) - .take(2) - .count() - == 1 - } - JoinAction::Delete => { - table_of_record - .get_matching_records(join_key, false) - .take(1) - .count() - == 0 - } - }; - - let mut output_records = vec![]; - for matching_record in table_to_match.get_matching_records(join_key, false) { - let join_record = join_records(matching_record); - - if need_to_act_on_default_record { - let default_join_record = default_join_records(matching_record); - match action { - JoinAction::Insert => { - // delete the default join record - output_records.push((JoinAction::Delete, default_join_record)); - // insert the new join record - output_records.push((JoinAction::Insert, join_record)); - } - JoinAction::Delete => { - output_records.push((JoinAction::Delete, join_record)); - output_records.push((JoinAction::Insert, default_join_record)); - } - } - } else { - output_records.push((action, join_record)); - } - } - - output_records - } - - fn join( - &self, - action: JoinAction, - join_key: &JoinKey, - record: &Record, - record_branch: JoinBranch, - ) -> Vec<(JoinAction, Record)> { - match (&self.join_type, record_branch) { - (JoinType::Inner, _) => self.inner_join(action, join_key, record, record_branch, false), - (JoinType::LeftOuter, JoinBranch::Left) => { - self.inner_join(action, join_key, record, JoinBranch::Left, true) - } - (JoinType::LeftOuter, JoinBranch::Right) => { - self.outer_join(action, join_key, record, JoinBranch::Right) - } - (JoinType::RightOuter, JoinBranch::Left) => { - self.outer_join(action, join_key, record, JoinBranch::Left) - } - (JoinType::RightOuter, JoinBranch::Right) => { - self.inner_join(action, join_key, record, JoinBranch::Right, true) - } - } - } - - pub fn delete( - &mut self, - from: JoinBranch, - old: &Record, - old_decoded: &Record, - ) -> Vec<(JoinAction, Record)> { - let join_key = match from { - JoinBranch::Left => self.left.remove(old_decoded), - JoinBranch::Right => self.right.remove(old_decoded), - }; - - self.join(JoinAction::Delete, &join_key, old, from) - } - - pub fn insert( - &mut self, - from: JoinBranch, - new: &Record, - new_decoded: &Record, - ) -> JoinResult> { - let join_key = match from { - JoinBranch::Left => self.left.insert(new.clone(), new_decoded)?, - JoinBranch::Right => self.right.insert(new.clone(), new_decoded)?, - }; - - Ok(self.join(JoinAction::Insert, &join_key, new, from)) - } - - pub fn evict_index(&mut self, now: &Timestamp) { - self.left.evict_index(now); - self.right.evict_index(now); - } -} - -fn create_join_records_fn( - record: &Record, - record_branch: JoinBranch, -) -> impl Fn(&Record) -> Record + '_ { - let lifetime = record.get_lifetime(); - move |matching_record| { - let matching_lifetime = matching_record.get_lifetime(); - - let mut output_record = match record_branch { - JoinBranch::Left => { - let len = record.values().len() + matching_record.values().len(); - let mut data = Vec::with_capacity(len); - data.extend_from_slice(record.values()); - data.extend_from_slice(matching_record.values()); - Record::new(data) - } - JoinBranch::Right => { - let len = record.values().len() + matching_record.values().len(); - let mut data = Vec::with_capacity(len); - data.extend_from_slice(matching_record.values()); - data.extend_from_slice(record.values()); - Record::new(data) - } - }; - - if let Some(lifetime) = &lifetime { - if let Some(matching_lifetime) = matching_lifetime { - if lifetime.reference > matching_lifetime.reference { - output_record.set_lifetime(Some(lifetime.clone())); - } else { - output_record.set_lifetime(Some(matching_lifetime)); - } - } else { - output_record.set_lifetime(Some(lifetime.clone())); - } - } else if let Some(matching_lifetime) = matching_lifetime { - output_record.set_lifetime(Some(matching_lifetime)); - } - - output_record - } -} diff --git a/dozer-sql/src/product/join/operator/table.rs b/dozer-sql/src/product/join/operator/table.rs deleted file mode 100644 index a16e7094e9..0000000000 --- a/dozer-sql/src/product/join/operator/table.rs +++ /dev/null @@ -1,227 +0,0 @@ -use std::{ - collections::{ - hash_map::{self, Values}, - HashMap, - }, - iter::{once, Flatten, Once}, -}; - -use dozer_types::{ - chrono, - types::{Field, Record, Schema, Timestamp}, -}; -use linked_hash_map::LinkedHashMap; - -use crate::{ - errors::JoinError, - utils::record_hashtable_key::{get_record_hash, RecordKey}, -}; - -pub type JoinKey = RecordKey; -type IndexKey = (JoinKey, u64); // (join_key, primary_key) - -#[derive(Debug, Clone)] -pub struct JoinTable { - join_key_indexes: Vec, - primary_key_indexes: Vec, - default_record: Record, - map: HashMap>>, - lifetime_map: LinkedHashMap>, - accurate_keys: bool, -} - -impl JoinTable { - pub fn new( - schema: &Schema, - join_key_indexes: Vec, - accurate_keys: bool, - ) -> Result { - let primary_key_indexes = if schema.primary_index.is_empty() { - (0..schema.fields.len()).collect() - } else { - schema.primary_index.clone() - }; - - Ok(Self { - join_key_indexes, - primary_key_indexes, - default_record: Record::nulls_from_schema(schema), - map: Default::default(), - lifetime_map: Default::default(), - accurate_keys, - }) - } - - pub fn get_matching_records<'a>( - &'a self, - join_key: &JoinKey, - default_if_no_match: bool, - ) -> MatchingRecords<'a> { - if let Some(records_map) = self.map.get(join_key) { - MatchingRecords::Values(records_map.values().flatten()) - } else if default_if_no_match { - MatchingRecords::Default(once(&self.default_record)) - } else { - MatchingRecords::Empty - } - } - - pub fn default_record(&self) -> &Record { - &self.default_record - } - - pub fn insert( - &mut self, - record: Record, - record_decoded: &Record, - ) -> Result { - let join_key = self.get_join_key(record_decoded); - let primary_key = get_record_key_hash(record_decoded, &self.primary_key_indexes); - - if let Some(lifetime) = record.get_lifetime() { - let Some(eviction_instant) = - lifetime - .reference - .checked_add_signed(chrono::Duration::nanoseconds( - lifetime.duration.as_nanos() as i64, - )) - else { - return Err(JoinError::EvictionTimeOverflow); - }; - - self.lifetime_map - .entry(eviction_instant) - .or_default() - .push((join_key.clone(), primary_key)); - } - - self.map - .entry(join_key.clone()) - .or_default() - .entry(primary_key) - .or_default() - .push(record); - - Ok(join_key) - } - - pub fn remove(&mut self, record: &Record) -> JoinKey { - let join_key = self.get_join_key(record); - if let hash_map::Entry::Occupied(record_map) = self.map.entry(join_key.clone()) { - let primary_key = get_record_key_hash(record, &self.primary_key_indexes); - remove_record_using_primary_key(record_map, primary_key); - } - join_key - } - - pub fn evict_index(&mut self, now: &Timestamp) { - let mut keys_to_remove = vec![]; - for (eviction_instant, join_index_keys) in self.lifetime_map.iter() { - if eviction_instant <= now { - keys_to_remove.push(*eviction_instant); - for (join_key, primary_key) in join_index_keys { - if let hash_map::Entry::Occupied(record_map) = self.map.entry(join_key.clone()) - { - remove_record_using_primary_key(record_map, *primary_key); - } - } - } else { - break; - } - } - - for key in keys_to_remove { - self.lifetime_map.remove(&key); - } - } - - fn get_join_key(&self, record: &Record) -> JoinKey { - if self.accurate_keys { - JoinKey::Accurate(get_record_key_fields(record, &self.join_key_indexes)) - } else { - JoinKey::Hash(get_record_key_hash(record, &self.join_key_indexes)) - } - } -} - -#[derive(Debug)] -pub enum MatchingRecords<'a> { - Values(Flatten>>), - Default(Once<&'a Record>), - Empty, -} - -impl<'a> Iterator for MatchingRecords<'a> { - type Item = &'a Record; - - fn next(&mut self) -> Option { - match self { - MatchingRecords::Values(values) => values.next(), - MatchingRecords::Default(default) => default.next(), - MatchingRecords::Empty => None, - } - } -} - -fn get_record_key_hash(record: &Record, key_indexes: &[usize]) -> u64 { - let key_fields = key_indexes.iter().map(|i| &record.values[*i]); - get_record_hash(key_fields) -} - -fn get_record_key_fields(record: &Record, key_indexes: &[usize]) -> Vec { - key_indexes - .iter() - .map(|i| record.values[*i].clone()) - .collect() -} - -fn remove_record_using_primary_key( - mut record_map: hash_map::OccupiedEntry>>, - primary_key: u64, -) { - if let hash_map::Entry::Occupied(mut record_vec) = record_map.get_mut().entry(primary_key) { - record_vec.get_mut().pop(); - if record_vec.get().is_empty() { - record_vec.remove(); - } - } - - if record_map.get().is_empty() { - record_map.remove(); - } -} - -#[cfg(test)] -mod tests { - use dozer_types::types::{FieldDefinition, FieldType}; - - use super::*; - - #[test] - fn test_match_insert_remove() { - let schema = Schema { - fields: vec![FieldDefinition { - name: "a".to_string(), - typ: FieldType::Int, - nullable: false, - source: Default::default(), - description: None, - }], - primary_index: vec![0], - }; - let mut table = JoinTable::new(&schema, vec![0], true).unwrap(); - - let record = Record::new(vec![Field::Int(1)]); - let join_key = table.get_join_key(&record); - assert_eq!(table.get_matching_records(&join_key, true).count(), 1); - assert_eq!(table.get_matching_records(&join_key, false).count(), 0); - - let join_key = table.insert(record.clone(), &record).unwrap(); - assert_eq!(table.get_matching_records(&join_key, true).count(), 1); - assert_eq!(table.get_matching_records(&join_key, false).count(), 1); - - let join_key = table.remove(&record); - assert_eq!(table.get_matching_records(&join_key, true).count(), 1); - assert_eq!(table.get_matching_records(&join_key, false).count(), 0); - } -} diff --git a/dozer-sql/src/product/join/processor.rs b/dozer-sql/src/product/join/processor.rs deleted file mode 100644 index eb187a8464..0000000000 --- a/dozer-sql/src/product/join/processor.rs +++ /dev/null @@ -1,437 +0,0 @@ -use dozer_core::channels::ProcessorChannelForwarder; -use dozer_core::epoch::Epoch; -use dozer_core::node::Processor; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_types::errors::internal::BoxedError; -use dozer_types::types::{Lifetime, Operation, TableOperation}; - -use crate::errors::PipelineError; - -use super::operator::{JoinAction, JoinBranch, JoinOperator}; - -#[derive(Debug)] -pub struct ProductProcessor { - join_operator: JoinOperator, -} - -impl ProductProcessor { - pub fn new(_id: String, join_operator: JoinOperator) -> Self { - Self { join_operator } - } - - fn update_eviction_index(&mut self, lifetime: Lifetime) { - self.join_operator.evict_index(&lifetime.reference); - } -} - -impl Processor for ProductProcessor { - fn commit(&self, _epoch: &Epoch) -> Result<(), BoxedError> { - Ok(()) - } - - fn process( - &mut self, - op: TableOperation, - fw: &mut dyn ProcessorChannelForwarder, - ) -> Result<(), BoxedError> { - let from_branch = match op.port { - 0 => JoinBranch::Left, - 1 => JoinBranch::Right, - _ => return Err(PipelineError::InvalidPortHandle(op.port).into()), - }; - let records = match op.op { - Operation::Delete { old } => { - if let Some(lifetime) = old.get_lifetime() { - self.update_eviction_index(lifetime); - } - - self.join_operator.delete(from_branch, &old, &old) - } - Operation::Insert { new } => { - if let Some(lifetime) = new.get_lifetime() { - self.update_eviction_index(lifetime); - } - - self.join_operator - .insert(from_branch, &new, &new) - .map_err(PipelineError::JoinError)? - } - Operation::Update { old, new } => { - if let Some(lifetime) = old.get_lifetime() { - self.update_eviction_index(lifetime); - } - - let mut old_records = self.join_operator.delete(from_branch, &old, &old); - - let new_records = self - .join_operator - .insert(from_branch, &new, &new) - .map_err(PipelineError::JoinError)?; - - old_records.extend(new_records); - old_records - } - Operation::BatchInsert { new } => { - for record in &new { - self.process( - TableOperation::without_id( - Operation::Insert { - new: record.clone(), - }, - op.port, - ), - fw, - )?; - } - return Ok(()); - } - }; - - for (action, record) in records { - match action { - JoinAction::Insert => { - fw.send(TableOperation::without_id( - Operation::Insert { new: record }, - DEFAULT_PORT_HANDLE, - )); - } - JoinAction::Delete => { - fw.send(TableOperation::without_id( - Operation::Delete { old: record }, - DEFAULT_PORT_HANDLE, - )); - } - } - } - - Ok(()) - } -} - -#[cfg(test)] -mod tests { - use std::collections::HashMap; - - use dozer_core::{event::EventHub, node::ProcessorFactory}; - use dozer_sql_expression::builder::NameOrAlias; - use dozer_sql_expression::sqlparser::ast::JoinOperator as SqlJoinOperator; - use dozer_types::types::{Field, FieldDefinition, Record, Schema}; - - use crate::product::join::{ - factory::{LEFT_JOIN_PORT, RIGHT_JOIN_PORT}, - operator::JoinType, - }; - use crate::{product::join::factory::JoinProcessorFactory, tests::utils::get_select}; - - use super::*; - - struct TestChannelForwarder { - operations: Vec, - } - - impl ProcessorChannelForwarder for TestChannelForwarder { - fn send(&mut self, op: TableOperation) { - self.operations.push(op); - } - } - - fn create_schema(table_name: &'static str) -> Schema { - let mut schema = Schema::new(); - schema - .field( - FieldDefinition { - name: "joinkey".into(), - typ: dozer_types::types::FieldType::UInt, - nullable: false, - source: dozer_types::types::SourceDefinition::Table { - connection: "test".into(), - name: table_name.into(), - }, - description: None, - }, - true, - ) - .field( - FieldDefinition { - name: "data".into(), - typ: dozer_types::types::FieldType::UInt, - nullable: false, - source: dozer_types::types::SourceDefinition::Table { - connection: "test".into(), - name: table_name.into(), - }, - description: None, - }, - false, - ); - schema - } - - enum JoinSide { - Left, - Right, - } - - struct Executor { - processor: Box, - forwarder: TestChannelForwarder, - } - - impl Executor { - async fn new(kind: JoinType) -> Self { - let left_schema = create_schema("left"); - let right_schema = create_schema("right"); - - let stmt = get_select( - "SELECT left.joinkey FROM left INNER JOIN right ON left.joinkey = right.joinkey", - ) - .unwrap(); - let join = &stmt.from[0].joins[0]; - let join_op = join.join_operator.clone(); - let SqlJoinOperator::Inner(constraint) = join_op else { - unreachable!() - }; - let join_op = match kind { - JoinType::Inner => SqlJoinOperator::Inner(constraint), - JoinType::LeftOuter => SqlJoinOperator::LeftOuter(constraint), - JoinType::RightOuter => SqlJoinOperator::RightOuter(constraint), - }; - let factory = JoinProcessorFactory::new( - "test".into(), - Some(NameOrAlias("left".into(), None)), - Some(NameOrAlias("right".into(), None)), - join_op, - false, - ); - - let schemas = [ - (LEFT_JOIN_PORT, left_schema), - (RIGHT_JOIN_PORT, right_schema), - ] - .into_iter() - .collect(); - let processor = factory - .build(schemas, HashMap::new(), EventHub::new(1)) - .await - .unwrap(); - - let forwarder = TestChannelForwarder { operations: vec![] }; - Executor { - processor, - forwarder, - } - } - - fn do_op(&mut self, operation: Operation, side: JoinSide) -> Vec { - let port = match side { - JoinSide::Left => LEFT_JOIN_PORT, - JoinSide::Right => RIGHT_JOIN_PORT, - }; - self.processor - .process( - TableOperation::without_id(operation, port), - &mut self.forwarder, - ) - .unwrap(); - let output_ops = self.forwarder.operations.clone(); - self.forwarder.operations.clear(); - output_ops.into_iter().map(|op| op.op).collect() - } - - fn insert(&mut self, side: JoinSide, values: &[Field]) -> (Record, Vec) { - let record = Record::new(values.to_vec()); - let op = Operation::Insert { - new: record.clone(), - }; - (record, self.do_op(op, side)) - } - - fn update( - &mut self, - side: JoinSide, - old: Record, - new: &[Field], - ) -> (Record, Vec) { - let record = Record::new(new.to_vec()); - let op = Operation::Update { - old, - new: record.clone(), - }; - (record, self.do_op(op, side)) - } - - fn delete(&mut self, side: JoinSide, old: Record) -> Vec { - let op = Operation::Delete { old }; - self.do_op(op, side) - } - } - - fn join_record(left: Record, right: Record) -> Record { - let mut values = left.values; - values.extend(right.values); - Record::new(values) - } - - #[tokio::test] - async fn test_inner_join() { - let mut exec = Executor::new(JoinType::Inner).await; - - let (left_record, ops) = exec.insert(JoinSide::Left, &[Field::UInt(0), Field::UInt(1)]); - assert_eq!(ops, &[]); - - let (right_record, ops) = exec.insert(JoinSide::Right, &[Field::UInt(0), Field::UInt(2)]); - assert_eq!( - ops, - &[Operation::Insert { - new: join_record(left_record.clone(), right_record.clone()) - }] - ); - let (new_left_record, ops) = exec.update( - JoinSide::Left, - left_record.clone(), - &[Field::UInt(0), Field::UInt(2)], - ); - assert_eq!( - ops, - &[ - Operation::Delete { - old: join_record(left_record.clone(), right_record.clone()) - }, - Operation::Insert { - new: join_record(new_left_record.clone(), right_record.clone()) - } - ] - ); - - assert_eq!( - exec.delete(JoinSide::Right, right_record.clone()), - &[Operation::Delete { - old: join_record(new_left_record.clone(), right_record.clone()) - },] - ); - } - - #[tokio::test] - async fn test_left_outer_join() { - let mut exec = Executor::new(JoinType::LeftOuter).await; - - let null_record = Record::new(vec![Field::Null, Field::Null]); - - let (left_record, ops) = exec.insert(JoinSide::Left, &[Field::UInt(0), Field::UInt(1)]); - assert_eq!( - ops, - &[Operation::Insert { - new: join_record(left_record.clone(), null_record.clone()) - }] - ); - - let (right_record, ops) = exec.insert(JoinSide::Right, &[Field::UInt(0), Field::UInt(2)]); - assert_eq!( - ops, - &[ - Operation::Delete { - old: join_record(left_record.clone(), null_record.clone()), - }, - Operation::Insert { - new: join_record(left_record.clone(), right_record.clone()) - } - ] - ); - let (new_left_record, ops) = exec.update( - JoinSide::Left, - left_record.clone(), - &[Field::UInt(0), Field::UInt(2)], - ); - assert_eq!( - ops, - &[ - Operation::Delete { - old: join_record(left_record.clone(), right_record.clone()) - }, - Operation::Insert { - new: join_record(new_left_record.clone(), right_record.clone()) - } - ] - ); - - assert_eq!( - exec.delete(JoinSide::Right, right_record.clone()), - &[ - Operation::Delete { - old: join_record(new_left_record.clone(), right_record.clone()) - }, - Operation::Insert { - new: join_record(new_left_record.clone(), null_record.clone(),) - }, - ] - ); - let (right_record, _) = exec.insert(JoinSide::Right, &[Field::UInt(0), Field::UInt(2)]); - - assert_eq!( - exec.delete(JoinSide::Left, new_left_record.clone()), - &[Operation::Delete { - old: join_record(new_left_record.clone(), right_record.clone()) - },] - ); - } - - #[tokio::test] - async fn test_right_outer_join() { - let mut exec = Executor::new(JoinType::RightOuter).await; - - let null_record = Record::new(vec![Field::Null, Field::Null]); - - let (left_record, ops) = exec.insert(JoinSide::Left, &[Field::UInt(0), Field::UInt(1)]); - assert_eq!(ops, &[]); - - let (right_record, ops) = exec.insert(JoinSide::Right, &[Field::UInt(0), Field::UInt(2)]); - assert_eq!( - ops, - &[Operation::Insert { - new: join_record(left_record.clone(), right_record.clone()) - }] - ); - let (new_left_record, ops) = exec.update( - JoinSide::Left, - left_record.clone(), - &[Field::UInt(0), Field::UInt(2)], - ); - assert_eq!( - ops, - &[ - Operation::Delete { - old: join_record(left_record.clone(), right_record.clone()) - }, - Operation::Insert { - new: join_record(null_record.clone(), right_record.clone()) - }, - Operation::Delete { - old: join_record(null_record.clone(), right_record.clone()) - }, - Operation::Insert { - new: join_record(new_left_record.clone(), right_record.clone()) - } - ] - ); - - assert_eq!( - exec.delete(JoinSide::Left, right_record.clone()), - &[ - Operation::Delete { - old: join_record(new_left_record.clone(), right_record.clone()) - }, - Operation::Insert { - new: join_record(null_record.clone(), right_record.clone(),) - }, - ] - ); - let (new_left_record, _) = exec.insert(JoinSide::Left, &[Field::UInt(0), Field::UInt(2)]); - - assert_eq!( - exec.delete(JoinSide::Right, new_left_record.clone()), - &[Operation::Delete { - old: join_record(new_left_record.clone(), right_record.clone()) - },] - ); - } -} diff --git a/dozer-sql/src/product/mod.rs b/dozer-sql/src/product/mod.rs deleted file mode 100644 index c3631b42ea..0000000000 --- a/dozer-sql/src/product/mod.rs +++ /dev/null @@ -1,3 +0,0 @@ -pub(crate) mod join; -pub(crate) mod set; -pub(crate) mod table; diff --git a/dozer-sql/src/product/set/mod.rs b/dozer-sql/src/product/set/mod.rs deleted file mode 100644 index 9da13861bf..0000000000 --- a/dozer-sql/src/product/set/mod.rs +++ /dev/null @@ -1,5 +0,0 @@ -pub mod set_factory; -mod set_processor; - -pub(crate) mod operator; -pub(crate) mod record_map; diff --git a/dozer-sql/src/product/set/operator.rs b/dozer-sql/src/product/set/operator.rs deleted file mode 100644 index 1f142f43e3..0000000000 --- a/dozer-sql/src/product/set/operator.rs +++ /dev/null @@ -1,96 +0,0 @@ -use super::record_map::{CountingRecordMap, CountingRecordMapEnum}; -use crate::errors::PipelineError; -use dozer_sql_expression::sqlparser::ast::{SetOperator, SetQuantifier}; -use dozer_types::types::Record; - -#[derive(Clone, Debug, PartialEq, Eq, Copy)] -pub enum SetAction { - Insert, - Delete, - // Update, -} - -#[derive(Clone, Debug)] -pub struct SetOperation { - pub op: SetOperator, - pub quantifier: SetQuantifier, -} - -impl SetOperation { - pub fn _new(op: SetOperator) -> Self { - Self { - op, - quantifier: SetQuantifier::None, - } - } - - pub fn execute( - &self, - action: SetAction, - record: Record, - record_map: &mut CountingRecordMapEnum, - ) -> Result, PipelineError> { - match (self.op, self.quantifier) { - (SetOperator::Union, SetQuantifier::All) => Ok(vec![(action, record)]), - (SetOperator::Union, SetQuantifier::None) => { - self.execute_union(action, record, record_map) - } - _ => Err(PipelineError::InvalidOperandType(self.op.to_string())), - } - } - - fn execute_union( - &self, - action: SetAction, - record: Record, - record_map: &mut CountingRecordMapEnum, - ) -> Result, PipelineError> { - match action { - SetAction::Insert => self.union_insert(action, record, record_map), - SetAction::Delete => self.union_delete(action, record, record_map), - } - } - - fn union_insert( - &self, - action: SetAction, - record: Record, - record_map: &mut CountingRecordMapEnum, - ) -> Result, PipelineError> { - let _count = self.update_map(record.clone(), false, record_map); - if _count == 1 { - Ok(vec![(action, record)]) - } else { - Ok(vec![]) - } - } - - fn union_delete( - &self, - action: SetAction, - record: Record, - record_map: &mut CountingRecordMapEnum, - ) -> Result, PipelineError> { - let _count = self.update_map(record.clone(), true, record_map); - if _count == 0 { - Ok(vec![(action, record)]) - } else { - Ok(vec![]) - } - } - - fn update_map( - &self, - record: Record, - decr: bool, - record_map: &mut CountingRecordMapEnum, - ) -> u64 { - if decr { - record_map.remove(&record); - } else { - record_map.insert(&record); - } - - record_map.estimate_count(&record) - } -} diff --git a/dozer-sql/src/product/set/record_map/bloom.rs b/dozer-sql/src/product/set/record_map/bloom.rs deleted file mode 100644 index c27b94d930..0000000000 --- a/dozer-sql/src/product/set/record_map/bloom.rs +++ /dev/null @@ -1,196 +0,0 @@ -//! Based on https://www.arunma.com/2023/03/19/build-your-own-counting-bloom-filter-in-rust/ - -use std::hash::Hash; - -use dozer_types::serde::{Deserialize, Serialize}; - -#[derive(Debug, Serialize, Deserialize)] -#[serde(crate = "dozer_types::serde")] -pub struct CountingBloomFilter { - #[serde(with = "dozer_types::serde_bytes")] - counters: Vec, - num_hashes: u32, - #[serde(skip)] - hasher: hash::BloomHasher, -} - -impl CountingBloomFilter { - pub fn with_rate(false_positive_rate: f32, expected_num_items: u32) -> Self { - let num_counters = optimal_num_counters(expected_num_items, false_positive_rate); - let num_hashes = optimal_num_hashes(expected_num_items, num_counters); - Self { - counters: vec![0; num_counters], - num_hashes, - hasher: Default::default(), - } - } - - pub fn insert(&mut self, value: &V) { - for slot in calculate_slots(&self.hasher, value, self.num_hashes, self.counters.len()) { - self.counters[slot] = self.counters[slot].saturating_add(1); - } - } - - pub fn remove(&mut self, value: &V) { - let slots = calculate_slots(&self.hasher, value, self.num_hashes, self.counters.len()) - .collect::>(); - if slots.iter().all(|slot| self.counters[*slot] > 0) { - for slot in slots { - self.counters[slot] = self.counters[slot].saturating_sub(1); - } - } - } - - pub fn estimate_count(&self, value: &V) -> u8 { - calculate_slots(&self.hasher, value, self.num_hashes, self.counters.len()) - .map(|slot| self.counters[slot]) - .min() - .unwrap_or(0) - } - - pub fn clear(&mut self) { - self.counters.iter_mut().for_each(|counter| *counter = 0); - } -} - -fn optimal_num_counters(num_items: u32, false_positive_rate: f32) -> usize { - -(num_items as f64 * (false_positive_rate as f64).ln() / (2.0f64.ln().powi(2))).ceil() as usize -} - -fn optimal_num_hashes(num_items: u32, num_counters: usize) -> u32 { - let k = (num_counters as f64 / num_items as f64 * 2.0f64.ln()).round() as u32; - if k < 1 { - 1 - } else { - k - } -} - -fn calculate_slots( - hasher: &hash::BloomHasher, - value: &V, - num_hashes: u32, - num_counters: usize, -) -> impl Iterator { - hasher - .calculate_hashes(value, num_hashes) - .map(move |hash| (hash % num_counters as u64) as usize) -} - -mod hash { - use std::hash::{BuildHasher, Hash}; - - use ahash::RandomState; - - #[derive(Debug)] - pub struct BloomHasher { - random_state_1: RandomState, - random_state_2: RandomState, - } - - impl Default for BloomHasher { - fn default() -> Self { - // We're using fixed keys because `RandomState` cannot be serialized. These are just two random numbers. - const KEY1: usize = 11636376767615148353; - const KEY2: usize = 1474968174732524820; - let random_state_1 = RandomState::with_seed(KEY1); - let random_state_2 = RandomState::with_seed(KEY2); - Self { - random_state_1, - random_state_2, - } - } - } - - impl BloomHasher { - pub fn calculate_hashes( - &self, - value: &impl Hash, - num_hashes: u32, - ) -> impl Iterator { - calculate_hashes_impl( - value, - num_hashes, - &self.random_state_1, - &self.random_state_2, - ) - } - } - - /// From paper "Less Hashing, Same Performance: Building a Better Bloom Filter" - fn calculate_hashes_impl( - value: &T, - num_hashes: u32, - hasher_builder1: &impl BuildHasher, - hasher_builder2: &impl BuildHasher, - ) -> impl Iterator { - let hash1 = hasher_builder1.hash_one(value); - let hash2 = hasher_builder2.hash_one(value); - (0..num_hashes).map(move |i| hash1.wrapping_add((i as u64).wrapping_mul(hash2))) - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn bloom_hashes_consistently() { - let hasher1 = hash::BloomHasher::default(); - let hasher2 = hash::BloomHasher::default(); - - for value in &["foo", "bar", "baz"] { - let hashes1 = hasher1.calculate_hashes(value, 3).collect::>(); - let hashes2 = hasher2.calculate_hashes(value, 3).collect::>(); - assert_eq!(hashes1, hashes2); - } - } - - #[test] - fn test_counting_bloom_filter() { - let mut filter = CountingBloomFilter::with_rate(0.01, 100); - assert_eq!(filter.estimate_count(&"foo"), 0); - assert_eq!(filter.estimate_count(&"bar"), 0); - assert_eq!(filter.estimate_count(&"baz"), 0); - - filter.insert(&"foo"); - assert_eq!(filter.estimate_count(&"foo"), 1); - assert_eq!(filter.estimate_count(&"bar"), 0); - assert_eq!(filter.estimate_count(&"baz"), 0); - - filter.insert(&"foo"); - assert_eq!(filter.estimate_count(&"foo"), 2); - assert_eq!(filter.estimate_count(&"bar"), 0); - assert_eq!(filter.estimate_count(&"baz"), 0); - - filter.insert(&"bar"); - assert_eq!(filter.estimate_count(&"foo"), 2); - assert_eq!(filter.estimate_count(&"bar"), 1); - assert_eq!(filter.estimate_count(&"baz"), 0); - - filter.insert(&"baz"); - assert_eq!(filter.estimate_count(&"foo"), 2); - assert_eq!(filter.estimate_count(&"bar"), 1); - assert_eq!(filter.estimate_count(&"baz"), 1); - - filter.remove(&"foo"); - assert_eq!(filter.estimate_count(&"foo"), 1); - assert_eq!(filter.estimate_count(&"bar"), 1); - assert_eq!(filter.estimate_count(&"baz"), 1); - - filter.remove(&"foo"); - assert_eq!(filter.estimate_count(&"foo"), 0); - assert_eq!(filter.estimate_count(&"bar"), 1); - assert_eq!(filter.estimate_count(&"baz"), 1); - - filter.remove(&"bar"); - assert_eq!(filter.estimate_count(&"foo"), 0); - assert_eq!(filter.estimate_count(&"bar"), 0); - assert_eq!(filter.estimate_count(&"baz"), 1); - - filter.remove(&"baz"); - assert_eq!(filter.estimate_count(&"foo"), 0); - assert_eq!(filter.estimate_count(&"bar"), 0); - assert_eq!(filter.estimate_count(&"baz"), 0); - } -} diff --git a/dozer-sql/src/product/set/record_map/mod.rs b/dozer-sql/src/product/set/record_map/mod.rs deleted file mode 100644 index 1d4119e18f..0000000000 --- a/dozer-sql/src/product/set/record_map/mod.rs +++ /dev/null @@ -1,159 +0,0 @@ -use dozer_types::{ - errors::types::DeserializationError, - serde::{Deserialize, Serialize}, - types::Record, -}; -use enum_dispatch::enum_dispatch; -use std::collections::HashMap; - -#[enum_dispatch(CountingRecordMap)] -pub enum CountingRecordMapEnum { - AccurateCountingRecordMap, - ProbabilisticCountingRecordMap, -} - -#[enum_dispatch] -pub trait CountingRecordMap { - /// Inserts a record, or increases its insertion count if it already exixts in the map. - fn insert(&mut self, record: &Record); - - /// Decreases the insertion count of a record, and removes it if the count reaches zero. - fn remove(&mut self, record: &Record); - - /// Returns an estimate of the number of times this record has been inserted into the filter. - /// Depending on the implementation, this number may not be accurate. - fn estimate_count(&self, record: &Record) -> u64; - - /// Clears the map, removing all records. - fn clear(&mut self); -} - -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct AccurateCountingRecordMap { - map: HashMap, -} - -impl AccurateCountingRecordMap { - pub fn new() -> Result { - Ok(Self { - map: Default::default(), - }) - } -} - -impl CountingRecordMap for AccurateCountingRecordMap { - fn insert(&mut self, record: &Record) { - let count = self.map.entry(record.clone()).or_insert(0); - if *count < u64::max_value() { - *count += 1; - } - } - - fn remove(&mut self, record: &Record) { - if let Some(count) = self.map.get_mut(record) { - *count -= 1; - if *count == 0 { - self.map.remove(record); - } - } - } - - fn estimate_count(&self, record: &Record) -> u64 { - self.map.get(record).copied().unwrap_or(0) - } - - fn clear(&mut self) { - self.map.clear(); - } -} - -#[derive(Debug, Serialize, Deserialize)] -#[serde(crate = "dozer_types::serde")] -pub struct ProbabilisticCountingRecordMap { - map: bloom::CountingBloomFilter, -} - -impl ProbabilisticCountingRecordMap { - const FALSE_POSITIVE_RATE: f32 = 0.01; - const EXPECTED_NUM_ITEMS: u32 = 10000000; - - pub fn new() -> Result { - Ok(Self { - map: bloom::CountingBloomFilter::with_rate( - Self::FALSE_POSITIVE_RATE, - Self::EXPECTED_NUM_ITEMS, - ), - }) - } -} - -impl CountingRecordMap for ProbabilisticCountingRecordMap { - fn insert(&mut self, record: &Record) { - self.map.insert(record); - } - - fn remove(&mut self, record: &Record) { - self.map.remove(record); - } - - fn estimate_count(&self, record: &Record) -> u64 { - self.map.estimate_count(record) as u64 - } - - fn clear(&mut self) { - self.map.clear(); - } -} - -mod bloom; - -#[cfg(test)] -mod tests { - use dozer_types::types::{Field, Record}; - - use super::{ - AccurateCountingRecordMap, CountingRecordMap, CountingRecordMapEnum, - ProbabilisticCountingRecordMap, - }; - - fn test_map(mut map: CountingRecordMapEnum) { - let make_record = Record::new; - - let a = make_record(vec![Field::String('a'.into())]); - let b = make_record(vec![Field::String('b'.into())]); - - assert_eq!(map.estimate_count(&a), 0); - assert_eq!(map.estimate_count(&b), 0); - - map.insert(&a); - map.insert(&b); - assert_eq!(map.estimate_count(&a), 1); - assert_eq!(map.estimate_count(&b), 1); - - map.insert(&b); - map.insert(&b); - assert_eq!(map.estimate_count(&a), 1); - assert_eq!(map.estimate_count(&b), 3); - - map.remove(&b); - assert_eq!(map.estimate_count(&a), 1); - assert_eq!(map.estimate_count(&b), 2); - - map.remove(&a); - assert_eq!(map.estimate_count(&a), 0); - assert_eq!(map.estimate_count(&b), 2); - - map.clear(); - assert_eq!(map.estimate_count(&a), 0); - assert_eq!(map.estimate_count(&b), 0); - } - - #[test] - fn test_maps() { - let accurate_map = AccurateCountingRecordMap::new().unwrap().into(); - test_map(accurate_map); - - let probabilistic_map = ProbabilisticCountingRecordMap::new().unwrap().into(); - test_map(probabilistic_map); - } -} diff --git a/dozer-sql/src/product/set/set_factory.rs b/dozer-sql/src/product/set/set_factory.rs deleted file mode 100644 index 6f4fd442f4..0000000000 --- a/dozer-sql/src/product/set/set_factory.rs +++ /dev/null @@ -1,116 +0,0 @@ -use std::collections::HashMap; - -use crate::errors::PipelineError; -use crate::errors::SetError; - -use dozer_core::event::EventHub; -use dozer_core::{ - node::{PortHandle, Processor, ProcessorFactory}, - DEFAULT_PORT_HANDLE, -}; -use dozer_sql_expression::sqlparser::ast::{SetOperator, SetQuantifier}; -use dozer_types::errors::internal::BoxedError; -use dozer_types::tonic::async_trait; -use dozer_types::types::{FieldDefinition, Schema, SourceDefinition}; - -use super::operator::SetOperation; -use super::set_processor::SetProcessor; - -#[derive(Debug)] -pub struct SetProcessorFactory { - id: String, - set_quantifier: SetQuantifier, - enable_probabilistic_optimizations: bool, -} - -impl SetProcessorFactory { - /// Creates a new [`FromProcessorFactory`]. - pub fn new( - id: String, - set_quantifier: SetQuantifier, - enable_probabilistic_optimizations: bool, - ) -> Self { - Self { - id, - set_quantifier, - enable_probabilistic_optimizations, - } - } -} - -#[async_trait] -impl ProcessorFactory for SetProcessorFactory { - fn id(&self) -> String { - self.id.clone() - } - - fn type_name(&self) -> String { - "Set".to_string() - } - fn get_input_ports(&self) -> Vec { - vec![0, 1] - } - - fn get_output_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - async fn get_output_schema( - &self, - _output_port: &PortHandle, - input_schemas: &HashMap, - ) -> Result { - let output_columns = validate_set_operation_input_schemas(input_schemas)?; - - let output_schema = Schema { - fields: output_columns, - primary_index: input_schemas[&0].primary_index.clone(), - }; - - Ok(output_schema) - } - - async fn build( - &self, - _input_schemas: HashMap, - _output_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - Ok(Box::new(SetProcessor::new( - self.id.clone(), - SetOperation { - op: SetOperator::Union, - quantifier: self.set_quantifier, - }, - self.enable_probabilistic_optimizations, - )?)) - } -} - -fn validate_set_operation_input_schemas( - input_schemas: &HashMap, -) -> Result, PipelineError> { - let mut left_columns = input_schemas[&0].fields.clone(); - let mut right_columns = input_schemas[&0].fields.clone(); - - left_columns.sort(); - right_columns.sort(); - - let mut output_fields = Vec::new(); - for (left, right) in left_columns.iter().zip(right_columns.iter()) { - if !is_similar_fields(left, right) { - return Err(PipelineError::SetError(SetError::InvalidInputSchemas)); - } - output_fields.push(FieldDefinition::new( - left.name.clone(), - left.typ, - left.nullable, - SourceDefinition::Dynamic, - )); - } - Ok(output_fields) -} - -fn is_similar_fields(left: &FieldDefinition, right: &FieldDefinition) -> bool { - left.name == right.name && left.typ == right.typ && left.nullable == right.nullable -} diff --git a/dozer-sql/src/product/set/set_processor.rs b/dozer-sql/src/product/set/set_processor.rs deleted file mode 100644 index 3a273e1c11..0000000000 --- a/dozer-sql/src/product/set/set_processor.rs +++ /dev/null @@ -1,186 +0,0 @@ -use super::operator::{SetAction, SetOperation}; -use super::record_map::{ - AccurateCountingRecordMap, CountingRecordMapEnum, ProbabilisticCountingRecordMap, -}; -use crate::errors::{PipelineError, ProductError, SetError}; -use dozer_core::channels::ProcessorChannelForwarder; -use dozer_core::epoch::Epoch; -use dozer_core::node::Processor; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_types::errors::internal::BoxedError; -use dozer_types::types::{Operation, Record, TableOperation}; -use std::fmt::{Debug, Formatter}; - -pub struct SetProcessor { - _id: String, - /// Set operations - operator: SetOperation, - /// Hashmap containing records with its occurrence - record_map: CountingRecordMapEnum, -} - -impl SetProcessor { - /// Creates a new [`SetProcessor`]. - pub fn new( - id: String, - operator: SetOperation, - enable_probabilistic_optimizations: bool, - ) -> Result { - Ok(Self { - _id: id, - operator, - record_map: if enable_probabilistic_optimizations { - ProbabilisticCountingRecordMap::new()?.into() - } else { - AccurateCountingRecordMap::new()?.into() - }, - }) - } - - fn delete(&mut self, record: Record) -> Result, ProductError> { - self.operator - .execute(SetAction::Delete, record, &mut self.record_map) - .map_err(|err| { - ProductError::DeleteError("UNION query error:".to_string(), Box::new(err)) - }) - } - - fn insert(&mut self, record: Record) -> Result, ProductError> { - self.operator - .execute(SetAction::Insert, record, &mut self.record_map) - .map_err(|err| { - ProductError::InsertError("UNION query error:".to_string(), Box::new(err)) - }) - } - - #[allow(clippy::type_complexity)] - fn update( - &mut self, - old: Record, - new: Record, - ) -> Result<(Vec<(SetAction, Record)>, Vec<(SetAction, Record)>), ProductError> { - let old_records = self - .operator - .execute(SetAction::Delete, old, &mut self.record_map) - .map_err(|err| { - ProductError::UpdateOldError("UNION query error:".to_string(), Box::new(err)) - })?; - - let new_records = self - .operator - .execute(SetAction::Insert, new, &mut self.record_map) - .map_err(|err| { - ProductError::UpdateNewError("UNION query error:".to_string(), Box::new(err)) - })?; - - Ok((old_records, new_records)) - } -} - -impl Debug for SetProcessor { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - f.debug_tuple("SetProcessor").field(&self.operator).finish() - } -} - -impl Processor for SetProcessor { - fn commit(&self, _epoch: &Epoch) -> Result<(), BoxedError> { - Ok(()) - } - - fn process( - &mut self, - op: TableOperation, - fw: &mut dyn ProcessorChannelForwarder, - ) -> Result<(), BoxedError> { - match op.op { - Operation::Delete { old } => { - let records = self.delete(old).map_err(PipelineError::ProductError)?; - - for (action, record) in records.into_iter() { - match action { - SetAction::Insert => { - fw.send(TableOperation::without_id( - Operation::Insert { new: record }, - DEFAULT_PORT_HANDLE, - )); - } - SetAction::Delete => { - fw.send(TableOperation::without_id( - Operation::Delete { old: record }, - DEFAULT_PORT_HANDLE, - )); - } - } - } - } - Operation::Insert { new } => { - let records = self.insert(new).map_err(PipelineError::ProductError)?; - - for (action, record) in records.into_iter() { - match action { - SetAction::Insert => { - fw.send(TableOperation::without_id( - Operation::Insert { new: record }, - DEFAULT_PORT_HANDLE, - )); - } - SetAction::Delete => { - fw.send(TableOperation::without_id( - Operation::Delete { old: record }, - DEFAULT_PORT_HANDLE, - )); - } - } - } - } - Operation::Update { old, new } => { - let (old_records, new_records) = - self.update(old, new).map_err(PipelineError::ProductError)?; - - for (action, old) in old_records.into_iter() { - match action { - SetAction::Insert => { - fw.send(TableOperation::without_id( - Operation::Insert { new: old }, - DEFAULT_PORT_HANDLE, - )); - } - SetAction::Delete => { - fw.send(TableOperation::without_id( - Operation::Delete { old }, - DEFAULT_PORT_HANDLE, - )); - } - } - } - - for (action, new) in new_records.into_iter() { - match action { - SetAction::Insert => { - fw.send(TableOperation::without_id( - Operation::Insert { new }, - DEFAULT_PORT_HANDLE, - )); - } - SetAction::Delete => { - fw.send(TableOperation::without_id( - Operation::Delete { old: new }, - DEFAULT_PORT_HANDLE, - )); - } - } - } - } - Operation::BatchInsert { new } => { - for record in new { - self.process( - TableOperation::without_id(Operation::Insert { new: record }, op.port), - fw, - )?; - } - } - } - Ok(()) - } -} diff --git a/dozer-sql/src/product/table/factory.rs b/dozer-sql/src/product/table/factory.rs deleted file mode 100644 index 793a0a7941..0000000000 --- a/dozer-sql/src/product/table/factory.rs +++ /dev/null @@ -1,66 +0,0 @@ -use std::collections::HashMap; - -use dozer_core::{ - event::EventHub, - node::{PortHandle, Processor, ProcessorFactory}, - DEFAULT_PORT_HANDLE, -}; -use dozer_sql_expression::builder::{extend_schema_source_def, NameOrAlias}; -use dozer_types::{errors::internal::BoxedError, tonic::async_trait, types::Schema}; - -use crate::errors::PipelineError; - -use super::processor::TableProcessor; - -#[derive(Debug)] -pub struct TableProcessorFactory { - id: String, - table: NameOrAlias, -} - -impl TableProcessorFactory { - pub fn new(id: String, table: NameOrAlias) -> Self { - Self { id, table } - } -} - -#[async_trait] -impl ProcessorFactory for TableProcessorFactory { - fn id(&self) -> String { - self.id.clone() - } - - fn type_name(&self) -> String { - "Table".to_string() - } - - fn get_input_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - fn get_output_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - async fn get_output_schema( - &self, - _output_port: &PortHandle, - input_schemas: &HashMap, - ) -> Result { - if let Some(input_schema) = input_schemas.get(&DEFAULT_PORT_HANDLE) { - let extended_input_schema = extend_schema_source_def(input_schema, &self.table); - Ok(extended_input_schema) - } else { - Err(PipelineError::InvalidPortHandle(DEFAULT_PORT_HANDLE).into()) - } - } - - async fn build( - &self, - _input_schemas: HashMap, - _output_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - Ok(Box::new(TableProcessor::new(self.id.clone()))) - } -} diff --git a/dozer-sql/src/product/table/mod.rs b/dozer-sql/src/product/table/mod.rs deleted file mode 100644 index 467cebdda3..0000000000 --- a/dozer-sql/src/product/table/mod.rs +++ /dev/null @@ -1,2 +0,0 @@ -pub mod factory; -mod processor; diff --git a/dozer-sql/src/product/table/processor.rs b/dozer-sql/src/product/table/processor.rs deleted file mode 100644 index 0e0eeb1a68..0000000000 --- a/dozer-sql/src/product/table/processor.rs +++ /dev/null @@ -1,33 +0,0 @@ -use dozer_core::channels::ProcessorChannelForwarder; -use dozer_core::epoch::Epoch; -use dozer_core::node::Processor; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_types::errors::internal::BoxedError; -use dozer_types::types::TableOperation; - -#[derive(Debug)] -pub struct TableProcessor { - _id: String, -} - -impl TableProcessor { - pub fn new(id: String) -> Self { - Self { _id: id } - } -} - -impl Processor for TableProcessor { - fn commit(&self, _epoch: &Epoch) -> Result<(), BoxedError> { - Ok(()) - } - - fn process( - &mut self, - mut op: TableOperation, - fw: &mut dyn ProcessorChannelForwarder, - ) -> Result<(), BoxedError> { - op.port = DEFAULT_PORT_HANDLE; - fw.send(op); - Ok(()) - } -} diff --git a/dozer-sql/src/projection/factory.rs b/dozer-sql/src/projection/factory.rs deleted file mode 100644 index ea835febb9..0000000000 --- a/dozer-sql/src/projection/factory.rs +++ /dev/null @@ -1,175 +0,0 @@ -use std::{collections::HashMap, sync::Arc}; - -use dozer_core::{ - event::EventHub, - node::{PortHandle, Processor, ProcessorFactory}, - DEFAULT_PORT_HANDLE, -}; -use dozer_sql_expression::{ - builder::ExpressionBuilder, - execution::Expression, - sqlparser::ast::{Expr, Ident, SelectItem}, -}; -use dozer_types::{ - errors::internal::BoxedError, - types::{FieldDefinition, Schema}, -}; -use dozer_types::{models::udf_config::UdfConfig, tonic::async_trait}; -use tokio::runtime::Runtime; - -use crate::errors::PipelineError; - -use super::processor::ProjectionProcessor; - -#[derive(Debug)] -pub struct ProjectionProcessorFactory { - select: Vec, - id: String, - udfs: Vec, - runtime: Arc, -} - -impl ProjectionProcessorFactory { - /// Creates a new [`ProjectionProcessorFactory`]. - pub fn _new( - id: String, - select: Vec, - udfs: Vec, - runtime: Arc, - ) -> Self { - Self { - select, - id, - udfs, - runtime, - } - } -} - -#[async_trait] -impl ProcessorFactory for ProjectionProcessorFactory { - fn id(&self) -> String { - self.id.clone() - } - fn type_name(&self) -> String { - "Projection".to_string() - } - - fn get_input_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - fn get_output_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - async fn get_output_schema( - &self, - _output_port: &PortHandle, - input_schemas: &HashMap, - ) -> Result { - let input_schema = input_schemas.get(&DEFAULT_PORT_HANDLE).unwrap(); - - let mut select_expr: Vec<(String, Expression)> = vec![]; - for s in self.select.iter() { - match s { - SelectItem::Wildcard(_) => { - let fields: Vec = input_schema - .fields - .iter() - .map(|col| { - SelectItem::UnnamedExpr(Expr::Identifier(Ident::new( - col.to_owned().name, - ))) - }) - .collect(); - for f in fields { - if let Ok(res) = parse_sql_select_item( - &f, - input_schema, - &self.udfs, - self.runtime.clone(), - ) - .await - { - select_expr.push(res) - } - } - } - _ => { - if let Ok(res) = - parse_sql_select_item(s, input_schema, &self.udfs, self.runtime.clone()) - .await - { - select_expr.push(res) - } - } - } - } - - let mut output_schema = input_schema.clone(); - let mut fields = vec![]; - for e in select_expr.iter() { - let field_name = e.0.clone(); - let field_type = e.1.get_type(input_schema)?; - fields.push(FieldDefinition::new( - field_name, - field_type.return_type, - field_type.nullable, - field_type.source, - )); - } - output_schema.fields = fields; - - Ok(output_schema) - } - - async fn build( - &self, - input_schemas: HashMap, - _output_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - let schema = match input_schemas.get(&DEFAULT_PORT_HANDLE) { - Some(schema) => Ok(schema), - None => Err(PipelineError::InvalidPortHandle(DEFAULT_PORT_HANDLE)), - }?; - - let mut expressions = vec![]; - for select in &self.select { - expressions.push( - parse_sql_select_item(select, schema, &self.udfs, self.runtime.clone()).await?, - ); - } - Ok(Box::new(ProjectionProcessor::new( - schema.clone(), - expressions.into_iter().map(|e| e.1).collect(), - )?)) - } -} - -pub(crate) async fn parse_sql_select_item( - sql: &SelectItem, - schema: &Schema, - udfs: &[UdfConfig], - runtime: Arc, -) -> Result<(String, Expression), PipelineError> { - match sql { - SelectItem::UnnamedExpr(sql_expr) => { - let expr = ExpressionBuilder::new(0, runtime) - .parse_sql_expression(true, sql_expr, schema, udfs) - .await?; - Ok((sql_expr.to_string(), expr)) - } - SelectItem::ExprWithAlias { expr, alias } => { - let expr = ExpressionBuilder::new(0, runtime) - .parse_sql_expression(true, expr, schema, udfs) - .await?; - Ok((alias.value.clone(), expr)) - } - SelectItem::Wildcard(_) => Err(PipelineError::InvalidOperator("*".to_string())), - SelectItem::QualifiedWildcard(ref object_name, ..) => { - Err(PipelineError::InvalidOperator(object_name.to_string())) - } - } -} diff --git a/dozer-sql/src/projection/mod.rs b/dozer-sql/src/projection/mod.rs deleted file mode 100644 index 9a12ba12cc..0000000000 --- a/dozer-sql/src/projection/mod.rs +++ /dev/null @@ -1,2 +0,0 @@ -pub mod factory; -pub mod processor; diff --git a/dozer-sql/src/projection/processor.rs b/dozer-sql/src/projection/processor.rs deleted file mode 100644 index b83b131285..0000000000 --- a/dozer-sql/src/projection/processor.rs +++ /dev/null @@ -1,101 +0,0 @@ -use crate::errors::PipelineError; -use dozer_sql_expression::execution::Expression; - -use dozer_core::channels::ProcessorChannelForwarder; -use dozer_core::epoch::Epoch; -use dozer_core::node::Processor; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_types::errors::internal::BoxedError; -use dozer_types::types::{Operation, Record, Schema, TableOperation}; - -#[derive(Debug)] -pub struct ProjectionProcessor { - expressions: Vec, - input_schema: Schema, -} - -impl ProjectionProcessor { - pub fn new(input_schema: Schema, expressions: Vec) -> Result { - Ok(Self { - input_schema, - expressions, - }) - } - - fn delete(&mut self, record: &Record) -> Result { - let mut results = vec![]; - - for expr in &mut self.expressions { - results.push(expr.evaluate(record, &self.input_schema)?); - } - - let mut output_record = Record::new(results); - output_record.set_lifetime(record.lifetime.to_owned()); - - Ok(Operation::Delete { old: output_record }) - } - - fn insert(&mut self, record: &Record) -> Result { - let mut results = vec![]; - - for expr in &mut self.expressions { - results.push(expr.evaluate(record, &self.input_schema)?); - } - - let mut output_record = Record::new(results); - output_record.set_lifetime(record.lifetime.to_owned()); - Ok(output_record) - } - - fn update(&mut self, old: &Record, new: &Record) -> Result { - let mut old_results = vec![]; - let mut new_results = vec![]; - - for expr in &mut self.expressions { - old_results.push(expr.evaluate(old, &self.input_schema)?); - new_results.push(expr.evaluate(new, &self.input_schema)?); - } - - let mut old_output_record = Record::new(old_results); - old_output_record.set_lifetime(old.lifetime.to_owned()); - let mut new_output_record = Record::new(new_results); - new_output_record.set_lifetime(new.lifetime.to_owned()); - Ok(Operation::Update { - old: old_output_record, - new: new_output_record, - }) - } -} - -impl Processor for ProjectionProcessor { - fn process( - &mut self, - op: TableOperation, - fw: &mut dyn ProcessorChannelForwarder, - ) -> Result<(), BoxedError> { - let output_op = match op.op { - Operation::Delete { ref old } => self.delete(old)?, - Operation::Insert { ref new } => Operation::Insert { - new: self.insert(new)?, - }, - Operation::Update { ref old, ref new } => self.update(old, new)?, - Operation::BatchInsert { new } => { - let records = new - .iter() - .map(|record| self.insert(record)) - .collect::, _>>()?; - Operation::BatchInsert { new: records } - } - }; - fw.send(TableOperation { - id: op.id, - op: output_op, - port: DEFAULT_PORT_HANDLE, - }); - Ok(()) - } - - fn commit(&self, _epoch: &Epoch) -> Result<(), BoxedError> { - Ok(()) - } -} diff --git a/dozer-sql/src/selection/factory.rs b/dozer-sql/src/selection/factory.rs deleted file mode 100644 index ab5567584c..0000000000 --- a/dozer-sql/src/selection/factory.rs +++ /dev/null @@ -1,90 +0,0 @@ -use std::{collections::HashMap, sync::Arc}; - -use crate::errors::PipelineError; -use dozer_core::{ - event::EventHub, - node::{PortHandle, Processor, ProcessorFactory}, - DEFAULT_PORT_HANDLE, -}; -use dozer_sql_expression::builder::ExpressionBuilder; -use dozer_sql_expression::sqlparser::ast::Expr as SqlExpr; -use dozer_types::{errors::internal::BoxedError, types::Schema}; -use dozer_types::{models::udf_config::UdfConfig, tonic::async_trait}; -use tokio::runtime::Runtime; - -use super::processor::SelectionProcessor; - -#[derive(Debug)] -pub struct SelectionProcessorFactory { - statement: SqlExpr, - id: String, - udfs: Vec, - runtime: Arc, -} - -impl SelectionProcessorFactory { - /// Creates a new [`SelectionProcessorFactory`]. - pub fn new( - id: String, - statement: SqlExpr, - udf_config: Vec, - runtime: Arc, - ) -> Self { - Self { - statement, - id, - udfs: udf_config, - runtime, - } - } -} - -#[async_trait] -impl ProcessorFactory for SelectionProcessorFactory { - fn id(&self) -> String { - self.id.clone() - } - fn type_name(&self) -> String { - "Selection".to_string() - } - fn get_input_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - fn get_output_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - async fn get_output_schema( - &self, - _output_port: &PortHandle, - input_schemas: &HashMap, - ) -> Result { - let schema = input_schemas - .get(&DEFAULT_PORT_HANDLE) - .ok_or(PipelineError::InvalidPortHandle(DEFAULT_PORT_HANDLE))?; - Ok(schema.clone()) - } - - async fn build( - &self, - input_schemas: HashMap, - _output_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - let schema = input_schemas - .get(&DEFAULT_PORT_HANDLE) - .ok_or(PipelineError::InvalidPortHandle(DEFAULT_PORT_HANDLE))?; - - match ExpressionBuilder::new(schema.fields.len(), self.runtime.clone()) - .build(false, &self.statement, schema, &self.udfs) - .await - { - Ok(expression) => Ok(Box::new(SelectionProcessor::new( - schema.clone(), - expression, - )?)), - Err(e) => Err(e.into()), - } - } -} diff --git a/dozer-sql/src/selection/mod.rs b/dozer-sql/src/selection/mod.rs deleted file mode 100644 index 9a12ba12cc..0000000000 --- a/dozer-sql/src/selection/mod.rs +++ /dev/null @@ -1,2 +0,0 @@ -pub mod factory; -pub mod processor; diff --git a/dozer-sql/src/selection/processor.rs b/dozer-sql/src/selection/processor.rs deleted file mode 100644 index 8881286e86..0000000000 --- a/dozer-sql/src/selection/processor.rs +++ /dev/null @@ -1,106 +0,0 @@ -use dozer_core::channels::ProcessorChannelForwarder; -use dozer_core::epoch::Epoch; -use dozer_core::node::Processor; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_sql_expression::execution::Expression; -use dozer_types::errors::internal::BoxedError; -use dozer_types::types::{Field, Operation, Record, Schema, TableOperation}; - -use crate::errors::PipelineError; - -#[derive(Debug)] -pub struct SelectionProcessor { - expression: Expression, - input_schema: Schema, -} - -impl SelectionProcessor { - pub fn new(input_schema: Schema, expression: Expression) -> Result { - Ok(Self { - input_schema, - expression, - }) - } - - fn filter(&mut self, record: &Record) -> Result { - Ok(self.expression.evaluate(record, &self.input_schema)? == Field::Boolean(true)) - } -} - -impl Processor for SelectionProcessor { - fn commit(&self, _epoch: &Epoch) -> Result<(), BoxedError> { - Ok(()) - } - - fn process( - &mut self, - mut op: TableOperation, - fw: &mut dyn ProcessorChannelForwarder, - ) -> Result<(), BoxedError> { - match op.op { - Operation::Delete { ref old } => { - if self.filter(old)? { - op.port = DEFAULT_PORT_HANDLE; - fw.send(op); - } - } - Operation::Insert { ref new } => { - if self.filter(new)? { - op.port = DEFAULT_PORT_HANDLE; - fw.send(op); - } - } - Operation::Update { old, new } => { - let old_fulfilled = self.filter(&old)?; - let new_fulfilled = self.filter(&new)?; - match (old_fulfilled, new_fulfilled) { - (true, true) => { - // both records fulfills the WHERE condition, forward the operation - fw.send(TableOperation { - id: op.id, - op: Operation::Update { old, new }, - port: DEFAULT_PORT_HANDLE, - }); - } - (true, false) => { - // the old record fulfills the WHERE condition while then new one doesn't, forward a delete operation - fw.send(TableOperation { - id: op.id, - op: Operation::Delete { old }, - port: DEFAULT_PORT_HANDLE, - }); - } - (false, true) => { - // the old record doesn't fulfill the WHERE condition while then new one does, forward an insert operation - fw.send(TableOperation { - id: op.id, - op: Operation::Insert { new }, - port: DEFAULT_PORT_HANDLE, - }); - } - (false, false) => { - // both records doesn't fulfill the WHERE condition, don't forward the operation - } - } - } - Operation::BatchInsert { new } => { - let records = new - .into_iter() - .filter_map(|record| { - self.filter(&record) - .map(|fulfilled| if fulfilled { Some(record) } else { None }) - .transpose() - }) - .collect::, _>>()?; - if !records.is_empty() { - fw.send(TableOperation { - id: op.id, - op: Operation::BatchInsert { new: records }, - port: DEFAULT_PORT_HANDLE, - }); - } - } - } - Ok(()) - } -} diff --git a/dozer-sql/src/table_operator/factory.rs b/dozer-sql/src/table_operator/factory.rs deleted file mode 100644 index d97b32c7e9..0000000000 --- a/dozer-sql/src/table_operator/factory.rs +++ /dev/null @@ -1,356 +0,0 @@ -use std::{collections::HashMap, sync::Arc, time::Duration}; - -use dozer_core::{ - event::EventHub, - node::{PortHandle, Processor, ProcessorFactory}, - DEFAULT_PORT_HANDLE, -}; -use dozer_sql_expression::{ - builder::ExpressionBuilder, - execution::Expression, - sqlparser::ast::{Expr, FunctionArg, FunctionArgExpr, Value}, -}; -use dozer_types::{errors::internal::BoxedError, types::Schema}; -use dozer_types::{models::udf_config::UdfConfig, tonic::async_trait}; -use tokio::runtime::Runtime; - -use crate::{ - builder::{TableOperatorArg, TableOperatorDescriptor}, - errors::{PipelineError, TableOperatorError}, -}; - -use super::{ - lifetime::LifetimeTableOperator, - operator::{TableOperator, TableOperatorType}, - processor::TableOperatorProcessor, -}; - -const _SOURCE_TABLE_ARGUMENT: usize = 0; - -#[derive(Debug)] -pub struct TableOperatorProcessorFactory { - id: String, - table: TableOperatorDescriptor, - name: String, - udfs: Vec, - runtime: Arc, -} - -impl TableOperatorProcessorFactory { - pub fn new( - id: String, - table: TableOperatorDescriptor, - udfs: Vec, - runtime: Arc, - ) -> Self { - Self { - id: id.clone(), - table, - name: id, - udfs, - runtime, - } - } -} - -#[async_trait] -impl ProcessorFactory for TableOperatorProcessorFactory { - fn id(&self) -> String { - self.id.clone() - } - - fn type_name(&self) -> String { - self.name.clone() - } - fn get_input_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - fn get_output_ports(&self) -> Vec { - vec![DEFAULT_PORT_HANDLE] - } - - async fn get_output_schema( - &self, - _output_port: &PortHandle, - input_schemas: &HashMap, - ) -> Result { - let input_schema = input_schemas - .get(&DEFAULT_PORT_HANDLE) - .ok_or(PipelineError::InvalidPortHandle(DEFAULT_PORT_HANDLE))?; - - let output_schema = - match operator_from_descriptor( - &self.table, - input_schema, - &self.udfs, - self.runtime.clone(), - ) - .await? - { - Some(operator) => operator - .get_output_schema(input_schema) - .map_err(PipelineError::TableOperatorError)?, - None => { - return Err(PipelineError::TableOperatorError( - TableOperatorError::InternalError("Invalid Table Operator".into()), - ) - .into()) - } - }; - - Ok(output_schema) - } - - async fn build( - &self, - input_schemas: HashMap, - _output_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - let input_schema = input_schemas - .get(&DEFAULT_PORT_HANDLE) - .ok_or(PipelineError::InternalError( - "Invalid Window".to_string().into(), - ))? - .clone(); - - match operator_from_descriptor(&self.table, &input_schema, &self.udfs, self.runtime.clone()) - .await? - { - Some(operator) => Ok(Box::new(TableOperatorProcessor::new( - self.id.clone(), - operator, - input_schema, - ))), - None => Err( - PipelineError::TableOperatorError(TableOperatorError::InternalError( - "Invalid Table Operator".into(), - )) - .into(), - ), - } - } -} - -pub(crate) async fn operator_from_descriptor( - descriptor: &TableOperatorDescriptor, - schema: &Schema, - udfs: &[UdfConfig], - runtime: Arc, -) -> Result, PipelineError> { - if &descriptor.name.to_uppercase() == "TTL" { - let operator = lifetime_from_descriptor(descriptor, schema, udfs, runtime).await?; - - Ok(Some(operator.into())) - } else { - Err(PipelineError::InternalError(descriptor.name.clone().into())) - } -} - -async fn lifetime_from_descriptor( - descriptor: &TableOperatorDescriptor, - schema: &Schema, - udfs: &[UdfConfig], - runtime: Arc, -) -> Result { - let table_expression_arg = - descriptor - .args - .get(1) - .ok_or(TableOperatorError::MissingArgument( - descriptor.name.to_owned(), - ))?; - - let expression_arg = if let TableOperatorArg::Argument(argument) = table_expression_arg { - argument - } else { - return Err(TableOperatorError::InvalidReference( - descriptor.name.to_owned(), - format!("{:?}", table_expression_arg), - )); - }; - - let table_duration_arg = descriptor - .args - .get(2) - .ok_or(TableOperatorError::MissingArgument( - descriptor.name.to_owned(), - ))?; - let duration_arg = if let TableOperatorArg::Argument(argument) = table_duration_arg { - argument - } else { - return Err(TableOperatorError::InvalidInterval( - descriptor.name.to_owned(), - format!("{:?}", table_duration_arg), - )); - }; - - let expression = get_expression( - descriptor.name.to_owned(), - expression_arg, - schema, - udfs, - runtime, - ) - .await?; - let duration = get_interval(descriptor.name.to_owned(), duration_arg)?; - - let operator = LifetimeTableOperator::new(None, expression, duration); - - Ok(operator) -} - -fn get_interval( - function_name: String, - interval_arg: &FunctionArg, -) -> Result { - match interval_arg { - FunctionArg::Named { name, arg: _ } => { - let column_name = ExpressionBuilder::normalize_ident(name); - Err(TableOperatorError::InvalidInterval( - column_name, - function_name, - )) - } - FunctionArg::Unnamed(arg_expr) => match arg_expr { - FunctionArgExpr::Expr(expr) => match expr { - Expr::Value(Value::SingleQuotedString(s) | Value::DoubleQuotedString(s)) => { - let interval = - parse_duration_string(function_name.to_owned(), s).map_err(|_| { - TableOperatorError::InvalidInterval(s.to_owned(), function_name) - })?; - Ok(interval) - } - _ => Err(TableOperatorError::InvalidInterval( - expr.to_string(), - function_name, - )), - }, - FunctionArgExpr::QualifiedWildcard(_) => Err(TableOperatorError::InvalidInterval( - "*".to_string(), - function_name, - )), - FunctionArgExpr::Wildcard => Err(TableOperatorError::InvalidInterval( - "*".to_string(), - function_name, - )), - }, - } -} - -async fn get_expression( - function_name: String, - interval_arg: &FunctionArg, - schema: &Schema, - udfs: &[UdfConfig], - runtime: Arc, -) -> Result { - match interval_arg { - FunctionArg::Named { name, arg: _ } => { - let column_name = ExpressionBuilder::normalize_ident(name); - Err(TableOperatorError::InvalidReference( - column_name, - function_name, - )) - } - FunctionArg::Unnamed(arg_expr) => match arg_expr { - FunctionArgExpr::Expr(expr) => { - let mut builder = ExpressionBuilder::new(schema.fields.len(), runtime); - let expression = builder - .build(false, expr, schema, udfs) - .await - .map_err(|_| { - TableOperatorError::InvalidReference(expr.to_string(), function_name) - })?; - - Ok(expression) - } - FunctionArgExpr::QualifiedWildcard(_) => Err(TableOperatorError::InvalidReference( - "*".to_string(), - function_name, - )), - FunctionArgExpr::Wildcard => Err(TableOperatorError::InvalidReference( - "*".to_string(), - function_name, - )), - }, - } -} - -fn parse_duration_string( - function_name: String, - duration_string: &str, -) -> Result { - let duration_string = duration_string - .split_whitespace() - .collect::>() - .join(" "); - - let duration_tokens = duration_string.split(' ').collect::>(); - if duration_tokens.len() != 2 { - return Err(TableOperatorError::InvalidInterval( - duration_string, - function_name, - )); - } - - let duration_value = duration_tokens[0].parse::().map_err(|_| { - TableOperatorError::InvalidInterval(duration_string.to_owned(), function_name.clone()) - })?; - - let duration_unit = duration_tokens[1].to_uppercase(); - - match duration_unit.as_str() { - "MILLISECOND" | "MILLISECONDS" => Ok(Duration::from_millis(duration_value)), - "SECOND" | "SECONDS" => Ok(Duration::from_secs(duration_value)), - "MINUTE" | "MINUTES" => Ok(Duration::from_secs(duration_value * 60)), - "HOUR" | "HOURS" => Ok(Duration::from_secs(duration_value * 60 * 60)), - "DAY" | "DAYS" => Ok(Duration::from_secs(duration_value * 60 * 60 * 24)), - _ => Err(TableOperatorError::InvalidInterval( - duration_string, - function_name, - )), - } -} - -pub(crate) fn get_source_name( - function_name: &String, - arg: &FunctionArg, -) -> Result { - match arg { - FunctionArg::Named { name, arg: _ } => { - let source_name = ExpressionBuilder::normalize_ident(name); - Err(TableOperatorError::InvalidSourceArgument( - source_name, - function_name.to_string(), - )) - } - FunctionArg::Unnamed(arg_expr) => match arg_expr { - FunctionArgExpr::Expr(expr) => match expr { - Expr::Identifier(ident) => { - let source_name = ExpressionBuilder::normalize_ident(ident); - Ok(source_name) - } - Expr::CompoundIdentifier(ident) => { - let source_name = ExpressionBuilder::fullname_from_ident(ident); - Ok(source_name) - } - _ => Err(TableOperatorError::InvalidSourceArgument( - expr.to_string(), - function_name.to_string(), - )), - }, - FunctionArgExpr::QualifiedWildcard(_) => { - Err(TableOperatorError::InvalidSourceArgument( - "*".to_string(), - function_name.to_string(), - )) - } - FunctionArgExpr::Wildcard => Err(TableOperatorError::InvalidSourceArgument( - "*".to_string(), - function_name.to_string(), - )), - }, - } -} diff --git a/dozer-sql/src/table_operator/lifetime.rs b/dozer-sql/src/table_operator/lifetime.rs deleted file mode 100644 index 69dc50c6d7..0000000000 --- a/dozer-sql/src/table_operator/lifetime.rs +++ /dev/null @@ -1,88 +0,0 @@ -use dozer_sql_expression::execution::Expression; -use dozer_types::types::{Field, Lifetime, Record, Schema}; - -use crate::errors::TableOperatorError; - -use super::operator::{TableOperator, TableOperatorType}; - -#[derive(Debug)] -pub struct LifetimeTableOperator { - operator: Option>, - expression: Expression, - duration: std::time::Duration, -} - -impl LifetimeTableOperator { - pub fn new( - operator: Option>, - expression: Expression, - duration: std::time::Duration, - ) -> Self { - Self { - operator, - expression, - duration, - } - } -} - -impl TableOperator for LifetimeTableOperator { - fn get_name(&self) -> String { - "TTL".to_owned() - } - - fn execute( - &mut self, - record: &Record, - schema: &Schema, - ) -> Result, TableOperatorError> { - let mut ttl_records = vec![]; - if let Some(operator) = &mut self.operator { - let operator_records = operator.execute(record, schema)?; - - let schema = operator.get_output_schema(schema)?; - - let reference = match self - .expression - .evaluate(record, &schema) - .map_err(|err| TableOperatorError::InternalError(Box::new(err)))? - { - Field::Timestamp(timestamp) => timestamp, - other => return Err(TableOperatorError::InvalidTtlInputType(other)), - }; - - let lifetime = Some(Lifetime { - reference, - duration: self.duration, - }); - for mut operator_record in operator_records { - operator_record.set_lifetime(lifetime.clone()); - ttl_records.push(operator_record); - } - } else { - let reference = match self - .expression - .evaluate(record, schema) - .map_err(|err| TableOperatorError::InternalError(Box::new(err)))? - { - Field::Timestamp(timestamp) => timestamp, - other => return Err(TableOperatorError::InvalidTtlInputType(other)), - }; - - let lifetime = Some(Lifetime { - reference, - duration: self.duration, - }); - - let mut cloned_record = record.clone(); - cloned_record.set_lifetime(lifetime); - ttl_records.push(cloned_record); - } - - Ok(ttl_records) - } - - fn get_output_schema(&self, schema: &Schema) -> Result { - Ok(schema.clone()) - } -} diff --git a/dozer-sql/src/table_operator/mod.rs b/dozer-sql/src/table_operator/mod.rs deleted file mode 100644 index ebe1dab842..0000000000 --- a/dozer-sql/src/table_operator/mod.rs +++ /dev/null @@ -1,5 +0,0 @@ -pub(crate) mod factory; -mod lifetime; -mod operator; -mod processor; -mod tests; diff --git a/dozer-sql/src/table_operator/operator.rs b/dozer-sql/src/table_operator/operator.rs deleted file mode 100644 index 383088f061..0000000000 --- a/dozer-sql/src/table_operator/operator.rs +++ /dev/null @@ -1,22 +0,0 @@ -use crate::table_operator::lifetime::LifetimeTableOperator; -use dozer_types::types::{Record, Schema}; -use enum_dispatch::enum_dispatch; - -use crate::errors::TableOperatorError; - -#[enum_dispatch] -pub trait TableOperator: Send + Sync { - fn get_name(&self) -> String; - fn execute( - &mut self, - record: &Record, - schema: &Schema, - ) -> Result, TableOperatorError>; - fn get_output_schema(&self, schema: &Schema) -> Result; -} - -#[enum_dispatch(TableOperator)] -#[derive(Debug)] -pub enum TableOperatorType { - LifetimeTableOperator, -} diff --git a/dozer-sql/src/table_operator/processor.rs b/dozer-sql/src/table_operator/processor.rs deleted file mode 100644 index c2f41f936c..0000000000 --- a/dozer-sql/src/table_operator/processor.rs +++ /dev/null @@ -1,104 +0,0 @@ -use dozer_core::channels::ProcessorChannelForwarder; -use dozer_core::epoch::Epoch; -use dozer_core::node::Processor; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_types::errors::internal::BoxedError; -use dozer_types::types::{Operation, Schema, TableOperation}; - -use crate::errors::PipelineError; - -use super::operator::{TableOperator, TableOperatorType}; - -#[derive(Debug)] -pub struct TableOperatorProcessor { - _id: String, - operator: TableOperatorType, - input_schema: Schema, -} - -impl TableOperatorProcessor { - pub fn new(id: String, operator: TableOperatorType, input_schema: Schema) -> Self { - Self { - _id: id, - operator, - input_schema, - } - } -} - -impl Processor for TableOperatorProcessor { - fn commit(&self, _epoch: &Epoch) -> Result<(), BoxedError> { - Ok(()) - } - - fn process( - &mut self, - op: TableOperation, - fw: &mut dyn ProcessorChannelForwarder, - ) -> Result<(), BoxedError> { - match op.op { - Operation::Delete { ref old } => { - let records = self - .operator - .execute(old, &self.input_schema) - .map_err(PipelineError::TableOperatorError)?; - for record in records { - fw.send(TableOperation::without_id( - Operation::Delete { old: record }, - DEFAULT_PORT_HANDLE, - )); - } - } - Operation::Insert { ref new } => { - let records = self - .operator - .execute(new, &self.input_schema) - .map_err(PipelineError::TableOperatorError)?; - for record in records { - fw.send(TableOperation::without_id( - Operation::Insert { new: record }, - DEFAULT_PORT_HANDLE, - )); - } - } - Operation::Update { ref old, ref new } => { - let old_records = self - .operator - .execute(old, &self.input_schema) - .map_err(PipelineError::TableOperatorError)?; - for record in old_records { - fw.send(TableOperation::without_id( - Operation::Delete { old: record }, - DEFAULT_PORT_HANDLE, - )); - } - - let new_records = self - .operator - .execute(new, &self.input_schema) - .map_err(PipelineError::TableOperatorError)?; - for record in new_records { - fw.send(TableOperation::without_id( - Operation::Insert { new: record }, - DEFAULT_PORT_HANDLE, - )); - } - } - Operation::BatchInsert { new } => { - let mut records = vec![]; - for record in new { - records.extend( - self.operator - .execute(&record, &self.input_schema) - .map_err(PipelineError::TableOperatorError)?, - ); - } - fw.send(TableOperation::without_id( - Operation::BatchInsert { new: records }, - DEFAULT_PORT_HANDLE, - )); - } - } - Ok(()) - } -} diff --git a/dozer-sql/src/table_operator/tests/mod.rs b/dozer-sql/src/table_operator/tests/mod.rs deleted file mode 100644 index 5d901dee2c..0000000000 --- a/dozer-sql/src/table_operator/tests/mod.rs +++ /dev/null @@ -1,2 +0,0 @@ -#[cfg(test)] -mod operator_test; diff --git a/dozer-sql/src/table_operator/tests/operator_test.rs b/dozer-sql/src/table_operator/tests/operator_test.rs deleted file mode 100644 index d360a46c28..0000000000 --- a/dozer-sql/src/table_operator/tests/operator_test.rs +++ /dev/null @@ -1,61 +0,0 @@ -use std::time::Duration; - -use dozer_sql_expression::execution::Expression; -use dozer_types::{ - chrono::DateTime, - types::{Field, FieldDefinition, FieldType, Lifetime, Record, Schema, SourceDefinition}, -}; - -use crate::table_operator::{lifetime::LifetimeTableOperator, operator::TableOperator}; - -#[test] -fn test_lifetime() { - let schema = Schema::default() - .field( - FieldDefinition::new( - "id".to_string(), - FieldType::Int, - false, - SourceDefinition::Alias { - name: "alias".to_string(), - }, - ), - false, - ) - .field( - FieldDefinition::new( - "ref".to_string(), - FieldType::Timestamp, - false, - SourceDefinition::Alias { - name: "alias".to_string(), - }, - ), - false, - ) - .to_owned(); - - let record = Record::new(vec![ - Field::Int(0), - Field::Timestamp(DateTime::parse_from_rfc3339("2020-01-01T00:13:00Z").unwrap()), - ]); - - let mut table_operator = LifetimeTableOperator::new( - None, - Expression::Column { index: 1 }, - Duration::from_secs(60), - ); - - let result = table_operator.execute(&record, &schema).unwrap(); - assert_eq!(result.len(), 1); - let lifetime_record = result.first().unwrap(); - - let mut expected_record = record.clone(); - - expected_record.set_lifetime(Some(Lifetime { - reference: DateTime::parse_from_rfc3339("2020-01-01T00:13:00Z").unwrap(), - duration: Duration::from_secs(60), - })); - - assert_eq!(lifetime_record, &expected_record); -} diff --git a/dozer-sql/src/tests/builder_test.rs b/dozer-sql/src/tests/builder_test.rs deleted file mode 100644 index 4d2b398255..0000000000 --- a/dozer-sql/src/tests/builder_test.rs +++ /dev/null @@ -1,282 +0,0 @@ -use dozer_core::app::{App, AppPipeline}; -use dozer_core::appsource::{AppSourceManager, AppSourceMappings}; -use dozer_core::epoch::Epoch; -use dozer_core::event::EventHub; -use dozer_core::executor::DagExecutor; -use dozer_core::node::{ - OutputPortDef, OutputPortType, PortHandle, Sink, SinkFactory, Source, SourceFactory, -}; -use dozer_core::DEFAULT_PORT_HANDLE; -use dozer_types::chrono::DateTime; -use dozer_types::errors::internal::BoxedError; -use dozer_types::log::debug; -use dozer_types::models::ingestion_types::IngestionMessage; -use dozer_types::node::OpIdentifier; -use dozer_types::ordered_float::OrderedFloat; -use dozer_types::tonic::async_trait; -use dozer_types::types::{ - Field, FieldDefinition, FieldType, Operation, Record, Schema, SourceDefinition, TableOperation, -}; -use tokio::sync::mpsc::Sender; - -use std::collections::HashMap; -use std::future::pending; - -use crate::builder::statement_to_pipeline; -use crate::tests::utils::create_test_runtime; - -/// Test Source -#[derive(Debug)] -pub struct TestSourceFactory { - output_ports: Vec, -} - -impl TestSourceFactory { - pub fn new(output_ports: Vec) -> Self { - Self { output_ports } - } -} - -impl SourceFactory for TestSourceFactory { - fn get_output_ports(&self) -> Vec { - self.output_ports - .iter() - .map(|e| OutputPortDef::new(*e, OutputPortType::Stateless)) - .collect() - } - - fn get_output_schema(&self, _port: &PortHandle) -> Result { - Ok(Schema::default() - .field( - FieldDefinition::new( - String::from("CustomerID"), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("Country"), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("Spending"), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - String::from("timestamp"), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .clone()) - } - - fn get_output_port_name(&self, port: &PortHandle) -> String { - format!("port_{}", port) - } - - fn build( - &self, - _output_schemas: HashMap, - _event_hub: EventHub, - _state: Option>, - ) -> Result, BoxedError> { - Ok(Box::new(TestSource {})) - } -} - -#[derive(Debug)] -pub struct TestSource {} - -#[async_trait] -impl Source for TestSource { - async fn serialize_state(&self) -> Result, BoxedError> { - Ok(vec![]) - } - - async fn start( - &mut self, - sender: Sender<(PortHandle, IngestionMessage)>, - _last_checkpoint: Option, - ) -> Result<(), BoxedError> { - for _ in 0..10 { - sender - .send(( - DEFAULT_PORT_HANDLE, - IngestionMessage::OperationEvent { - table_index: 0, - op: Operation::Insert { - new: Record::new(vec![ - Field::Int(0), - Field::String("Italy".to_string()), - Field::Float(OrderedFloat(5.5)), - Field::Timestamp( - DateTime::parse_from_rfc3339("2020-01-01T00:13:00Z").unwrap(), - ), - ]), - }, - id: None, - }, - )) - .await - .unwrap(); - } - Ok(()) - } -} - -#[derive(Debug)] -pub struct TestSinkFactory { - input_ports: Vec, -} - -impl TestSinkFactory { - pub fn new(input_ports: Vec) -> Self { - Self { input_ports } - } -} - -#[async_trait] -impl SinkFactory for TestSinkFactory { - fn get_input_ports(&self) -> Vec { - self.input_ports.clone() - } - - fn get_input_port_name(&self, _port: &PortHandle) -> String { - "test".to_string() - } - - async fn build( - &self, - _input_schemas: HashMap, - _event_hub: EventHub, - ) -> Result, BoxedError> { - Ok(Box::new(TestSink {})) - } - - fn prepare(&self, _input_schemas: HashMap) -> Result<(), BoxedError> { - Ok(()) - } - - fn type_name(&self) -> String { - "test".to_string() - } -} - -#[derive(Debug)] -pub struct TestSink {} - -impl Sink for TestSink { - fn process(&mut self, op: TableOperation) -> Result<(), BoxedError> { - println!("Sink: {:?}", op); - Ok(()) - } - - fn commit(&mut self, _epoch_details: &Epoch) -> Result<(), BoxedError> { - Ok(()) - } - - fn on_source_snapshotting_started( - &mut self, - _connection_name: String, - ) -> Result<(), BoxedError> { - Ok(()) - } - - fn on_source_snapshotting_done( - &mut self, - _connection_name: String, - _id: Option, - ) -> Result<(), BoxedError> { - Ok(()) - } - - fn set_source_state(&mut self, _source_state: &[u8]) -> Result<(), BoxedError> { - Ok(()) - } - - fn get_source_state(&mut self) -> Result>, BoxedError> { - Ok(None) - } - - fn get_latest_op_id(&mut self) -> Result, BoxedError> { - Ok(None) - } -} - -#[test] -fn test_pipeline_builder() { - let mut pipeline = AppPipeline::new_with_default_flags(); - let runtime = create_test_runtime(); - let context = statement_to_pipeline( - "SELECT t.Spending \ - FROM TTL(TUMBLE(users, timestamp, '5 MINUTES'), timestamp, '1 MINUTE') t JOIN users u on t.CustomerID=u.CustomerID \ - WHERE t.Spending >= 1", - &mut pipeline, - Some("results".to_string()), - vec![], - runtime.clone() - ) - .unwrap(); - - let table_info = context.output_tables_map.get("results").unwrap(); - - let mut asm = AppSourceManager::new(); - asm.add( - Box::new(TestSourceFactory::new(vec![DEFAULT_PORT_HANDLE])), - AppSourceMappings::new( - "mem".to_string(), - vec![("users".to_string(), DEFAULT_PORT_HANDLE)] - .into_iter() - .collect(), - ), - ) - .unwrap(); - - pipeline.add_sink( - Box::new(TestSinkFactory::new(vec![DEFAULT_PORT_HANDLE])), - "sink".to_string(), - ); - pipeline.connect_nodes( - table_info.node.clone(), - table_info.port, - "sink".to_string(), - DEFAULT_PORT_HANDLE, - ); - - let mut app = App::new(asm); - app.add_pipeline(pipeline); - - let dag = app.into_dag().unwrap(); - - let now = std::time::Instant::now(); - - let runtime_clone = runtime.clone(); - let handle = runtime.block_on(async move { - DagExecutor::new(dag, Default::default()) - .await - .unwrap() - .start(pending::<()>(), Default::default(), runtime_clone) - .await - .unwrap() - }); - handle.join().unwrap(); - - let elapsed = now.elapsed(); - debug!("Elapsed: {:.2?}", elapsed); -} diff --git a/dozer-sql/src/tests/mod.rs b/dozer-sql/src/tests/mod.rs deleted file mode 100644 index 0ed1750729..0000000000 --- a/dozer-sql/src/tests/mod.rs +++ /dev/null @@ -1,2 +0,0 @@ -mod builder_test; -pub mod utils; diff --git a/dozer-sql/src/tests/utils.rs b/dozer-sql/src/tests/utils.rs deleted file mode 100644 index 8e327c0889..0000000000 --- a/dozer-sql/src/tests/utils.rs +++ /dev/null @@ -1,39 +0,0 @@ -use std::sync::Arc; - -use crate::errors::PipelineError; -use dozer_sql_expression::sqlparser::{ - ast::{Query, Select, SetExpr, Statement}, - dialect::DozerDialect, - parser::Parser, -}; -use tokio::runtime::Runtime; - -pub fn get_select(sql: &str) -> Result, PipelineError> { - let dialect = DozerDialect {}; - - let ast = Parser::parse_sql(&dialect, sql).unwrap(); - - let statement = ast.first().expect("First statement is missing").to_owned(); - if let Statement::Query(query) = statement { - Ok(get_query_select(&query)) - } else { - panic!("this is not supposed to be called"); - } -} - -pub fn get_query_select(query: &Query) -> Box
, -} - -impl S3Storage { - pub fn convert_to_table(&self) -> PrettyTable { - table!( - ["access_key_id", SECRET], - ["secret_access_key", SECRET], - ["region", self.details.region], - ["bucket_name", self.details.bucket_name] - ) - } -} -impl SchemaExample for S3Storage { - fn example() -> Self { - let s3_details = S3Details { - access_key_id: "".to_owned(), - secret_access_key: "".to_owned(), - region: "".to_owned(), - bucket_name: "".to_owned(), - }; - Self { - details: s3_details, - tables: vec![Table { - config: TableConfig::CSV(CsvConfig { - path: "path/to/file".to_owned(), - extension: ".csv".to_owned(), - marker_extension: None, - }), - name: "table_name".to_owned(), - }], - } - } -} - -#[derive(Debug, Serialize, Deserialize, Eq, PartialEq, Clone, Hash, JsonSchema)] -pub struct LocalDetails { - pub path: String, -} - -#[derive(Debug, Serialize, Deserialize, Eq, PartialEq, Clone, Hash, JsonSchema)] -#[schemars(example = "Self::example")] - -pub struct LocalStorage { - pub details: LocalDetails, - - pub tables: Vec
, -} - -impl LocalStorage { - pub fn convert_to_table(&self) -> PrettyTable { - table!(["path", self.details.path]) - } -} - -#[derive(Debug, Serialize, Deserialize, Eq, PartialEq, Clone, Hash, JsonSchema)] -pub struct DeltaTable { - pub path: String, - - pub name: String, -} - -impl DeltaTable { - pub fn convert_to_table(&self) -> PrettyTable { - todo!() - } -} - -#[derive(Debug, Serialize, Deserialize, Eq, PartialEq, Clone, Hash, JsonSchema)] -#[schemars(example = "Self::example")] - -pub struct DeltaLakeConfig { - pub tables: Vec, -} - -#[derive(Debug, Serialize, Deserialize, Eq, PartialEq, Clone, Hash, JsonSchema)] -#[schemars(example = "Self::example")] - -pub struct MongodbConfig { - pub connection_string: String, -} - -#[derive(Debug, Serialize, Deserialize, Eq, PartialEq, Clone, Hash, JsonSchema)] -#[schemars(example = "Self::example")] - -pub struct MySQLConfig { - pub url: String, - - pub server_id: Option, -} - -#[derive(Debug, Serialize, Deserialize, Eq, PartialEq, Clone, Hash, JsonSchema)] -pub struct NestedDozerConfig { - pub url: String, - #[serde(default, skip_serializing_if = "equal_default")] - pub log_options: NestedDozerLogOptions, -} - -#[derive(Debug, Serialize, Deserialize, Eq, PartialEq, Clone, Hash, JsonSchema, Default)] -pub struct NestedDozerLogOptions { - #[serde(skip_serializing_if = "Option::is_none")] - pub batch_size: Option, - - #[serde(skip_serializing_if = "Option::is_none")] - pub timeout_in_millis: Option, - - #[serde(skip_serializing_if = "Option::is_none")] - pub buffer_size: Option, -} - -pub fn default_log_batch_size() -> u32 { - 30 -} - -pub fn default_timeout() -> u32 { - 1000 -} - -pub fn default_buffer_size() -> u32 { - 1000 -} - -pub fn default_snowflake_poll_interval() -> Duration { - Duration::from_secs(60) -} - -impl SchemaExample for MongodbConfig { - fn example() -> Self { - Self { - connection_string: "mongodb://localhost:27017/db_name".to_owned(), - } - } -} - -impl SchemaExample for MySQLConfig { - fn example() -> Self { - Self { - url: "mysql://root:1234@localhost:3306/db_name".to_owned(), - server_id: Some((1).to_owned()), - } - } -} - -impl SchemaExample for GrpcConfig { - fn example() -> Self { - Self { - host: Some("localhost".to_owned()), - port: Some(50051), - schemas: ConfigSchemas::Path("schema.json".to_owned()), - adapter: Some("arrow".to_owned()), - } - } -} - -impl SchemaExample for KafkaConfig { - fn example() -> Self { - Self { - broker: "".to_owned(), - schema_registry_url: Some("".to_owned()), - } - } -} - -impl SchemaExample for DeltaLakeConfig { - fn example() -> Self { - Self { - tables: vec![DeltaTable { - path: "".to_owned(), - name: "".to_owned(), - }], - } - } -} - -impl SchemaExample for LocalStorage { - fn example() -> Self { - Self { - details: LocalDetails { - path: "path".to_owned(), - }, - tables: vec![Table { - config: TableConfig::CSV(CsvConfig { - path: "path/to/table".to_owned(), - extension: ".csv".to_owned(), - marker_extension: None, - }), - name: "table_name".to_owned(), - }], - } - } -} - -impl SchemaExample for SnowflakeConfig { - fn example() -> Self { - Self { - server: "..snowflakecomputing.com".to_owned(), - port: "443".to_owned(), - user: "bob".to_owned(), - password: "password".to_owned(), - database: "database".to_owned(), - schema: "schema".to_owned(), - warehouse: "warehouse".to_owned(), - driver: Some("SnowflakeDSIIDriver".to_owned()), - role: "role".to_owned(), - poll_interval_seconds: None, - } - } -} - -impl SchemaExample for EthConfig { - fn example() -> Self { - let eth_filter = EthFilter { - from_block: Some(0), - to_block: None, - addresses: vec![], - topics: vec![], - }; - Self { - provider: EthProviderConfig::Log(EthLogConfig { - wss_url: "".to_owned(), - filter: Some(eth_filter), - contracts: vec![], - }), - } - } -} - -#[derive(Debug, Serialize, Deserialize, Eq, PartialEq, Clone, Hash, Default, JsonSchema)] -pub struct JavaScriptConfig { - #[serde(skip_serializing_if = "Option::is_none")] - pub bootstrap_path: Option, -} - -pub fn default_bootstrap_path() -> String { - String::from("src/js/bootstrap.js") -} - -#[derive(Debug, Serialize, Deserialize, Eq, PartialEq, Clone, Hash, JsonSchema)] -#[schemars(example = "Self::example")] -pub struct WebhookConfig { - #[serde(skip_serializing_if = "Option::is_none")] - pub host: Option, - - #[serde(skip_serializing_if = "Option::is_none")] - pub port: Option, - - pub endpoints: Vec, -} -#[derive(Debug, Serialize, Deserialize, Eq, PartialEq, Clone, Hash, JsonSchema)] -#[schemars(example = "Self::example")] -pub struct WebhookEndpoint { - pub path: String, - pub verbs: Vec, - pub schema: WebhookConfigSchemas, -} -#[derive(Debug, Serialize, Deserialize, Eq, PartialEq, Clone, Hash, JsonSchema)] -#[schemars(example = "Self::example")] -pub enum WebhookVerb { - POST, // insert - PUT, // update - DELETE, // delete -} - -#[derive(Debug, Serialize, Deserialize, Eq, PartialEq, Clone, Hash, JsonSchema)] -pub enum WebhookConfigSchemas { - Inline(String), - Path(String), -} - -impl SchemaExample for WebhookConfig { - fn example() -> Self { - Self { - host: Some("localhost".to_owned()), - port: Some(50059), - endpoints: vec![WebhookEndpoint::example()], - } - } -} - -impl SchemaExample for WebhookEndpoint { - fn example() -> Self { - let user_schema = r#" - { - "users": { - "schema": { - "fields": [ - { - "name": "id", - "typ": "Int", - "nullable": false - }, - { - "name": "name", - "typ": "String", - "nullable": true - }, - { - "name": "json", - "typ": "Json", - "nullable": true - } - ] - } - } - } - "#; - Self { - path: "/ingest".to_owned(), - verbs: vec![WebhookVerb::POST, WebhookVerb::DELETE], - schema: WebhookConfigSchemas::Inline(user_schema.to_string()), - } - } -} -impl SchemaExample for WebhookVerb { - fn example() -> Self { - Self::POST - } -} - -#[derive(Debug, JsonSchema, Clone, Deserialize, Serialize, Hash, Eq, PartialEq)] -pub struct AerospikeConfig { - pub namespace: String, - pub sets: Vec, -} - -#[derive(Debug, Serialize, Deserialize, Eq, PartialEq, Clone, Hash, JsonSchema)] -pub struct OracleConfig { - pub user: String, - pub password: String, - pub host: String, - pub port: u16, - pub sid: String, - #[serde(skip_serializing_if = "Option::is_none")] - /// Only needed if using pluggable database - pub pdb: Option, - /// The schemas to consider when listing tables. If empty, will list all schemas, which can be slow. - #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub schemas: Vec, - /// Batch size during snapshotting - pub batch_size: Option, - pub replicator: OracleReplicator, -} - -#[derive(Debug, Serialize, Deserialize, Eq, PartialEq, Clone, Copy, Hash, JsonSchema)] -pub enum OracleReplicator { - LogMiner { poll_interval_in_milliseconds: u64 }, - DozerLogReader, -} diff --git a/dozer-types/src/models/json_schema_helper.rs b/dozer-types/src/models/json_schema_helper.rs deleted file mode 100644 index d3fddfea2f..0000000000 --- a/dozer-types/src/models/json_schema_helper.rs +++ /dev/null @@ -1,54 +0,0 @@ -use schemars::{schema::RootSchema, schema_for}; -use serde::{Deserialize, Serialize}; - -use crate::models::ingestion_types; - -use super::{config::Config, connection::PostgresConfig}; - -#[derive(Debug, Serialize, Deserialize)] -pub struct Schema { - pub name: String, - pub schema: RootSchema, -} -pub fn get_dozer_schema() -> Result { - let schema = schema_for!(Config); - let schema_json = serde_json::to_string_pretty(&schema)?; - Ok(schema_json) -} - -pub fn get_connection_schemas() -> Result { - let mut schemas = vec![]; - - let configs = [ - ("postgres", schema_for!(PostgresConfig)), - ("ethereum", schema_for!(ingestion_types::EthConfig)), - ("grpc", schema_for!(ingestion_types::GrpcConfig)), - ("snowflake", schema_for!(ingestion_types::SnowflakeConfig)), - ("kafka", schema_for!(ingestion_types::KafkaConfig)), - ("s3", schema_for!(ingestion_types::S3Storage)), - ("local_storage", schema_for!(ingestion_types::LocalStorage)), - ("deltalake", schema_for!(ingestion_types::DeltaLakeConfig)), - ("mongodb", schema_for!(ingestion_types::MongodbConfig)), - ("mysql", schema_for!(ingestion_types::MySQLConfig)), - ("dozer", schema_for!(ingestion_types::NestedDozerConfig)), - ]; - for (name, schema) in configs.iter() { - schemas.push(Schema { - name: name.to_string(), - schema: schema.clone(), - }); - } - let schema_json = serde_json::to_string_pretty(&schemas)?; - Ok(schema_json) -} - -#[cfg(test)] -mod tests { - use super::get_connection_schemas; - - #[test] - fn get_schemas() { - let schemas = get_connection_schemas(); - assert!(schemas.is_ok()); - } -} diff --git a/dozer-types/src/models/lambda_config.rs b/dozer-types/src/models/lambda_config.rs deleted file mode 100644 index 6e0c378443..0000000000 --- a/dozer-types/src/models/lambda_config.rs +++ /dev/null @@ -1,15 +0,0 @@ -use schemars::JsonSchema; -use serde::{Deserialize, Serialize}; - -#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, JsonSchema)] -#[serde(deny_unknown_fields)] -pub enum LambdaConfig { - JavaScript(JavaScriptLambda), -} - -#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, JsonSchema)] -#[serde(deny_unknown_fields)] -pub struct JavaScriptLambda { - pub endpoint: String, - pub module: String, -} diff --git a/dozer-types/src/models/mod.rs b/dozer-types/src/models/mod.rs deleted file mode 100644 index 188074ed5b..0000000000 --- a/dozer-types/src/models/mod.rs +++ /dev/null @@ -1,19 +0,0 @@ -pub mod api_config; -pub mod api_security; -pub mod app_config; -pub mod config; -pub mod connection; -pub mod flags; -pub mod ingestion_types; -mod json_schema_helper; -pub mod lambda_config; -pub mod sink; -pub mod sink_config; -pub mod source; -pub mod telemetry; -pub mod udf_config; -pub use json_schema_helper::{get_connection_schemas, get_dozer_schema}; - -fn equal_default(t: &T) -> bool { - t == &T::default() -} diff --git a/dozer-types/src/models/sink.rs b/dozer-types/src/models/sink.rs deleted file mode 100644 index ac86af10bb..0000000000 --- a/dozer-types/src/models/sink.rs +++ /dev/null @@ -1,287 +0,0 @@ -use std::{num::NonZeroUsize, path::Display}; - -use super::equal_default; -use schemars::JsonSchema; -use serde::{Deserialize, Serialize}; -use std::fmt; - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Default, Eq, PartialEq, Clone)] -#[serde(deny_unknown_fields)] -pub struct ApiIndex { - #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub primary_key: Vec, - - #[serde(default, skip_serializing_if = "equal_default")] - pub secondary: SecondaryIndexConfig, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Default, Eq, PartialEq, Clone)] -#[serde(deny_unknown_fields)] -pub struct SecondaryIndexConfig { - #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub skip_default: Vec, - - #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub create: Vec, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Eq, PartialEq, Clone)] -#[serde(deny_unknown_fields)] -pub enum SecondaryIndex { - SortedInverted(SortedInverted), - FullText(FullText), -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Eq, PartialEq, Clone)] -#[serde(deny_unknown_fields)] -pub struct SortedInverted { - pub fields: Vec, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Eq, PartialEq, Clone)] -#[serde(deny_unknown_fields)] -pub struct FullText { - pub field: String, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Eq, PartialEq, Clone, Copy, Default)] -#[serde(deny_unknown_fields)] -pub enum OnInsertResolutionTypes { - #[default] - Nothing, - Update, - Panic, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Eq, PartialEq, Clone, Copy, Default)] -#[serde(deny_unknown_fields)] -pub enum OnUpdateResolutionTypes { - #[default] - Nothing, - Upsert, - Panic, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Eq, PartialEq, Clone, Copy, Default)] -#[serde(deny_unknown_fields)] -pub enum OnDeleteResolutionTypes { - #[default] - Nothing, - Panic, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Default, Eq, PartialEq, Clone, Copy)] -#[serde(deny_unknown_fields)] -pub struct ConflictResolution { - #[serde(default, skip_serializing_if = "equal_default")] - pub on_insert: OnInsertResolutionTypes, - #[serde(default, skip_serializing_if = "equal_default")] - pub on_update: OnUpdateResolutionTypes, - #[serde(default, skip_serializing_if = "equal_default")] - pub on_delete: OnDeleteResolutionTypes, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Default, Eq, PartialEq, Clone)] -#[serde(deny_unknown_fields)] -pub struct LogReaderOptions { - #[serde(skip_serializing_if = "Option::is_none")] - pub batch_size: Option, - - #[serde(skip_serializing_if = "Option::is_none")] - pub timeout_in_millis: Option, - - #[serde(skip_serializing_if = "Option::is_none")] - pub buffer_size: Option, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Eq, PartialEq, Clone)] -#[serde(deny_unknown_fields)] -pub struct Sink { - pub name: String, - pub config: SinkConfig, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Eq, PartialEq, Clone)] -#[serde(deny_unknown_fields)] -#[allow(clippy::large_enum_variant)] -pub enum SinkConfig { - Dummy(DummySinkConfig), - Aerospike(AerospikeSinkConfig), - Clickhouse(ClickhouseSinkConfig), - Oracle(OracleSinkConfig), -} -impl SinkConfig { - pub fn name(&self) -> String { - let name = match self { - SinkConfig::Dummy(_) => "dummy", - SinkConfig::Aerospike(_) => "aerospike", - SinkConfig::Clickhouse(_) => "clickhouse", - SinkConfig::Oracle(_) => "oracle", - }; - return name.to_string(); - } -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Clone, PartialEq, Eq)] -#[serde(deny_unknown_fields)] -pub struct DummySinkConfig { - pub table_name: String, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Clone, PartialEq, Eq)] -#[serde(untagged)] -pub enum DenormColumn { - Direct(String), - Renamed { source: String, destination: String }, -} - -impl DenormColumn { - pub fn to_src_dst(&self) -> (&str, &str) { - match self { - DenormColumn::Direct(name) => (name, name), - DenormColumn::Renamed { - source, - destination, - } => (source, destination), - } - } -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Clone, PartialEq, Eq)] -#[serde(untagged)] -pub enum DenormKey { - Simple(String), - Composite(Vec), -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Clone, PartialEq, Eq)] -#[serde(deny_unknown_fields)] -pub struct AerospikeDenormalizations { - pub from_namespace: String, - pub from_set: String, - pub key: DenormKey, - pub columns: Vec, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Clone, PartialEq, Eq)] -pub struct AerospikeSet { - pub namespace: String, - pub set: String, - pub primary_key: Vec, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Clone, PartialEq, Eq)] -#[serde(deny_unknown_fields)] -pub struct AerospikeSinkTable { - pub source_table_name: String, - pub namespace: String, - pub set_name: String, - #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub denormalize: Vec, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub write_denormalized_to: Option, - #[serde(default)] - pub primary_key: Vec, - #[serde(default)] - pub aggregate_by_pk: bool, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Eq, PartialEq, Clone)] -#[serde(deny_unknown_fields)] -pub struct AerospikeSinkConfig { - pub connection: String, - pub n_threads: Option, - #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub tables: Vec, - pub max_batch_duration_ms: Option, - pub preferred_batch_size: Option, - pub metadata_namespace: String, - #[serde(default)] - pub metadata_set: Option, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Eq, PartialEq, Clone, Default)] -#[serde(deny_unknown_fields)] -pub struct ClickhouseSinkConfig { - #[serde(default = "ClickhouseSinkConfig::default_host")] - pub host: String, - #[serde(default = "ClickhouseSinkConfig::default_port")] - pub port: u16, - #[serde(default = "ClickhouseSinkConfig::default_user")] - pub user: String, - #[serde(default)] - pub password: Option, - #[serde(default = "ClickhouseSinkConfig::default_scheme")] - pub scheme: String, - #[serde(default = "ClickhouseSinkConfig::default_database")] - pub database: String, - pub options: Vec<(String, String)>, - pub source_table_name: String, - pub sink_table_name: String, - pub create_table_options: Option, -} - -impl ClickhouseSinkConfig { - fn default_database() -> String { - "default".to_string() - } - fn default_scheme() -> String { - "tcp".to_string() - } - fn default_host() -> String { - "0.0.0.0".to_string() - } - fn default_port() -> u16 { - 9000 - } - fn default_user() -> String { - "default".to_string() - } -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Eq, PartialEq, Clone)] -#[serde(deny_unknown_fields)] -pub struct ClickhouseTableOptions { - pub engine: Option, - pub primary_keys: Option>, - pub partition_by: Option, - pub sample_by: Option, - pub order_by: Option>, - pub cluster: Option, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Eq, PartialEq, Clone)] -#[serde(deny_unknown_fields)] -pub struct OracleSinkConfig { - pub connection: String, - pub table_name: String, - #[serde(default)] - pub unique_key: Vec, - #[serde(default)] - pub owner: Option, -} - -pub fn default_log_reader_batch_size() -> u32 { - 1000 -} - -pub fn default_log_reader_timeout_in_millis() -> u32 { - 300 -} - -pub fn default_log_reader_buffer_size() -> u32 { - 1000 -} - -impl std::fmt::Display for SecondaryIndex { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - SecondaryIndex::SortedInverted(SortedInverted { fields }) => { - write!(f, "type: SortedInverted, fields: {}", fields.join(", ")) - } - SecondaryIndex::FullText(FullText { field }) => { - write!(f, "type: FullText, field: {}", field) - } - } - } -} diff --git a/dozer-types/src/models/sink_config.rs b/dozer-types/src/models/sink_config.rs deleted file mode 100644 index fbbea728d0..0000000000 --- a/dozer-types/src/models/sink_config.rs +++ /dev/null @@ -1,96 +0,0 @@ -use schemars::JsonSchema; -use serde::{Deserialize, Serialize}; - -#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, JsonSchema)] -#[serde(deny_unknown_fields)] -pub enum SinkConfig { - Snowflake(Snowflake), -} - -#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, JsonSchema)] -#[serde(deny_unknown_fields)] -pub struct Snowflake { - pub connection: snowflake::ConnectionParameters, - pub endpoint: String, - pub destination: snowflake::Destination, - - #[serde(default, skip_serializing_if = "Option::is_none")] - pub options: Option, -} - -pub mod snowflake { - use std::time::Duration; - - use crate::helper::{deserialize_duration_secs_f64, f64_schema, serialize_duration_secs_f64}; - use schemars::JsonSchema; - use serde::{Deserialize, Serialize}; - - #[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, JsonSchema)] - #[serde(deny_unknown_fields)] - pub struct ConnectionParameters { - pub server: String, - - #[serde(default, skip_serializing_if = "Option::is_none")] - pub port: Option, - - pub user: String, - pub password: String, - - #[serde(default, skip_serializing_if = "Option::is_none")] - pub role: Option, - - #[serde(default, skip_serializing_if = "Option::is_none")] - pub driver: Option, - - pub warehouse: String, - } - - #[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, JsonSchema)] - #[serde(deny_unknown_fields)] - pub struct Destination { - pub database: String, - pub schema: String, - pub table: String, - } - - #[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, JsonSchema)] - #[serde(deny_unknown_fields)] - pub struct Options { - #[serde(default, skip_serializing_if = "Option::is_none")] - pub batch_size: Option, - - #[serde( - default, - skip_serializing_if = "Option::is_none", - deserialize_with = "deserialize_duration_secs_f64", - serialize_with = "serialize_duration_secs_f64" - )] - #[schemars(schema_with = "f64_schema")] - pub batch_interval_seconds: Option, - - #[serde(default, skip_serializing_if = "Option::is_none")] - pub suspend_warehouse_after_each_batch: Option, - } - - impl Default for Options { - fn default() -> Self { - Self { - batch_size: Some(default_batch_size()), - batch_interval_seconds: Some(default_batch_interval()), - suspend_warehouse_after_each_batch: Some(default_suspend_warehouse()), - } - } - } - - pub fn default_batch_size() -> usize { - 1000000 - } - - pub fn default_batch_interval() -> Duration { - Duration::from_secs(1) - } - - pub fn default_suspend_warehouse() -> bool { - true - } -} diff --git a/dozer-types/src/models/source.rs b/dozer-types/src/models/source.rs deleted file mode 100644 index 8f0c449ec7..0000000000 --- a/dozer-types/src/models/source.rs +++ /dev/null @@ -1,39 +0,0 @@ -use schemars::JsonSchema; -use serde::{Deserialize, Serialize}; - -use super::equal_default; - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Default, Eq, PartialEq, Clone)] -#[serde(deny_unknown_fields)] -pub struct Source { - /// name of the source - to distinguish between multiple sources; Type: String - pub name: String, - - /// name of the table in source database; Type: String - pub table_name: String, - - #[serde(default, skip_serializing_if = "Vec::is_empty")] - /// list of columns gonna be used in the source table; Type: String[] - pub columns: Vec, - - /// reference to pre-defined connection name; Type: String - pub connection: String, - - #[serde(skip_serializing_if = "Option::is_none")] - /// name of schema source database; Type: String - pub schema: Option, - - #[serde(default, skip_serializing_if = "equal_default")] - /// setting for how to refresh the data; Default: RealTime - pub refresh_config: RefreshConfig, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Eq, PartialEq, Clone, Default)] -#[serde(deny_unknown_fields)] -pub enum RefreshConfig { - // Hour { minute: u32 }, - // Day { time: String }, - // CronExpression { expression: String }, - #[default] - RealTime, -} diff --git a/dozer-types/src/models/telemetry.rs b/dozer-types/src/models/telemetry.rs deleted file mode 100644 index 6763b8c385..0000000000 --- a/dozer-types/src/models/telemetry.rs +++ /dev/null @@ -1,53 +0,0 @@ -use schemars::JsonSchema; -use serde::{Deserialize, Serialize}; -#[derive(Debug, Serialize, Deserialize, JsonSchema, Default, PartialEq, Eq, Clone)] -#[serde(deny_unknown_fields)] -pub struct TelemetryConfig { - pub trace: Option, - pub metrics: Option, - #[serde(default)] - pub application_id: usize, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, PartialEq, Eq, Clone)] -pub enum TelemetryTraceConfig { - XRay(XRayConfig), -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Default, PartialEq, Eq, Clone)] -#[serde(deny_unknown_fields)] -pub struct DozerTelemetryConfig { - #[serde(skip_serializing_if = "Option::is_none")] - pub endpoint: Option, - - #[serde(skip_serializing_if = "Option::is_none")] - pub adapter: Option, - - #[serde(skip_serializing_if = "Option::is_none")] - pub sample_percent: Option, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, PartialEq, Eq, Clone)] -#[serde(deny_unknown_fields)] -pub struct XRayConfig { - pub endpoint: String, - pub timeout_in_seconds: u64, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, PartialEq, Eq, Clone)] -#[serde(deny_unknown_fields)] -pub enum TelemetryMetricsConfig { - Prometheus(PrometheusConfig), -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, PartialEq, Eq, Clone)] -#[serde(deny_unknown_fields)] -pub struct PrometheusConfig { - #[serde(default = "PrometheusConfig::default_address")] - pub address: String, -} -impl PrometheusConfig { - pub fn default_address() -> String { - "0.0.0.0:8089".to_string() - } -} diff --git a/dozer-types/src/models/udf_config.rs b/dozer-types/src/models/udf_config.rs deleted file mode 100644 index 512fba3a57..0000000000 --- a/dozer-types/src/models/udf_config.rs +++ /dev/null @@ -1,33 +0,0 @@ -use schemars::JsonSchema; - -use crate::serde::{Deserialize, Serialize}; - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Eq, PartialEq, Clone)] -#[serde(deny_unknown_fields)] -pub struct UdfConfig { - /// name of the model function - pub name: String, - /// setting for what type of udf to use; Default: Onnx - pub config: UdfType, -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Eq, PartialEq, Clone)] -#[serde(deny_unknown_fields)] -pub enum UdfType { - Onnx(OnnxConfig), - JavaScript(JavaScriptConfig), -} - -#[derive(Debug, Serialize, Deserialize, JsonSchema, Eq, PartialEq, Clone)] -#[serde(deny_unknown_fields)] -pub struct OnnxConfig { - /// path to the model file - pub path: String, -} - -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] -#[serde(deny_unknown_fields)] -pub struct JavaScriptConfig { - /// path to the module file - pub module: String, -} diff --git a/dozer-types/src/node.rs b/dozer-types/src/node.rs deleted file mode 100644 index b5b5562ae5..0000000000 --- a/dozer-types/src/node.rs +++ /dev/null @@ -1,146 +0,0 @@ -use serde::{self, Deserialize, Serialize}; - -use std::{ - collections::HashMap, - fmt::{Display, Formatter}, - str::from_utf8, -}; -#[derive( - Clone, Debug, PartialEq, Eq, Hash, Serialize, Deserialize, bincode::Encode, bincode::Decode, -)] -pub struct NodeHandle { - pub ns: Option, - pub id: String, -} - -impl NodeHandle { - pub fn new(ns: Option, id: String) -> Self { - Self { ns, id } - } -} - -impl NodeHandle { - pub fn to_bytes(&self) -> Vec { - let mut r = Vec::::with_capacity(5); - match self.ns { - Some(ns) => { - r.push(1_u8); - r.extend(ns.to_le_bytes()); - } - None => r.push(0_u8), - } - let id_buf = self.id.as_bytes(); - r.extend((id_buf.len() as u16).to_le_bytes()); - r.extend(id_buf); - r - } - - pub fn from_bytes(buffer: &[u8]) -> NodeHandle { - match buffer[0] { - 1_u8 => { - let ns = u16::from_le_bytes(buffer[1..3].try_into().unwrap()); - let id_len: u16 = u16::from_le_bytes(buffer[3..5].try_into().unwrap()); - let id = from_utf8(&buffer[5..5 + id_len as usize]).unwrap(); - NodeHandle::new(Some(ns), id.to_string()) - } - _ => { - let id_len: u16 = u16::from_le_bytes(buffer[1..3].try_into().unwrap()); - let id = from_utf8(&buffer[3..3 + id_len as usize]).unwrap(); - NodeHandle::new(None, id.to_string()) - } - } - } -} - -impl Display for NodeHandle { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - let ns_str = match self.ns { - Some(ns) => ns.to_string(), - None => "r".to_string(), - }; - f.write_str(&format!("{}_{}", ns_str, self.id)) - } -} - -#[derive( - Clone, - Debug, - Copy, - PartialEq, - Eq, - PartialOrd, - Ord, - Hash, - Default, - Serialize, - Deserialize, - bincode::Encode, - bincode::Decode, -)] -/// A identifier made of two `u64`s. -pub struct OpIdentifier { - /// High 64 bits of the identifier. - pub txid: u64, - /// Low 64 bits of the identifier. - pub seq_in_tx: u64, -} - -impl OpIdentifier { - pub fn new(txid: u64, seq_in_tx: u64) -> Self { - Self { txid, seq_in_tx } - } - - pub fn to_bytes(&self) -> [u8; 16] { - let mut result = [0_u8; 16]; - result[0..8].copy_from_slice(&self.txid.to_be_bytes()); - result[8..16].copy_from_slice(&self.seq_in_tx.to_be_bytes()); - result - } - - pub fn from_bytes(bytes: [u8; 16]) -> Self { - let txid = u64::from_be_bytes(bytes[0..8].try_into().unwrap()); - let seq_in_tx = u64::from_be_bytes(bytes[8..16].try_into().unwrap()); - Self::new(txid, seq_in_tx) - } -} - -#[derive( - Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize, bincode::Encode, bincode::Decode, -)] -/// A source's ingestion state. -pub enum SourceState { - /// This source hasn't been ingested. - NotStarted, - /// This source has some data ingested, and it can't be restarted. - NonRestartable, - /// This source has some data ingested, and it can be restarted if it's given the op id. - Restartable(OpIdentifier), -} - -impl SourceState { - pub fn op_id(&self) -> Option<&OpIdentifier> { - if let Self::Restartable(op_id) = self { - Some(op_id) - } else { - None - } - } -} - -/// Map from a `Source` node's handle to it state. -/// -/// This uniquely identifies the state of the Dozer pipeline. -pub type SourceStates = HashMap; - -#[test] -fn test_handle_to_from_bytes() { - let original = NodeHandle::new(Some(10), 100.to_string()); - let sz = original.to_bytes(); - let _decoded = NodeHandle::from_bytes(sz.as_slice()); - - let original = NodeHandle::new(None, 100.to_string()); - let sz = original.to_bytes(); - let decoded = NodeHandle::from_bytes(sz.as_slice()); - - assert_eq!(original, decoded) -} diff --git a/dozer-types/src/tests/api_config_yaml_deserialize.rs b/dozer-types/src/tests/api_config_yaml_deserialize.rs deleted file mode 100644 index e64be614fe..0000000000 --- a/dozer-types/src/tests/api_config_yaml_deserialize.rs +++ /dev/null @@ -1,159 +0,0 @@ -use crate::models::{ - api_config::{GrpcApiOptions, RestApiOptions}, - api_security::ApiSecurity, - config::Config, -}; - -#[test] -fn override_rest_port() { - let input_config = r#" - app_name: working_app - version: 1 - api: - rest: - port: 9876 - home_dir: './.dozer' - "#; - let deserialize_result = serde_yaml::from_str::(input_config); - let api_config = deserialize_result.unwrap().api; - let expected_rest_config = RestApiOptions { - port: Some(9876), - host: None, - cors: None, - enabled: None, - enable_sql: None, - }; - assert_eq!(api_config.rest, expected_rest_config); -} - -#[test] -fn override_rest_host() { - let input_config = r#" - app_name: working_app - version: 1 - api: - rest: - host: localhost - home_dir: './.dozer' - "#; - let deserialize_result = serde_yaml::from_str::(input_config); - let api_config = deserialize_result.unwrap().api; - let expected_rest_config = RestApiOptions { - port: None, - host: Some("localhost".to_owned()), - cors: None, - enabled: None, - enable_sql: None, - }; - assert_eq!(api_config.rest, expected_rest_config); -} - -#[test] -fn override_rest_enabled() { - let input_config = r#" - app_name: working_app - version: 1 - api: - rest: - enabled: false - home_dir: './.dozer' - "#; - let deserialize_result = serde_yaml::from_str::(input_config); - let api_config = deserialize_result.unwrap().api; - let expected_rest_config = RestApiOptions { - port: None, - host: None, - cors: None, - enabled: Some(false), - enable_sql: None, - }; - assert_eq!(api_config.rest, expected_rest_config); -} - -#[test] -fn override_grpc_port() { - let input_config = r#" - app_name: working_app - version: 1 - api: - grpc: - port: 4232 - home_dir: './.dozer' -"#; - let deserialize_result = serde_yaml::from_str::(input_config); - let api_config = deserialize_result.unwrap().api; - let expected_grpc_config = GrpcApiOptions { - port: Some(4232), - host: None, - cors: None, - web: None, - enabled: None, - }; - assert_eq!(api_config.grpc, expected_grpc_config); -} - -#[test] -fn override_grpc_enabled() { - let input_config = r#" - app_name: working_app - version: 1 - api: - grpc: - enabled: false - home_dir: './.dozer' -"#; - let deserialize_result = serde_yaml::from_str::(input_config); - let api_config = deserialize_result.unwrap().api; - let expected_grpc_config = GrpcApiOptions { - enabled: Some(false), - port: None, - host: None, - cors: None, - web: None, - }; - assert_eq!(api_config.grpc, expected_grpc_config); -} - -#[test] -fn override_jwt() { - let input_config = r#" - app_name: working_app - version: 1 - api: - api_security: !Jwt - Vv44T1GugX - home_dir: './.dozer' -"#; - let deserialize_result = serde_yaml::from_str::(input_config); - let api_config = deserialize_result.unwrap().api; - let api_security = api_config.api_security; - assert!(api_security.is_some()); - let api_security = api_security.unwrap(); - let expected_api_security = ApiSecurity::Jwt("Vv44T1GugX".to_owned()); - assert_eq!(api_security, expected_api_security); -} - -#[test] -fn override_pipeline_port() { - let input_config = r#" - app_name: working_app - version: 1 - api: - grpc: - port: 4232 - rest: - port: 3324 - api_security: !Jwt - Vv44T1GugX - app_grpc: - port: 3993 - - home_dir: './.dozer' -"#; - - let deserialize_result = serde_yaml::from_str::(input_config); - let api_config = deserialize_result.unwrap().api; - let app_grpc = api_config.app_grpc; - assert_eq!(app_grpc.port, Some(3993)); - assert_eq!(app_grpc.host, None); -} diff --git a/dozer-types/src/tests/dozer_yaml_deserialize.rs b/dozer-types/src/tests/dozer_yaml_deserialize.rs deleted file mode 100644 index 929fd8ed7c..0000000000 --- a/dozer-types/src/tests/dozer_yaml_deserialize.rs +++ /dev/null @@ -1,169 +0,0 @@ -use crate::models::config::Config; - -#[test] -#[ignore = "We removed the connection name validation, but should add it back in the future as part of a `validation` step"] -fn error_wrong_reference_connection_name() { - let input_config = r#" - app_name: working_app - version: 1 - home_dir: './.dozer' - connections: - - authentication: !Postgres - user: postgres - password: postgres - host: localhost - port: 5432 - database: users - db_type: Postgres - name: users - sources: - - name: users - table_name: users - columns: - - id - - email - - phone - connection: wrong_connection_name - sinks: - - name: users - path: /users - sql: select id, email, phone from users where 1=1; - index: - primary_key: - - id - - "#; - let deserialize_result = serde_yaml::from_str::(input_config); - let error = deserialize_result.err(); - assert!(error.is_some()); - assert!(error - .unwrap() - .to_string() - .starts_with("sources[0]: Cannot find Ref connection name: wrong_connection_name")); -} - -#[test] -fn error_missing_field_general() { - let input_config = r#" - app_name: working_app - version: 1 - home_dir: './.dozer' - connections: - - config: !Postgres - user: postgres - password: postgres - host: localhost - port: 5432 - database: users - name: users - sources: - - table_name: users - columns: - - id - - email - - phone - connection: users - sinks: - - path: /eth/stats - sql: select block_number, sum(id) from eth_logs where 1=1 group by block_number; - index: - primary_key: - - block_number - "#; - let deserialize_result = serde_yaml::from_str::(input_config); - let error = deserialize_result.err(); - assert!(error.is_some()); - assert!(error - .unwrap() - .to_string() - .starts_with("sources[0]: missing field `name`")); -} -#[test] -fn error_missing_field_in_source() { - let input_config = r#" - app_name: working_app - version: 1 - home_dir: './.dozer' - connections: - - config: !Postgres - user: postgres - password: postgres - host: localhost - port: 5432 - database: users - name: users - sources: - - table_name: users - columns: - - id - - email - - phone - connection: users - "#; - let deserialize_result = serde_yaml::from_str::(input_config); - let error = deserialize_result.err(); - assert!(error.is_some()); - assert!(error - .unwrap() - .to_string() - .starts_with("sources[0]: missing field `name`")); -} - -#[test] -fn error_missing_field_connection_ref_in_source() { - let input_config = r#" - app_name: working_app - version: 1 - home_dir: './.dozer' - connections: - - config: !Postgres - user: postgres - password: postgres - host: localhost - port: 5432 - database: users - name: users - sources: - - name: users - table_name: users - columns: - - id - - email - - phone - "#; - let deserialize_result = serde_yaml::from_str::(input_config); - let error = deserialize_result.err(); - assert!(error.is_some()); - assert!(error - .unwrap() - .to_string() - .starts_with("sources[0]: missing field `connection`")); -} - -#[test] -fn error_missing_connection_ref() { - let input_config = r#" - app_name: working_app - version: 1 - home_dir: './.dozer' - connections: - - config: !Postgres - user: postgres - host: localhost - port: 5432 - database: users - name: users - sources: - - name: users - table_name: users - columns: - - id - - email - - phone - "#; - let deserialize_result = serde_yaml::from_str::(input_config); - let error = deserialize_result.unwrap_err(); - assert!(error - .to_string() - .starts_with("sources[0]: missing field `connection`")); -} diff --git a/dozer-types/src/tests/eth_yaml_deserialize.rs b/dozer-types/src/tests/eth_yaml_deserialize.rs deleted file mode 100644 index 1354891d49..0000000000 --- a/dozer-types/src/tests/eth_yaml_deserialize.rs +++ /dev/null @@ -1,60 +0,0 @@ -use crate::models::{ - connection::ConnectionConfig, - ingestion_types::{EthConfig, EthFilter, EthLogConfig, EthProviderConfig}, -}; -#[test] -fn standard() { - let eth_config = r#" - !Ethereum - provider: !Log - wss_url: wss://link - filter: - from_block: 0 - addresses: [] - topics: [] - contracts: [] - "#; - let deserializer_result = serde_yaml::from_str::(eth_config).unwrap(); - let expected_eth_filter = EthFilter { - from_block: Some(0), - to_block: None, - addresses: vec![], - topics: vec![], - }; - let expected_eth_config = EthConfig { - provider: EthProviderConfig::Log(EthLogConfig { - filter: Some(expected_eth_filter), - wss_url: "wss://link".to_owned(), - contracts: vec![], - }), - }; - let expected = ConnectionConfig::Ethereum(expected_eth_config); - assert_eq!(expected, deserializer_result); -} - -#[test] -fn config_without_empty_array() { - let eth_config = r#" - !Ethereum - provider: !Log - wss_url: wss://link - filter: - from_block: 499203 - "#; - let deserializer_result = serde_yaml::from_str::(eth_config).unwrap(); - let expected_eth_filter = EthFilter { - from_block: Some(499203), - to_block: None, - addresses: vec![], - topics: vec![], - }; - let expected_eth_config = EthConfig { - provider: EthProviderConfig::Log(EthLogConfig { - wss_url: "wss://link".to_owned(), - filter: Some(expected_eth_filter), - contracts: vec![], - }), - }; - let expected = ConnectionConfig::Ethereum(expected_eth_config); - assert_eq!(expected, deserializer_result); -} diff --git a/dozer-types/src/tests/field_serialize_test.rs b/dozer-types/src/tests/field_serialize_test.rs deleted file mode 100644 index 3e48751a7d..0000000000 --- a/dozer-types/src/tests/field_serialize_test.rs +++ /dev/null @@ -1,40 +0,0 @@ -use bincode::config; - -use crate::types::{field_test_cases, Field}; - -#[test] -fn test_field_serialize_roundtrip() { - for field in field_test_cases() { - let bytes = field.encode(); - let deserialized = Field::decode(&bytes).unwrap(); - assert_eq!(field, deserialized); - } -} - -#[test] -fn test_field_bincode_serialize_roundtrip() { - for field in field_test_cases() { - let bytes = bincode::encode_to_vec(&field, config::legacy()).unwrap(); - let (deserialized, _): (Field, _) = bincode::decode_from_slice(&bytes, config::legacy()) - .unwrap_or_else(|e| { - panic!("Failed to deserialize field: {field:?} from bytes: {bytes:?}. {e}") - }); - assert_eq!(field, deserialized); - } -} - -#[test] -fn field_serialization_should_never_be_empty() { - for field in field_test_cases() { - let bytes = field.encode(); - assert!(!bytes.is_empty()); - } -} - -#[test] -fn encoding_len_must_agree_with_encode() { - for field in field_test_cases() { - let bytes = field.encode(); - assert_eq!(bytes.len(), field.encoding_len()); - } -} diff --git a/dozer-types/src/tests/flags_config_yaml_deserialize.rs b/dozer-types/src/tests/flags_config_yaml_deserialize.rs deleted file mode 100644 index fa9724a797..0000000000 --- a/dozer-types/src/tests/flags_config_yaml_deserialize.rs +++ /dev/null @@ -1,23 +0,0 @@ -use crate::models::{config::Config, flags::Flags}; - -#[test] -fn test_partial_flag_config_input() { - let input_config_with_flag = r#" - app_name: working_app - version: 1 - flags: - grpc_web: false - push_events: false -"#; - let deserializer_result = serde_yaml::from_str::(input_config_with_flag).unwrap(); - let default_flags = Flags::default(); - let flags_deserialize = deserializer_result.flags; - assert_eq!(flags_deserialize.dynamic, None); - assert_eq!(flags_deserialize.grpc_web, Some(false)); - assert_eq!(flags_deserialize.push_events, Some(false)); - assert_eq!( - flags_deserialize.authenticate_server_reflection, - default_flags.authenticate_server_reflection, - ); - assert_eq!(flags_deserialize.dynamic, default_flags.dynamic); -} diff --git a/dozer-types/src/tests/mod.rs b/dozer-types/src/tests/mod.rs deleted file mode 100644 index 99dca538eb..0000000000 --- a/dozer-types/src/tests/mod.rs +++ /dev/null @@ -1,8 +0,0 @@ -mod api_config_yaml_deserialize; -mod dozer_yaml_deserialize; -mod eth_yaml_deserialize; -mod field_serialize_test; -mod flags_config_yaml_deserialize; -mod postgres_yaml_deserialize; -mod secondary_index_yaml_deserialize; -mod udf_yaml_deserialize; diff --git a/dozer-types/src/tests/postgres_yaml_deserialize.rs b/dozer-types/src/tests/postgres_yaml_deserialize.rs deleted file mode 100644 index c92fc55d25..0000000000 --- a/dozer-types/src/tests/postgres_yaml_deserialize.rs +++ /dev/null @@ -1,212 +0,0 @@ -use crate::models::connection::{ConnectionConfig, PostgresConfig}; -#[test] -fn standard() { - let postgres_config = r#" - !Postgres - user: postgres - password: postgres - host: localhost - port: 5432 - database: users - "#; - let deserializer_result = serde_yaml::from_str::(postgres_config).unwrap(); - let postgres_auth = PostgresConfig { - user: Some("postgres".to_string()), - password: Some("postgres".to_string()), - host: Some("localhost".to_string()), - port: Some(5432), - database: Some("users".to_string()), - sslmode: None, - connection_url: None, - schema: None, - batch_size: None, - }; - let expected = ConnectionConfig::Postgres(postgres_auth); - assert_eq!(expected, deserializer_result); - - if let ConnectionConfig::Postgres(config) = expected { - let expected_replenished = config.replenish().unwrap(); - if let ConnectionConfig::Postgres(deserialized) = deserializer_result { - let deserialized_replenished = deserialized.replenish().unwrap(); - assert_eq!(expected_replenished, deserialized_replenished); - } - } -} - -#[test] -fn standard_with_ssl_mode() { - let postgres_config = r#" - !Postgres - user: postgres - password: postgres - host: localhost - port: 5432 - database: users - sslmode: require - "#; - let deserializer_result = serde_yaml::from_str::(postgres_config).unwrap(); - let postgres_auth = PostgresConfig { - user: Some("postgres".to_string()), - password: Some("postgres".to_string()), - host: Some("localhost".to_string()), - port: Some(5432), - database: Some("users".to_string()), - sslmode: Some("require".to_string()), - connection_url: None, - schema: None, - batch_size: None, - }; - let expected = ConnectionConfig::Postgres(postgres_auth); - assert_eq!(expected, deserializer_result); - - if let ConnectionConfig::Postgres(config) = expected { - let expected_replenished = config.replenish().unwrap(); - if let ConnectionConfig::Postgres(deserialized) = deserializer_result { - let deserialized_replenished = deserialized.replenish().unwrap(); - assert_eq!(expected_replenished, deserialized_replenished); - } - } -} - -#[test] -fn standard_url() { - let postgres_config = r#" - !Postgres - connection_url: postgres://postgres:postgres@ep-silent-bread-370191.ap-southeast-1.aws.neon.tech:5432/neondb?sslmode=prefer - "#; - let deserializer_result = serde_yaml::from_str::(postgres_config).unwrap(); - let postgres_auth = PostgresConfig { - user: None, - password: None, - host: None, - port: None, - database: None, - sslmode: None, - connection_url: Some("postgres://postgres:postgres@ep-silent-bread-370191.ap-southeast-1.aws.neon.tech:5432/neondb?sslmode=prefer".to_string()), - schema: None, - batch_size: None, - }; - let expected = ConnectionConfig::Postgres(postgres_auth); - assert_eq!(expected, deserializer_result); - - if let ConnectionConfig::Postgres(config) = expected { - let expected_replenished = config.replenish().unwrap(); - if let ConnectionConfig::Postgres(deserialized) = deserializer_result { - let deserialized_replenished = deserialized.replenish().unwrap(); - assert_eq!(expected_replenished, deserialized_replenished); - } - } -} - -#[test] -#[should_panic(expected = "MissingFieldInPostgresConfig(\"user\")")] -fn standard_url_missing_user() { - let postgres_config = r#" - !Postgres - connection_url: postgresql://localhost:5432/stocks?sslmode=prefer - "#; - let deserializer_result = serde_yaml::from_str::(postgres_config).unwrap(); - let postgres_auth = PostgresConfig { - user: None, - password: None, - host: None, - port: None, - database: None, - sslmode: None, - connection_url: Some("postgresql://localhost:5432/stocks?sslmode=prefer".to_string()), - schema: None, - batch_size: None, - }; - let expected = ConnectionConfig::Postgres(postgres_auth); - assert_eq!(expected, deserializer_result); - - if let ConnectionConfig::Postgres(config) = expected { - let expected_replenished = config.replenish().unwrap(); - if let ConnectionConfig::Postgres(deserialized) = deserializer_result { - let deserialized_replenished = deserialized.replenish().unwrap(); - assert_eq!(expected_replenished, deserialized_replenished); - } - } -} - -#[test] -#[should_panic(expected = "MissingFieldInPostgresConfig(\"password\")")] -fn standard_url_missing_password() { - let postgres_config = r#" - !Postgres - user: postgres - connection_url: postgresql://localhost:5432/stocks?sslmode=prefer - "#; - let deserializer_result = serde_yaml::from_str::(postgres_config).unwrap(); - let postgres_auth = PostgresConfig { - user: Some("postgres".to_string()), - password: None, - host: None, - port: None, - database: None, - sslmode: None, - connection_url: Some("postgresql://localhost:5432/stocks?sslmode=prefer".to_string()), - schema: None, - batch_size: None, - }; - let expected = ConnectionConfig::Postgres(postgres_auth); - assert_eq!(expected, deserializer_result); - - if let ConnectionConfig::Postgres(config) = expected { - let expected_replenished = config.replenish().unwrap(); - if let ConnectionConfig::Postgres(deserialized) = deserializer_result { - let deserialized_replenished = deserialized.replenish().unwrap(); - assert_eq!(expected_replenished, deserialized_replenished); - } - } -} - -#[test] -fn standard_url_2() { - let postgres_config = r#" - !Postgres - user: postgres - password: postgres - connection_url: postgresql://localhost:5432/stocks?sslmode=prefer - "#; - let deserializer_result = serde_yaml::from_str::(postgres_config).unwrap(); - let postgres_auth = PostgresConfig { - user: Some("postgres".to_string()), - password: Some("postgres".to_string()), - host: None, - port: None, - database: None, - sslmode: None, - connection_url: Some("postgresql://localhost:5432/stocks?sslmode=prefer".to_string()), - schema: None, - batch_size: None, - }; - let expected = ConnectionConfig::Postgres(postgres_auth); - assert_eq!(expected, deserializer_result); - - if let ConnectionConfig::Postgres(config) = expected { - let expected_replenished = config.replenish().unwrap(); - if let ConnectionConfig::Postgres(deserialized) = deserializer_result { - let deserialized_replenished = deserialized.replenish().unwrap(); - assert_eq!(expected_replenished, deserialized_replenished); - } - } -} - -#[test] -fn error_wrong_tag() { - let posgres_config = r#" - !Postgres112 - user: postgres - host: localhost - port: 5432 - database: users - "#; - let deserializer_result = serde_yaml::from_str::(posgres_config); - assert!(deserializer_result.is_err()); - assert!(deserializer_result - .err() - .unwrap() - .to_string() - .starts_with("unknown variant `Postgres112`")) -} diff --git a/dozer-types/src/tests/secondary_index_yaml_deserialize.rs b/dozer-types/src/tests/secondary_index_yaml_deserialize.rs deleted file mode 100644 index 9978e8af16..0000000000 --- a/dozer-types/src/tests/secondary_index_yaml_deserialize.rs +++ /dev/null @@ -1,37 +0,0 @@ -use crate::models::sink::{FullText, SecondaryIndex, SecondaryIndexConfig, SortedInverted}; - -#[test] -fn standard() { - let secondary = r#"skip_default: - - field1 - - field2 -create: - - !SortedInverted - fields: - - field1 - - field2 - - !FullText - field: field3 -"#; - let config: SecondaryIndexConfig = serde_yaml::from_str(secondary).unwrap(); - assert_eq!(config.skip_default, vec!["field1", "field2"]); - assert_eq!( - config.create, - vec![ - SecondaryIndex::SortedInverted(SortedInverted { - fields: vec!["field1".to_string(), "field2".to_string()] - }), - SecondaryIndex::FullText(FullText { - field: "field3".to_string() - }), - ] - ); -} - -#[test] -fn empty() { - let secondary = ""; - let config: SecondaryIndexConfig = serde_yaml::from_str(secondary).unwrap(); - assert_eq!(config.skip_default, Vec::::new()); - assert_eq!(config.create, vec![]); -} diff --git a/dozer-types/src/tests/udf_yaml_deserialize.rs b/dozer-types/src/tests/udf_yaml_deserialize.rs deleted file mode 100644 index 6cec087a5b..0000000000 --- a/dozer-types/src/tests/udf_yaml_deserialize.rs +++ /dev/null @@ -1,19 +0,0 @@ -use crate::models::udf_config::{OnnxConfig, UdfConfig, UdfType}; - -#[test] -fn standard() { - let udf_config = r#" - name: is_fraudulent - config: !Onnx - path: ./models/model_file - "#; - let deserializer_result = serde_yaml::from_str::(udf_config).unwrap(); - let udf_conf = UdfConfig { - config: UdfType::Onnx(OnnxConfig { - path: "./models/model_file".to_string(), - }), - name: "is_fraudulent".to_string(), - }; - let expected = udf_conf; - assert_eq!(expected, deserializer_result); -} diff --git a/dozer-types/src/types/field.rs b/dozer-types/src/types/field.rs deleted file mode 100644 index af8d9671ad..0000000000 --- a/dozer-types/src/types/field.rs +++ /dev/null @@ -1,1230 +0,0 @@ -use crate::errors::types::DeserializationError; -use crate::json_types::{ - json_cmp, json_from_bytes, json_from_str, json_to_bytes, json_to_bytes_size, JsonValue, -}; -use crate::types::{ - DozerDuration, DozerPoint, FieldDefinition, Schema, SourceDefinition, TimeUnit, -}; -#[allow(unused_imports)] -use chrono::{DateTime, Datelike, FixedOffset, LocalResult, NaiveDate, TimeZone, Utc}; -use ijson::DestructuredRef; -use ordered_float::OrderedFloat; -use rust_decimal::prelude::{FromPrimitive, ToPrimitive}; -use rust_decimal::Decimal; -use serde::{self, Deserialize, Serialize}; -use std::borrow::Cow; -use std::fmt::{Display, Formatter}; -use std::str::FromStr; -use std::time::Duration; - -pub const DATE_FORMAT: &str = "%Y-%m-%d"; -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, Hash)] -#[cfg_attr(feature = "arbitrary", derive(arbitrary::Arbitrary))] -pub enum Field { - UInt(u64), - U128(u128), - Int(i64), - Int8(i8), - I128(i128), - Float(#[cfg_attr(feature= "arbitrary", arbitrary(with = arbitrary_float))] OrderedFloat), - Boolean(bool), - String(String), - Text(String), - Binary(#[serde(with = "serde_bytes")] Vec), - Decimal(Decimal), - Timestamp(DateTime), - Date(NaiveDate), - Json(#[cfg_attr(feature= "arbitrary", arbitrary(with = arb_json::arbitrary_json))] JsonValue), - Point(DozerPoint), - Duration(DozerDuration), - Null, -} - -impl bincode::Decode for Field { - fn decode( - decoder: &mut D, - ) -> Result { - let first_byte = u32::decode(decoder)?; - match first_byte { - 0 => Ok(Field::UInt(u64::decode(decoder)?)), - 1 => Ok(Field::U128(u128::decode(decoder)?)), - 2 => Ok(Field::Int(i64::decode(decoder)?)), - 3 => Ok(Field::I128(i128::decode(decoder)?)), - 4 => Ok(Field::Float(OrderedFloat(f64::decode(decoder)?))), - 5 => Ok(Field::Boolean(bool::decode(decoder)?)), - 6 => Ok(Field::String(String::decode(decoder)?)), - 7 => Ok(Field::Text(String::decode(decoder)?)), - 8 => Ok(Field::Binary(Vec::::decode(decoder)?)), - 9 => { - let decoded = bincode::serde::Compat::decode(decoder)?; - Ok(Field::Decimal(decoded.0)) - } - 10 => { - let decoded = bincode::serde::Compat::decode(decoder)?; - Ok(Field::Timestamp(decoded.0)) - } - 11 => { - let decoded = bincode::serde::Compat::decode(decoder)?; - Ok(Field::Date(decoded.0)) - } - 12 => { - let bytes = Vec::::decode(decoder)?; - Ok(Field::Json(rmp_serde::from_slice(&bytes).map_err(|e| { - bincode::error::DecodeError::OtherString(e.to_string()) - })?)) - } - 13 => Ok(Field::Point(DozerPoint::decode(decoder)?)), - 14 => Ok(Field::Duration(DozerDuration::decode(decoder)?)), - 15 => Ok(Field::Null), - other => Err(bincode::error::DecodeError::UnexpectedVariant { - type_name: "Field", - allowed: &bincode::error::AllowedEnumVariants::Range { min: 0, max: 15 }, - found: other, - }), - } - } -} - -impl<'de> bincode::BorrowDecode<'de> for Field { - fn borrow_decode>( - decoder: &mut D, - ) -> Result { - let first_byte = u32::borrow_decode(decoder)?; - match first_byte { - 0 => Ok(Field::UInt(u64::borrow_decode(decoder)?)), - 1 => Ok(Field::U128(u128::borrow_decode(decoder)?)), - 2 => Ok(Field::Int(i64::borrow_decode(decoder)?)), - 3 => Ok(Field::I128(i128::borrow_decode(decoder)?)), - 4 => Ok(Field::Float(OrderedFloat(f64::borrow_decode(decoder)?))), - 5 => Ok(Field::Boolean(bool::borrow_decode(decoder)?)), - 6 => Ok(Field::String(String::borrow_decode(decoder)?)), - 7 => Ok(Field::Text(String::borrow_decode(decoder)?)), - 8 => Ok(Field::Binary(Vec::::borrow_decode(decoder)?)), - 9 => { - let decoded = bincode::serde::Compat::borrow_decode(decoder)?; - Ok(Field::Decimal(decoded.0)) - } - 10 => { - let decoded = bincode::serde::Compat::borrow_decode(decoder)?; - Ok(Field::Timestamp(decoded.0)) - } - 11 => { - let decoded = bincode::serde::Compat::borrow_decode(decoder)?; - Ok(Field::Date(decoded.0)) - } - 12 => { - let bytes = <&[u8]>::borrow_decode(decoder)?; - Ok(Field::Json(rmp_serde::from_slice(bytes).map_err(|e| { - bincode::error::DecodeError::OtherString(e.to_string()) - })?)) - } - 13 => Ok(Field::Point(DozerPoint::borrow_decode(decoder)?)), - 14 => Ok(Field::Duration(DozerDuration::borrow_decode(decoder)?)), - 15 => Ok(Field::Null), - other => Err(bincode::error::DecodeError::UnexpectedVariant { - type_name: "Field", - allowed: &bincode::error::AllowedEnumVariants::Range { min: 0, max: 15 }, - found: other, - }), - } - } -} - -impl bincode::Encode for Field { - fn encode( - &self, - encoder: &mut E, - ) -> Result<(), bincode::error::EncodeError> { - (self.get_type_prefix() as u32).encode(encoder)?; - match self { - Field::UInt(v) => v.encode(encoder), - Field::U128(v) => v.encode(encoder), - Field::Int(v) => v.encode(encoder), - Field::Int8(v) => v.encode(encoder), - Field::I128(v) => v.encode(encoder), - Field::Float(v) => v.encode(encoder), - Field::Boolean(v) => v.encode(encoder), - Field::String(v) => v.encode(encoder), - Field::Text(v) => v.encode(encoder), - Field::Binary(v) => v.encode(encoder), - Field::Decimal(v) => bincode::serde::Compat(v).encode(encoder), - Field::Timestamp(v) => bincode::serde::Compat(v).encode(encoder), - Field::Date(v) => bincode::serde::Compat(v).encode(encoder), - Field::Json(v) => { - let bytes = rmp_serde::to_vec(v) - .map_err(|e| bincode::error::EncodeError::OtherString(e.to_string()))?; - bytes.encode(encoder) - } - Field::Point(v) => v.encode(encoder), - Field::Duration(v) => v.encode(encoder), - Field::Null => Ok(()), - } - } -} - -impl Ord for Field { - fn cmp(&self, other: &Self) -> std::cmp::Ordering { - match (self, other) { - (Self::UInt(l), Self::UInt(r)) => l.cmp(r), - (Self::U128(l), Self::U128(r)) => l.cmp(r), - (Self::Int(l), Self::Int(r)) => l.cmp(r), - (Self::I128(l), Self::I128(r)) => l.cmp(r), - (Self::Float(l), Self::Float(r)) => l.cmp(r), - (Self::Boolean(l), Self::Boolean(r)) => l.cmp(r), - (Self::String(l), Self::String(r)) => l.cmp(r), - (Self::Text(l), Self::Text(r)) => l.cmp(r), - (Self::Binary(l), Self::Binary(r)) => l.cmp(r), - (Self::Decimal(l), Self::Decimal(r)) => l.cmp(r), - (Self::Timestamp(l), Self::Timestamp(r)) => l.cmp(r), - (Self::Date(l), Self::Date(r)) => l.cmp(r), - (Self::Json(l), Self::Json(r)) => json_cmp(l, r), - (Self::Point(l), Self::Point(r)) => l.cmp(r), - (Self::Duration(l), Self::Duration(r)) => l.cmp(r), - (Self::Null, Self::Null) => std::cmp::Ordering::Equal, - (Self::Null, _) => std::cmp::Ordering::Greater, - (_, Self::Null) => std::cmp::Ordering::Less, - (l, r) => l.ty().cmp(&r.ty()), - } - } -} -impl PartialOrd for Field { - fn partial_cmp(&self, other: &Self) -> Option { - Some(self.cmp(other)) - } -} - -#[cfg(feature = "arbitrary")] -pub(crate) fn arbitrary_float( - arbitrary: &mut arbitrary::Unstructured, -) -> arbitrary::Result> { - Ok(OrderedFloat(arbitrary.arbitrary()?)) -} - -#[cfg(feature = "arbitrary")] -mod arb_json { - use arbitrary::Arbitrary; - use ijson::{IArray, IObject}; - - use super::JsonValue; - - struct ArbitraryJson(JsonValue); - - impl<'a> arbitrary::Arbitrary<'a> for ArbitraryJson { - fn arbitrary(u: &mut arbitrary::Unstructured<'a>) -> arbitrary::Result { - use ijson::ValueType; - - let v = match u.choose(&[ - ValueType::Null, - ValueType::Bool, - ValueType::Number, - ValueType::String, - ValueType::Array, - ValueType::Object, - ])? { - ValueType::Null => JsonValue::NULL, - ValueType::Bool => bool::arbitrary(u)?.into(), - ValueType::Number => f64::arbitrary(u)?.into(), - ValueType::String => String::arbitrary(u)?.into(), - ValueType::Array => { - let mut values = IArray::new(); - for json_value in u.arbitrary_iter::()? { - values.push(json_value?.0); - } - values.into() - } - ValueType::Object => { - let mut object = IObject::new(); - for result in u.arbitrary_iter::<(String, ArbitraryJson)>()? { - let (key, value) = result?; - object.insert(key, value.0); - } - object.into() - } - }; - Ok(ArbitraryJson(v)) - } - } - pub(crate) fn arbitrary_json( - arbitrary: &mut arbitrary::Unstructured, - ) -> arbitrary::Result { - Ok(ArbitraryJson::arbitrary(arbitrary)?.0) - } -} - -impl Field { - pub(crate) fn data_encoding_len(&self) -> usize { - match self { - Field::UInt(_) => 8, - Field::U128(_) => 16, - Field::Int(_) => 8, - Field::Int8(_) => 8, - Field::I128(_) => 16, - Field::Float(_) => 8, - Field::Boolean(_) => 1, - Field::String(s) => s.len(), - Field::Text(s) => s.len(), - Field::Binary(b) => b.len(), - Field::Decimal(_) => 16, - Field::Timestamp(_) => 8, - Field::Date(_) => 10, - // todo: should optimize with better serialization method - Field::Json(b) => json_to_bytes_size(b), - Field::Point(_p) => 16, - Field::Duration(_) => 17, - Field::Null => 0, - } - } - - pub(crate) fn encode_data(&self) -> Cow<[u8]> { - match self { - Field::UInt(i) => Cow::Owned(i.to_be_bytes().into()), - Field::U128(i) => Cow::Owned(i.to_be_bytes().into()), - Field::Int(i) => Cow::Owned(i.to_be_bytes().into()), - Field::Int8(i) => Cow::Owned(i.to_be_bytes().into()), - Field::I128(i) => Cow::Owned(i.to_be_bytes().into()), - Field::Float(f) => Cow::Owned(f.to_be_bytes().into()), - Field::Boolean(b) => Cow::Owned(if *b { [1] } else { [0] }.into()), - Field::String(s) => Cow::Borrowed(s.as_bytes()), - Field::Text(s) => Cow::Borrowed(s.as_bytes()), - Field::Binary(b) => Cow::Borrowed(b.as_slice()), - Field::Decimal(d) => Cow::Owned(d.serialize().into()), - Field::Timestamp(t) => Cow::Owned(t.timestamp_millis().to_be_bytes().into()), - Field::Date(t) => Cow::Owned(t.to_string().into()), - Field::Json(b) => Cow::Owned(json_to_bytes(b)), - Field::Point(p) => Cow::Owned(p.to_bytes().into()), - Field::Duration(d) => Cow::Owned(d.to_bytes().into()), - Field::Null => Cow::Owned([].into()), - } - } - - pub fn encode_buf(&self, destination: &mut [u8]) { - let prefix = self.get_type_prefix(); - let data = self.encode_data(); - destination[0] = prefix; - destination[1..].copy_from_slice(&data); - } - - pub fn encoding_len(&self) -> usize { - self.data_encoding_len() + 1 - } - - pub fn encode(&self) -> Vec { - let mut result = vec![0; self.encoding_len()]; - self.encode_buf(&mut result); - result - } - - pub fn decode(buf: &[u8]) -> Result { - let first_byte = *buf.first().ok_or(DeserializationError::EmptyInput)?; - let val = &buf[1..]; - match first_byte { - 0 => Ok(Field::UInt(u64::from_be_bytes( - val.try_into() - .map_err(|_| DeserializationError::BadDataLength)?, - ))), - 1 => Ok(Field::U128(u128::from_be_bytes( - val.try_into() - .map_err(|_| DeserializationError::BadDataLength)?, - ))), - 2 => Ok(Field::Int(i64::from_be_bytes( - val.try_into() - .map_err(|_| DeserializationError::BadDataLength)?, - ))), - 3 => Ok(Field::I128(i128::from_be_bytes( - val.try_into() - .map_err(|_| DeserializationError::BadDataLength)?, - ))), - 4 => Ok(Field::Float(OrderedFloat(f64::from_be_bytes( - val.try_into() - .map_err(|_| DeserializationError::BadDataLength)?, - )))), - 5 => Ok(Field::Boolean(val[0] == 1)), - 6 => Ok(Field::String(std::str::from_utf8(val)?.to_string())), - 7 => Ok(Field::Text(std::str::from_utf8(val)?.to_string())), - 8 => Ok(Field::Binary(val.to_vec())), - 9 => Ok(Field::Decimal(Decimal::deserialize( - val.try_into() - .map_err(|_| DeserializationError::BadDataLength)?, - ))), - 10 => { - let timestamp = Utc.timestamp_millis_opt(i64::from_be_bytes( - val.try_into() - .map_err(|_| DeserializationError::BadDataLength)?, - )); - - match timestamp { - LocalResult::Single(v) => Ok(Field::Timestamp(DateTime::from(v))), - LocalResult::Ambiguous(_, _) => Err(DeserializationError::Custom( - "Ambiguous timestamp".to_string().into(), - )), - LocalResult::None => Err(DeserializationError::Custom( - "Invalid timestamp".to_string().into(), - )), - } - } - 11 => Ok(Field::Date(NaiveDate::parse_from_str( - std::str::from_utf8(val)?, - DATE_FORMAT, - )?)), - 12 => Ok(Field::Json(json_from_bytes(val)?)), - 13 => Ok(Field::Point( - DozerPoint::from_bytes(val).map_err(|_| DeserializationError::BadDataLength)?, - )), - 14 => Ok(Field::Duration( - DozerDuration::from_bytes(val).map_err(|_| DeserializationError::BadDataLength)?, - )), - 15 => Ok(Field::Null), - other => Err(DeserializationError::UnrecognisedFieldType(other)), - } - } - - fn get_type_prefix(&self) -> u8 { - match self { - Field::UInt(_) => 0, - Field::U128(_) => 1, - Field::Int(_) => 2, - Field::I128(_) => 3, - Field::Float(_) => 4, - Field::Boolean(_) => 5, - Field::String(_) => 6, - Field::Text(_) => 7, - Field::Binary(_) => 8, - Field::Decimal(_) => 9, - Field::Timestamp(_) => 10, - Field::Date(_) => 11, - Field::Json(_) => 12, - Field::Point(_) => 13, - Field::Duration(_) => 14, - Field::Null => 15, - Field::Int8(_) => 16, - } - } - - pub fn ty(&self) -> Option { - match self { - Field::UInt(_) => Some(FieldType::UInt), - Field::U128(_) => Some(FieldType::U128), - Field::Int(_) => Some(FieldType::Int), - Field::Int8(_) => Some(FieldType::Int), - Field::I128(_) => Some(FieldType::I128), - Field::Float(_) => Some(FieldType::Float), - Field::Boolean(_) => Some(FieldType::Boolean), - Field::String(_) => Some(FieldType::String), - Field::Text(_) => Some(FieldType::Text), - Field::Binary(_) => Some(FieldType::Binary), - Field::Decimal(_) => Some(FieldType::Decimal), - Field::Timestamp(_) => Some(FieldType::Timestamp), - Field::Date(_) => Some(FieldType::Date), - Field::Json(_) => Some(FieldType::Json), - Field::Point(_) => Some(FieldType::Point), - Field::Duration(_) => Some(FieldType::Duration), - Field::Null => None, - } - } - - pub fn as_uint(&self) -> Option { - match self { - Field::UInt(i) => Some(*i), - Field::Json(v) => v.as_number()?.to_u64(), - _ => None, - } - } - - pub fn as_u128(&self) -> Option { - match self { - Field::U128(i) => Some(*i), - Field::Json(j) => Some(j.to_f64_lossy()? as u128), - _ => None, - } - } - - pub fn is_u128(&self) -> bool { - matches!(self, Field::U128(_)) - } - - pub fn as_int(&self) -> Option { - match self { - Field::Int(i) => Some(*i), - Field::Json(j) => j.to_i64(), - _ => None, - } - } - - pub fn as_i128(&self) -> Option { - match self { - Field::I128(i) => Some(*i), - Field::Json(j) => Some(j.to_f64_lossy()? as i128), - _ => None, - } - } - - pub fn is_i128(&self) -> bool { - matches!(self, Field::I128(_)) - } - - pub fn as_float(&self) -> Option { - match self { - Field::Float(f) => Some(f.0), - Field::Json(j) => j.to_f64_lossy(), - _ => None, - } - } - - pub fn as_boolean(&self) -> Option { - match self { - Field::Boolean(b) => Some(*b), - Field::Json(j) => j.to_bool(), - _ => None, - } - } - - pub fn as_string(&self) -> Option<&str> { - match self { - Field::String(s) => Some(s), - Field::Json(j) => Some(j.as_string()?.as_str()), - _ => None, - } - } - - pub fn as_text(&self) -> Option<&str> { - match self { - Field::Text(s) => Some(s), - Field::Json(j) => Some(j.as_string()?.as_str()), - _ => None, - } - } - - pub fn as_binary(&self) -> Option<&[u8]> { - match self { - Field::Binary(b) => Some(b), - _ => None, - } - } - - pub fn as_decimal(&self) -> Option { - match self { - Field::Decimal(d) => Some(*d), - _ => None, - } - } - - pub fn is_decimal(&self) -> bool { - matches!(self, Field::Decimal(_)) - } - - pub fn as_timestamp(&self) -> Option> { - match self { - Field::Timestamp(t) => Some(*t), - _ => None, - } - } - - pub fn as_date(&self) -> Option { - match self { - Field::Date(d) => Some(*d), - _ => None, - } - } - - pub fn as_json(&self) -> Option<&JsonValue> { - match self { - Field::Json(b) => Some(b), - _ => None, - } - } - - pub fn as_point(&self) -> Option { - match self { - Field::Point(b) => Some(*b), - _ => None, - } - } - - pub fn as_duration(&self) -> Option { - match self { - Field::UInt(d) => Some(DozerDuration( - Duration::from_nanos(*d), - TimeUnit::Nanoseconds, - )), - Field::U128(d) => Some(DozerDuration( - Duration::from_nanos(*d as u64), - TimeUnit::Nanoseconds, - )), - Field::Int(d) => Some(DozerDuration( - Duration::from_nanos(*d as u64), - TimeUnit::Nanoseconds, - )), - Field::I128(d) => Some(DozerDuration( - Duration::from_nanos(*d as u64), - TimeUnit::Nanoseconds, - )), - Field::Duration(d) => Some(*d), - _ => None, - } - } - - pub fn as_null(&self) -> Option<()> { - match self { - Field::Null => Some(()), - Field::Json(j) => { - if j.is_null() { - Some(()) - } else { - None - } - } - _ => None, - } - } - - pub fn to_uint(&self) -> Option { - match self { - Field::UInt(u) => Some(*u), - Field::U128(u) => u64::from_u128(*u), - Field::Int(i) => u64::from_i64(*i), - Field::I128(i) => u64::from_i128(*i), - Field::Float(f) => u64::from_f64(f.0), - Field::Decimal(d) => d.to_u64(), - Field::String(s) => s.parse::().ok(), - Field::Text(s) => s.parse::().ok(), - Field::Json(j) => match j.destructure_ref() { - DestructuredRef::Number(n) => Some(n.to_f64_lossy() as u64), - DestructuredRef::String(s) => s.parse::().ok(), - _ => None, - }, - Field::Null => Some(0_u64), - _ => None, - } - } - - pub fn to_u128(&self) -> Option { - match self { - Field::UInt(u) => u128::from_u64(*u), - Field::U128(u) => Some(*u), - Field::Int(i) => u128::from_i64(*i), - Field::I128(i) => u128::from_i128(*i), - Field::Float(f) => u128::from_f64(f.0), - Field::Decimal(d) => d.to_u128(), - Field::String(s) => s.parse::().ok(), - Field::Text(s) => s.parse::().ok(), - Field::Json(j) => match j.destructure_ref() { - DestructuredRef::Number(n) => Some(n.to_f64_lossy() as u128), - DestructuredRef::String(s) => s.parse::().ok(), - _ => None, - }, - Field::Null => Some(0), - _ => None, - } - } - - pub fn to_int8(&self) -> Option { - match self { - Field::UInt(u) => i8::from_u64(*u), - Field::U128(u) => i8::from_u128(*u), - Field::Int(i) => Some((*i).try_into().unwrap()), - Field::I128(i) => i8::from_i128(*i), - Field::Float(f) => i8::from_f64(f.0), - Field::Decimal(d) => d.to_i8(), - Field::String(s) => s.parse::().ok(), - Field::Text(s) => s.parse::().ok(), - Field::Json(j) => match j.destructure_ref() { - DestructuredRef::Number(n) => Some(n.to_f64_lossy() as i8), - DestructuredRef::String(s) => s.parse::().ok(), - _ => None, - }, - Field::Null => Some(0_i8), - _ => None, - } - } - - pub fn to_int(&self) -> Option { - match self { - Field::UInt(u) => i64::from_u64(*u), - Field::U128(u) => i64::from_u128(*u), - Field::Int(i) => Some(*i), - Field::I128(i) => i64::from_i128(*i), - Field::Float(f) => i64::from_f64(f.0), - Field::Decimal(d) => d.to_i64(), - Field::String(s) => s.parse::().ok(), - Field::Text(s) => s.parse::().ok(), - Field::Json(j) => match j.destructure_ref() { - DestructuredRef::Number(n) => Some(n.to_f64_lossy() as i64), - DestructuredRef::String(s) => s.parse::().ok(), - _ => None, - }, - Field::Null => Some(0_i64), - _ => None, - } - } - - pub fn to_i128(&self) -> Option { - match self { - Field::UInt(u) => i128::from_u64(*u), - Field::U128(u) => i128::from_u128(*u), - Field::Int(i) => i128::from_i64(*i), - Field::I128(i) => Some(*i), - Field::Float(f) => i128::from_f64(f.0), - Field::Decimal(d) => d.to_i128(), - Field::String(s) => s.parse::().ok(), - Field::Text(s) => s.parse::().ok(), - Field::Json(j) => match j.destructure_ref() { - DestructuredRef::Number(n) => Some(n.to_f64_lossy() as i128), - DestructuredRef::String(s) => s.parse::().ok(), - _ => None, - }, - Field::Null => Some(0_i128), - _ => None, - } - } - - pub fn to_float(&self) -> Option { - match self { - Field::UInt(u) => f64::from_u64(*u), - Field::U128(u) => f64::from_u128(*u), - Field::Int(i) => f64::from_i64(*i), - Field::I128(i) => f64::from_i128(*i), - Field::Float(f) => Some(f.0), - Field::Decimal(d) => d.to_f64(), - Field::String(s) => s.parse::().ok(), - Field::Text(s) => s.parse::().ok(), - Field::Json(j) => match j.destructure_ref() { - DestructuredRef::Number(n) => Some(n.to_f64_lossy()), - DestructuredRef::String(s) => s.parse::().ok(), - _ => None, - }, - Field::Null => Some(0_f64), - _ => None, - } - } - - pub fn to_boolean(&self) -> Option { - match self { - Field::UInt(u) => Some(*u > 0_u64), - Field::U128(u) => Some(*u > 0_u128), - Field::Int(i) => Some(*i > 0_i64), - Field::I128(i) => Some(*i > 0_i128), - Field::Float(i) => Some(i.0 > 0_f64), - Field::Decimal(i) => Some(i.gt(&Decimal::from(0_u64))), - Field::Boolean(b) => Some(*b), - Field::String(s) => s.parse::().ok(), - Field::Text(s) => s.parse::().ok(), - Field::Json(j) => j.to_bool(), - Field::Null => Some(false), - _ => None, - } - } - - pub fn to_text(&self) -> String { - self.to_string() - } - - pub fn to_binary(&self) -> Option<&[u8]> { - match self { - Field::Binary(b) => Some(b), - _ => None, - } - } - - pub fn to_decimal(&self) -> Option { - match self { - Field::UInt(u) => Decimal::from_u64(*u), - Field::U128(u) => Decimal::from_u128(*u), - Field::Int(i) => Decimal::from_i64(*i), - Field::I128(i) => Decimal::from_i128(*i), - Field::Float(f) => Decimal::from_f64_retain(f.0), - Field::Decimal(d) => Some(*d), - Field::String(s) => Decimal::from_str_exact(s).ok(), - Field::Null => Some(Decimal::from(0)), - _ => None, - } - } - - pub fn to_timestamp(&self) -> Option> { - match self { - Field::String(s) => DateTime::parse_from_rfc3339(s.as_str()).ok(), - Field::Text(s) => DateTime::parse_from_rfc3339(s.as_str()).ok(), - Field::Timestamp(t) => Some(*t), - Field::Date(d) => match Utc.with_ymd_and_hms(d.year(), d.month(), d.day(), 0, 0, 0) { - LocalResult::Single(v) => Some(v.into()), - _ => unreachable!(), - }, - _ => None, - } - } - - pub fn to_date(&self) -> Option { - match self { - Field::String(s) => NaiveDate::parse_from_str(s, DATE_FORMAT).ok(), - Field::Timestamp(t) => Some(t.date_naive()), - Field::Date(d) => Some(*d), - _ => None, - } - } - - pub fn to_json(&self) -> Option { - match self { - Field::Json(b) => Some(b.to_owned()), - Field::UInt(u) => Some((*u).into()), - Field::U128(u) => Some((*u as f64).into()), - Field::Int(i) => Some((*i).into()), - Field::I128(i) => Some((*i as f64).into()), - Field::Float(OrderedFloat(f)) => Some((*f).into()), - Field::Boolean(b) => Some((*b).into()), - Field::String(s) => json_from_str(s.as_str()).ok(), - Field::Text(t) => Some(t.into()), - Field::Null => Some(JsonValue::NULL), - _ => None, - } - } - - pub fn to_point(&self) -> Option { - match self { - Field::Point(p) => Some(*p), - _ => None, - } - } - - pub fn to_duration(&self) -> Option { - match self { - Field::UInt(d) => Some(DozerDuration( - Duration::from_nanos(*d), - TimeUnit::Nanoseconds, - )), - Field::U128(d) => Some(DozerDuration( - Duration::from_nanos(u64::try_from(*d).ok()?), - TimeUnit::Nanoseconds, - )), - Field::Int(d) => Some(DozerDuration( - Duration::from_nanos(u64::try_from(*d).ok()?), - TimeUnit::Nanoseconds, - )), - Field::I128(d) => Some(DozerDuration( - Duration::from_nanos(u64::try_from(*d).ok()?), - TimeUnit::Nanoseconds, - )), - Field::Duration(d) => Some(*d), - Field::String(d) | Field::Text(d) => DozerDuration::from_str(d.as_str()).ok(), - Field::Null => Some(DozerDuration( - Duration::from_nanos(0), - TimeUnit::Nanoseconds, - )), - _ => None, - } - } - - pub fn to_null(&self) -> Option<()> { - match self { - Field::Null => Some(()), - _ => None, - } - } -} - -impl Display for Field { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - match self { - Field::UInt(u) => write!(f, "{u}"), - Field::U128(u) => write!(f, "{u}"), - Field::Int(i) => write!(f, "{i}"), - Field::Int8(i) => write!(f, "{i}"), - Field::I128(i) => write!(f, "{i}"), - Field::Float(OrderedFloat(fl)) => write!(f, "{fl}"), - Field::Decimal(d) => write!(f, "{d}"), - Field::Boolean(false) => { - write!(f, "FALSE") - } - Field::Boolean(true) => { - write!(f, "TRUE") - } - Field::String(s) => f.write_str(s), - Field::Text(t) => write!(f, "{t}"), - Field::Date(d) => write!(f, "{}", d.format(DATE_FORMAT)), - Field::Timestamp(t) => write!(f, "{}", t.to_rfc3339()), - Field::Binary(b) => write!(f, "{b:X?}"), - Field::Json(j) => write!(f, "{j:?}"), - Field::Point(p) => { - let (x, y) = p.0.x_y(); - write!(f, "POINT({}, {})", x.0, y.0) - } - Field::Duration(d) => write!(f, "PT{},{:09}S", d.0.as_secs(), d.0.subsec_nanos()), - Field::Null => write!(f, ""), - } - } -} - -#[derive( - Clone, - Copy, - Serialize, - Deserialize, - Debug, - PartialEq, - Eq, - PartialOrd, - Ord, - Hash, - bincode::Encode, - bincode::Decode, -)] -/// All field types supported in Dozer. -pub enum FieldType { - /// Unsigned 64-bit integer. - UInt, - /// Unsigned 128-bit integer. - U128, - /// Signed 64-bit integer. - Int, - Int8, - /// Signed 128-bit integer. - I128, - /// 64-bit floating point number. - Float, - /// `true` or `false`. - Boolean, - /// A string with limited length. - String, - /// A long string. - Text, - /// Raw bytes. - Binary, - /// `Decimal` represents a 128 bit representation of a fixed-precision decimal number. - /// The finite set of values of type `Decimal` are of the form m / 10e, - /// where m is an integer such that -296 < m < 296, and e is an integer - /// between 0 and 28 inclusive. - Decimal, - /// Timestamp up to nanoseconds. - Timestamp, - /// Allows for every date from Jan 1, 262145 BCE to Dec 31, 262143 CE. - Date, - /// JsonValue. - Json, - /// A geographic point. - Point, - /// Duration up to nanoseconds. - Duration, -} - -impl TryFrom<&str> for FieldType { - type Error = String; - - fn try_from(value: &str) -> Result { - let res = match value.to_lowercase().as_str() { - "uint" => FieldType::UInt, - "u128" => FieldType::U128, - "int" => FieldType::Int, - "i128" => FieldType::I128, - "float" => FieldType::Float, - "decimal" => FieldType::Decimal, - "boolean" => FieldType::Boolean, - "string" => FieldType::String, - "text" => FieldType::Text, - "binary" => FieldType::Binary, - "timestamp" => FieldType::Timestamp, - "date" => FieldType::Date, - "json" => FieldType::Json, - "jsonb" => FieldType::Json, - "json_array" => FieldType::Json, - "jsonb_array" => FieldType::Json, - "point" => FieldType::Point, - "duration" => FieldType::Duration, - _ => return Err(format!("Unsupported '{value}' type")), - }; - - Ok(res) - } -} - -impl Display for FieldType { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - match self { - FieldType::UInt => f.write_str("64-bit unsigned int"), - FieldType::U128 => f.write_str("128-bit unsigned int"), - FieldType::Int => f.write_str("64-bit int"), - FieldType::Int8 => f.write_str("8-bit int"), - - FieldType::I128 => f.write_str("128-bit int"), - FieldType::Float => f.write_str("float"), - FieldType::Boolean => f.write_str("boolean"), - FieldType::String => f.write_str("string"), - FieldType::Text => f.write_str("text"), - FieldType::Binary => f.write_str("binary"), - FieldType::Decimal => f.write_str("decimal"), - FieldType::Timestamp => f.write_str("timestamp"), - FieldType::Date => f.write_str("date"), - FieldType::Json => f.write_str("json"), - FieldType::Point => f.write_str("point"), - FieldType::Duration => f.write_str("duration"), - } - } -} - -/// Can't put it in `tests` module because of -/// and we need this function in `dozer-cache`. -pub fn field_test_cases() -> impl Iterator { - [ - Field::UInt(0_u64), - Field::UInt(1_u64), - Field::U128(0_u128), - Field::U128(1_u128), - Field::Int(0_i64), - Field::Int(1_i64), - Field::I128(0_i128), - Field::I128(1_i128), - Field::Float(OrderedFloat::from(0_f64)), - Field::Float(OrderedFloat::from(1_f64)), - Field::Decimal(Decimal::new(0, 0)), - Field::Decimal(Decimal::new(1, 0)), - Field::Boolean(true), - Field::Boolean(false), - Field::String("".to_string()), - Field::String("1".to_string()), - Field::Text("".to_string()), - Field::Text("1".to_string()), - Field::Binary(vec![]), - Field::Binary(vec![1]), - Field::Timestamp(DateTime::from(Utc.timestamp_millis_opt(0).unwrap())), - Field::Timestamp(DateTime::parse_from_rfc3339("2020-01-01T00:00:00Z").unwrap()), - Field::Date(NaiveDate::from_ymd_opt(1970, 1, 1).unwrap()), - Field::Date(NaiveDate::from_ymd_opt(2020, 1, 1).unwrap()), - Field::Json(Vec::::new().into()), - Field::Json( - vec![ - 123_f64, 34_f64, 97_f64, 98_f64, 99_f64, 34_f64, 58_f64, 34_f64, 102_f64, 111_f64, - 111_f64, 34_f64, - ] - .into(), - ), - Field::Null, - ] - .into_iter() -} - -pub fn arrow_field_test_cases() -> impl Iterator { - field_test_cases().filter(|case| !case.is_u128() && !case.is_i128() && !case.is_decimal()) -} - -pub fn arrow_field_test_cases_schema() -> Schema { - Schema::default() - .field( - FieldDefinition::new( - "uint1".to_string(), - FieldType::UInt, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "uint2".to_string(), - FieldType::UInt, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "int1".to_string(), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "int2".to_string(), - FieldType::Int, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "float1".to_string(), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "float2".to_string(), - FieldType::Float, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "boolean1".to_string(), - FieldType::Boolean, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "boolean2".to_string(), - FieldType::Boolean, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "string1".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "string2".to_string(), - FieldType::String, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "text1".to_string(), - FieldType::Text, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "text2".to_string(), - FieldType::Text, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "binary1".to_string(), - FieldType::Binary, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "binary2".to_string(), - FieldType::Binary, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "timestamp1".to_string(), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "timestamp2".to_string(), - FieldType::Timestamp, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "date1".to_string(), - FieldType::Date, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "date2".to_string(), - FieldType::Date, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "json1".to_string(), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "json2".to_string(), - FieldType::Json, - false, - SourceDefinition::Dynamic, - ), - false, - ) - .field( - FieldDefinition::new( - "null".to_string(), - FieldType::String, - true, - SourceDefinition::Dynamic, - ), - false, - ) - .clone() -} - -#[cfg(any( - feature = "python-auto-initialize", - feature = "python-extension-module" -))] -impl pyo3::ToPyObject for Field { - fn to_object(&self, py: pyo3::Python<'_>) -> pyo3::PyObject { - match self { - Field::UInt(val) => val.to_object(py), - Field::U128(val) => val.to_object(py), - Field::Int(val) => val.to_object(py), - Field::Int8(val) => val.to_object(py), - Field::I128(val) => val.to_object(py), - Field::Float(val) => val.0.to_object(py), - Field::Decimal(val) => val.to_f64().unwrap().to_object(py), - Field::Boolean(val) => val.to_object(py), - Field::String(val) => val.to_object(py), - Field::Text(val) => val.to_object(py), - Field::Binary(val) => val.to_object(py), - Field::Timestamp(val) => val.timestamp().to_object(py), - Field::Date(val) => { - pyo3::types::PyDate::new(py, val.year(), val.month() as u8, val.day() as u8) - .unwrap() - .to_object(py) - } - Field::Json(_val) => todo!(), - Field::Point(_val) => todo!(), - Field::Duration(_d) => todo!(), - Field::Null => unreachable!(), - } - } -} diff --git a/dozer-types/src/types/mod.rs b/dozer-types/src/types/mod.rs deleted file mode 100644 index 31d111d289..0000000000 --- a/dozer-types/src/types/mod.rs +++ /dev/null @@ -1,560 +0,0 @@ -use chrono::{DateTime, FixedOffset}; -use geo::{point, GeodesicDistance, Point}; -use ordered_float::OrderedFloat; -use std::array::TryFromSliceError; -use std::cmp::Ordering; -use std::fmt::{Display, Formatter}; -use std::hash::Hash; -use std::str::FromStr; - -use crate::errors::types::TypeError; -use crate::node::OpIdentifier; -use prettytable::{Cell, Row, Table}; -use serde::{self, Deserialize, Serialize}; - -pub mod field; - -#[cfg(test)] -mod tests; - -use crate::errors::internal::BoxedError; -use crate::errors::types::TypeError::InvalidFieldValue; -pub use field::{field_test_cases, Field, FieldType, DATE_FORMAT}; - -#[derive( - Clone, - Serialize, - Deserialize, - Debug, - PartialEq, - Eq, - PartialOrd, - Ord, - Default, - bincode::Encode, - bincode::Decode, -)] -pub enum SourceDefinition { - Table { - connection: String, - name: String, - }, - Alias { - name: String, - }, - #[default] - Dynamic, -} - -#[derive(Clone, Serialize, Deserialize, Debug, PartialEq, Eq, PartialOrd, Ord)] -pub struct FieldDefinition { - pub name: String, - pub typ: FieldType, - pub nullable: bool, - #[serde(default)] - pub source: SourceDefinition, - pub description: Option, -} - -impl FieldDefinition { - pub fn new(name: String, typ: FieldType, nullable: bool, source: SourceDefinition) -> Self { - Self { - name, - typ, - nullable, - source, - description: None, - } - } - - pub fn check_from(&self, table_name: String) -> bool { - match &self.source { - SourceDefinition::Table { name, .. } => *name == table_name, - SourceDefinition::Alias { name } => *name == table_name, - SourceDefinition::Dynamic => false, - } - } -} - -#[derive(Clone, Serialize, Deserialize, Debug, PartialEq, Eq, Default)] -pub struct Schema { - /// fields contains a list of FieldDefinition for all the fields that appear in a record. - /// Not necessarily all these fields will end up in the final object structure stored in - /// the cache. Some fields might only be used for indexing purposes only. - pub fields: Vec, - - /// Indexes of the fields forming the primary key for this schema. If the value is empty - /// only Insert Operation are supported. Updates and Deletes are not supported without a - /// primary key definition - #[serde(default)] - pub primary_index: Vec, -} - -impl Schema { - pub fn new() -> Schema { - Self::default() - } - - pub fn field(&mut self, f: FieldDefinition, pk: bool) -> &mut Self { - self.fields.push(f); - if pk { - self.primary_index.push(&self.fields.len() - 1) - } - self - } - - pub fn get_field_index(&self, name: &str) -> Result<(usize, &FieldDefinition), TypeError> { - let r = self - .fields - .iter() - .enumerate() - .find(|f| f.1.name.as_str() == name); - match r { - Some(v) => Ok(v), - _ => Err(TypeError::InvalidFieldName(name.to_string())), - } - } - - pub fn print(&self) -> Table { - let mut table = Table::new(); - table.add_row(row!["Field", "Type", "Nullable", "PK"]); - for (index, field) in self.fields.iter().enumerate() { - table.add_row(row![ - field.name, - format!("{:?}", field.typ), - field.nullable, - self.primary_index.contains(&index) - ]); - } - table - } - - /// Returns if this schema is append only. - /// - /// Append only schemas enable additional optimizations, however, the connectors and processors haven't properly implemented this yet. - pub fn is_append_only(&self) -> bool { - false - } -} - -impl Display for Schema { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - let table = self.print(); - table.fmt(f) - } -} - -#[derive(Clone, Serialize, Deserialize, Debug, PartialEq, Eq, bincode::Encode, bincode::Decode)] -pub enum IndexDefinition { - /// The sorted inverted index, supporting `Eq` filter on multiple fields and `LT`, `LTE`, `GT`, `GTE` filter on at most one field. - SortedInverted(Vec), - /// Full text index, supporting `Contains`, `MatchesAny` and `MatchesAll` filter on exactly one field. - FullText(usize), -} - -pub type SchemaWithIndex = (Schema, Vec); - -pub type Timestamp = DateTime; - -#[derive( - Clone, Serialize, Deserialize, Debug, PartialEq, Eq, Hash, bincode::Encode, bincode::Decode, -)] -pub struct Lifetime { - #[bincode(with_serde)] - pub reference: Timestamp, - pub duration: std::time::Duration, -} - -#[derive( - Debug, - Clone, - Serialize, - Deserialize, - PartialEq, - Eq, - Hash, - Default, - bincode::Encode, - bincode::Decode, -)] -pub struct Record { - /// List of values, following the definitions of `fields` of the associated schema - pub values: Vec, - - /// Time To Live for this record. If the value is None, the record will never expire. - pub lifetime: Option, -} - -impl Record { - pub fn new(values: Vec) -> Record { - Record { - values, - lifetime: None, - } - } - - pub fn nulls_from_schema(schema: &Schema) -> Record { - Self::nulls(schema.fields.len()) - } - - pub fn nulls(size: usize) -> Record { - Record { - values: vec![Field::Null; size], - lifetime: None, - } - } - - pub fn values(&self) -> &[Field] { - &self.values - } - - pub fn iter(&self) -> core::slice::Iter<'_, Field> { - self.values.iter() - } - - pub fn set_value(&mut self, idx: usize, value: Field) { - self.values[idx] = value; - } - - pub fn push_value(&mut self, value: Field) { - self.values.push(value); - } - - pub fn get_value(&self, idx: usize) -> Result<&Field, TypeError> { - match self.values.get(idx) { - Some(f) => Ok(f), - _ => Err(TypeError::InvalidFieldIndex(idx)), - } - } - - pub fn appended(existing: &Record, additional: &[Field]) -> Self { - let mut values = existing.values.clone(); - values.extend_from_slice(additional); - Self::new(values) - } - - pub fn get_key_fields(&self, schema: &Schema) -> Vec { - self.get_fields_by_indexes(&schema.primary_index) - } - - pub fn get_fields_by_indexes(&self, indexes: &[usize]) -> Vec { - debug_assert!(!&indexes.is_empty(), "Primary key indexes cannot be empty"); - - let mut fields = Vec::with_capacity(indexes.len()); - for i in indexes { - fields.push(self.values[*i].clone()); - } - fields - } - - pub fn get_key(&self, indexes: &Vec) -> Vec { - debug_assert!(!indexes.is_empty(), "Primary key indexes cannot be empty"); - - let mut tot_size = 0_usize; - let mut buffers = Vec::>::with_capacity(indexes.len()); - for i in indexes { - let bytes = self.values[*i].encode(); - tot_size += bytes.len(); - buffers.push(bytes); - } - - let mut res_buffer = Vec::::with_capacity(tot_size); - for i in buffers { - res_buffer.extend(i); - } - res_buffer - } - - pub fn set_lifetime(&mut self, lifetime: Option) { - self.lifetime = lifetime; - } - - pub fn get_lifetime(&self) -> Option { - self.lifetime.clone() - } -} - -impl Display for Record { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - let v = self - .values - .iter() - .map(|f| Cell::new(&f.to_string())) - .collect::>(); - - let mut table = Table::new(); - table.add_row(Row::new(v)); - table.fmt(f) - } -} - -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, bincode::Encode, bincode::Decode)] -/// A CDC event. -pub enum Operation { - Delete { old: Record }, - Insert { new: Record }, - Update { old: Record, new: Record }, - BatchInsert { new: Vec }, -} - -pub type PortHandle = u16; - -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize, bincode::Encode, bincode::Decode)] -pub struct TableOperation { - pub id: Option, - pub op: Operation, - /// For outputting an operation, node should fill the output port. - /// For received operation, the port is the input port. - /// Port mapping is done in forwarders. - pub port: PortHandle, -} - -impl TableOperation { - pub fn without_id(op: Operation, port: PortHandle) -> Self { - Self { id: None, op, port } - } -} - -// Helpful in interacting with external systems during ingestion and querying -// For example, nanoseconds can overflow. -#[derive( - Clone, - Copy, - Serialize, - Deserialize, - Debug, - PartialEq, - Eq, - PartialOrd, - Ord, - Hash, - bincode::Decode, - bincode::Encode, -)] -#[cfg_attr(feature = "arbitrary", derive(arbitrary::Arbitrary))] -pub enum TimeUnit { - Seconds, - Milliseconds, - Microseconds, - Nanoseconds, -} - -impl Display for TimeUnit { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - match self { - TimeUnit::Seconds => f.write_str("Seconds"), - TimeUnit::Milliseconds => f.write_str("Milliseconds"), - TimeUnit::Microseconds => f.write_str("Microseconds"), - TimeUnit::Nanoseconds => f.write_str("Nanoseconds"), - } - } -} - -impl FromStr for TimeUnit { - type Err = TypeError; - - fn from_str(str: &str) -> Result { - let error = || InvalidFieldValue { - field_type: FieldType::Duration, - nullable: false, - value: str.to_string(), - }; - let string = str.parse::().map_err(|_| error())?; - match string.as_str() { - "Seconds" => Ok(TimeUnit::Seconds), - "Milliseconds" => Ok(TimeUnit::Milliseconds), - "Microseconds" => Ok(TimeUnit::Microseconds), - "Nanoseconds" => Ok(TimeUnit::Nanoseconds), - &_ => Err(error()), - } - } -} - -impl TimeUnit { - pub fn to_bytes(&self) -> [u8; 1] { - match self { - TimeUnit::Seconds => [0_u8], - TimeUnit::Milliseconds => [1_u8], - TimeUnit::Microseconds => [2_u8], - TimeUnit::Nanoseconds => [3_u8], - } - } - - pub fn from_bytes(bytes: &[u8]) -> Result { - match bytes { - [0_u8] => Ok(TimeUnit::Seconds), - [1_u8] => Ok(TimeUnit::Milliseconds), - [2_u8] => Ok(TimeUnit::Microseconds), - [3_u8] => Ok(TimeUnit::Nanoseconds), - _ => Err(BoxedError::from("Unsupported unit".to_string())), - } - } -} - -#[derive( - Clone, - Copy, - Debug, - Serialize, - Deserialize, - Eq, - PartialEq, - Hash, - bincode::Encode, - bincode::Decode, -)] -#[cfg_attr(feature = "arbitrary", derive(arbitrary::Arbitrary))] -pub struct DozerDuration(pub std::time::Duration, pub TimeUnit); - -impl Ord for DozerDuration { - fn cmp(&self, other: &Self) -> Ordering { - std::time::Duration::cmp(&self.0, &other.0) - } -} - -impl PartialOrd for DozerDuration { - fn partial_cmp(&self, other: &Self) -> Option { - Some(self.cmp(other)) - } -} - -impl FromStr for DozerDuration { - type Err = TypeError; - - fn from_str(str: &str) -> Result { - let error = || InvalidFieldValue { - field_type: FieldType::Duration, - nullable: false, - value: str.to_string(), - }; - let val = str.parse::().map_err(|_| error())?; - Ok(Self( - std::time::Duration::from_nanos(val), - TimeUnit::Nanoseconds, - )) - } -} - -impl DozerDuration { - pub fn to_bytes(&self) -> [u8; 17] { - let mut result = [0_u8; 17]; - result[0..1].copy_from_slice(&self.1.to_bytes()); - result[1..17].copy_from_slice(&self.0.as_nanos().to_be_bytes()); - result - } - - pub fn from_bytes(bytes: &[u8]) -> Result { - let unit = TimeUnit::from_bytes(bytes[0..1].try_into()?).unwrap(); - let val = - std::time::Duration::from_nanos(u128::from_be_bytes(bytes[1..17].try_into()?) as u64); - - Ok(DozerDuration(val, unit)) - } -} - -#[derive( - Clone, - Copy, - Debug, - Serialize, - Deserialize, - Eq, - PartialEq, - Hash, - bincode::Encode, - bincode::Decode, -)] -pub struct DozerPoint(#[bincode(with_serde)] pub Point>); - -#[cfg(feature = "arbitrary")] -impl<'a> arbitrary::Arbitrary<'a> for DozerPoint { - fn arbitrary(u: &mut arbitrary::Unstructured<'a>) -> arbitrary::Result { - let x = self::field::arbitrary_float(u)?; - let y = self::field::arbitrary_float(u)?; - - Ok(Self(geo::Point::new(x, y))) - } -} - -impl GeodesicDistance> for DozerPoint { - fn geodesic_distance(&self, rhs: &Self) -> OrderedFloat { - let f = point! { x: self.0.x().0, y: self.0.y().0 }; - let t = point! { x: rhs.0.x().0, y: rhs.0.y().0 }; - OrderedFloat(f.geodesic_distance(&t)) - } -} - -impl Ord for DozerPoint { - fn cmp(&self, other: &Self) -> Ordering { - if self.0.x() == other.0.x() && self.0.y() == other.0.y() { - Ordering::Equal - } else if self.0.x() > other.0.x() - || (self.0.x() == other.0.x() && self.0.y() > other.0.y()) - { - Ordering::Greater - } else { - Ordering::Less - } - } -} - -impl PartialOrd for DozerPoint { - fn partial_cmp(&self, other: &Self) -> Option { - Some(self.cmp(other)) - } -} - -impl FromStr for DozerPoint { - type Err = TypeError; - - fn from_str(str: &str) -> Result { - let error = || InvalidFieldValue { - field_type: FieldType::Point, - nullable: false, - value: str.to_string(), - }; - - let s = str.replace('(', ""); - let s = s.replace(')', ""); - let mut cs = s.split(','); - let x = cs - .next() - .ok_or_else(error)? - .parse::() - .map_err(|_| error())?; - let y = cs - .next() - .ok_or_else(error)? - .parse::() - .map_err(|_| error())?; - Ok(Self(Point::from((OrderedFloat(x), OrderedFloat(y))))) - } -} - -impl From<(f64, f64)> for DozerPoint { - fn from((x, y): (f64, f64)) -> Self { - Self(point! {x: OrderedFloat(x), y: OrderedFloat(y)}) - } -} - -impl Display for DozerPoint { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - f.write_str(&format!("{:?}", self.0.x_y())) - } -} - -impl DozerPoint { - pub fn to_bytes(&self) -> [u8; 16] { - let mut result = [0_u8; 16]; - result[0..8].copy_from_slice(&self.0.x().to_be_bytes()); - result[8..16].copy_from_slice(&self.0.y().to_be_bytes()); - result - } - - pub fn from_bytes(bytes: &[u8]) -> Result { - let x = f64::from_be_bytes(bytes[0..8].try_into()?); - let y = f64::from_be_bytes(bytes[8..16].try_into()?); - - Ok(DozerPoint::from((x, y))) - } -} diff --git a/dozer-types/src/types/tests.rs b/dozer-types/src/types/tests.rs deleted file mode 100644 index 9bd4180637..0000000000 --- a/dozer-types/src/types/tests.rs +++ /dev/null @@ -1,567 +0,0 @@ -use crate::types::{field_test_cases, DozerDuration, DozerPoint, Field, TimeUnit}; -use chrono::{DateTime, NaiveDate, TimeZone, Utc}; -use ordered_float::OrderedFloat; -use rust_decimal::Decimal; - -#[test] -fn data_encoding_len_must_agree_with_encode() { - for field in field_test_cases() { - let bytes = field.encode_data(); - assert_eq!(bytes.len(), field.data_encoding_len()); - } -} - -#[test] -fn test_as_conversion() { - let field = Field::UInt(1); - assert!(field.as_uint().is_some()); - assert!(field.as_int().is_none()); - assert!(field.as_u128().is_none()); - assert!(field.as_i128().is_none()); - assert!(field.as_float().is_none()); - assert!(field.as_boolean().is_none()); - assert!(field.as_string().is_none()); - assert!(field.as_text().is_none()); - assert!(field.as_binary().is_none()); - assert!(field.as_decimal().is_none()); - assert!(field.as_timestamp().is_none()); - assert!(field.as_date().is_none()); - assert!(field.as_json().is_none()); - assert!(field.as_point().is_none()); - assert!(field.as_duration().is_some()); - assert!(field.as_null().is_none()); - - let field = Field::Int(1); - assert!(field.as_uint().is_none()); - assert!(field.as_int().is_some()); - assert!(field.as_u128().is_none()); - assert!(field.as_i128().is_none()); - assert!(field.as_float().is_none()); - assert!(field.as_boolean().is_none()); - assert!(field.as_string().is_none()); - assert!(field.as_text().is_none()); - assert!(field.as_binary().is_none()); - assert!(field.as_decimal().is_none()); - assert!(field.as_timestamp().is_none()); - assert!(field.as_date().is_none()); - assert!(field.as_json().is_none()); - assert!(field.as_point().is_none()); - assert!(field.as_duration().is_some()); - assert!(field.as_null().is_none()); - - let field = Field::U128(1); - assert!(field.as_uint().is_none()); - assert!(field.as_int().is_none()); - assert!(field.as_u128().is_some()); - assert!(field.as_i128().is_none()); - assert!(field.as_float().is_none()); - assert!(field.as_boolean().is_none()); - assert!(field.as_string().is_none()); - assert!(field.as_text().is_none()); - assert!(field.as_binary().is_none()); - assert!(field.as_decimal().is_none()); - assert!(field.as_timestamp().is_none()); - assert!(field.as_date().is_none()); - assert!(field.as_json().is_none()); - assert!(field.as_point().is_none()); - assert!(field.as_duration().is_some()); - assert!(field.as_null().is_none()); - - let field = Field::I128(1); - assert!(field.as_uint().is_none()); - assert!(field.as_int().is_none()); - assert!(field.as_u128().is_none()); - assert!(field.as_i128().is_some()); - assert!(field.as_float().is_none()); - assert!(field.as_boolean().is_none()); - assert!(field.as_string().is_none()); - assert!(field.as_text().is_none()); - assert!(field.as_binary().is_none()); - assert!(field.as_decimal().is_none()); - assert!(field.as_timestamp().is_none()); - assert!(field.as_date().is_none()); - assert!(field.as_json().is_none()); - assert!(field.as_point().is_none()); - assert!(field.as_duration().is_some()); - assert!(field.as_null().is_none()); - - let field = Field::Float(OrderedFloat::from(1.0)); - assert!(field.as_uint().is_none()); - assert!(field.as_int().is_none()); - assert!(field.as_u128().is_none()); - assert!(field.as_i128().is_none()); - assert!(field.as_float().is_some()); - assert!(field.as_boolean().is_none()); - assert!(field.as_string().is_none()); - assert!(field.as_text().is_none()); - assert!(field.as_binary().is_none()); - assert!(field.as_decimal().is_none()); - assert!(field.as_timestamp().is_none()); - assert!(field.as_date().is_none()); - assert!(field.as_json().is_none()); - assert!(field.as_point().is_none()); - assert!(field.as_duration().is_none()); - assert!(field.as_null().is_none()); - - let field = Field::Boolean(true); - assert!(field.as_uint().is_none()); - assert!(field.as_int().is_none()); - assert!(field.as_u128().is_none()); - assert!(field.as_i128().is_none()); - assert!(field.as_float().is_none()); - assert!(field.as_boolean().is_some()); - assert!(field.as_string().is_none()); - assert!(field.as_text().is_none()); - assert!(field.as_binary().is_none()); - assert!(field.as_decimal().is_none()); - assert!(field.as_timestamp().is_none()); - assert!(field.as_date().is_none()); - assert!(field.as_json().is_none()); - assert!(field.as_point().is_none()); - assert!(field.as_duration().is_none()); - assert!(field.as_null().is_none()); - - let field = Field::String("".to_string()); - assert!(field.as_uint().is_none()); - assert!(field.as_int().is_none()); - assert!(field.as_u128().is_none()); - assert!(field.as_i128().is_none()); - assert!(field.as_float().is_none()); - assert!(field.as_boolean().is_none()); - assert!(field.as_string().is_some()); - assert!(field.as_text().is_none()); - assert!(field.as_binary().is_none()); - assert!(field.as_decimal().is_none()); - assert!(field.as_timestamp().is_none()); - assert!(field.as_date().is_none()); - assert!(field.as_json().is_none()); - assert!(field.as_point().is_none()); - assert!(field.as_duration().is_none()); - assert!(field.as_null().is_none()); - - let field = Field::Text("".to_string()); - assert!(field.as_uint().is_none()); - assert!(field.as_int().is_none()); - assert!(field.as_u128().is_none()); - assert!(field.as_i128().is_none()); - assert!(field.as_float().is_none()); - assert!(field.as_boolean().is_none()); - assert!(field.as_string().is_none()); - assert!(field.as_text().is_some()); - assert!(field.as_binary().is_none()); - assert!(field.as_decimal().is_none()); - assert!(field.as_timestamp().is_none()); - assert!(field.as_date().is_none()); - assert!(field.as_json().is_none()); - assert!(field.as_point().is_none()); - assert!(field.as_duration().is_none()); - assert!(field.as_null().is_none()); - - let field = Field::Binary(vec![]); - assert!(field.as_uint().is_none()); - assert!(field.as_int().is_none()); - assert!(field.as_u128().is_none()); - assert!(field.as_i128().is_none()); - assert!(field.as_float().is_none()); - assert!(field.as_boolean().is_none()); - assert!(field.as_string().is_none()); - assert!(field.as_text().is_none()); - assert!(field.as_binary().is_some()); - assert!(field.as_decimal().is_none()); - assert!(field.as_timestamp().is_none()); - assert!(field.as_date().is_none()); - assert!(field.as_json().is_none()); - assert!(field.as_point().is_none()); - assert!(field.as_duration().is_none()); - assert!(field.as_null().is_none()); - - let field = Field::Decimal(Decimal::from(1)); - assert!(field.as_uint().is_none()); - assert!(field.as_int().is_none()); - assert!(field.as_u128().is_none()); - assert!(field.as_i128().is_none()); - assert!(field.as_float().is_none()); - assert!(field.as_boolean().is_none()); - assert!(field.as_string().is_none()); - assert!(field.as_text().is_none()); - assert!(field.as_binary().is_none()); - assert!(field.as_decimal().is_some()); - assert!(field.as_timestamp().is_none()); - assert!(field.as_date().is_none()); - assert!(field.as_json().is_none()); - assert!(field.as_point().is_none()); - assert!(field.as_duration().is_none()); - assert!(field.as_null().is_none()); - - let field = Field::Timestamp(DateTime::from(Utc.timestamp_millis_opt(0).unwrap())); - assert!(field.as_uint().is_none()); - assert!(field.as_int().is_none()); - assert!(field.as_u128().is_none()); - assert!(field.as_i128().is_none()); - assert!(field.as_float().is_none()); - assert!(field.as_boolean().is_none()); - assert!(field.as_string().is_none()); - assert!(field.as_text().is_none()); - assert!(field.as_binary().is_none()); - assert!(field.as_decimal().is_none()); - assert!(field.as_timestamp().is_some()); - assert!(field.as_date().is_none()); - assert!(field.as_json().is_none()); - assert!(field.as_point().is_none()); - assert!(field.as_duration().is_none()); - assert!(field.as_null().is_none()); - - let field = Field::Date(NaiveDate::from_ymd_opt(1970, 1, 1).unwrap()); - assert!(field.as_uint().is_none()); - assert!(field.as_int().is_none()); - assert!(field.as_u128().is_none()); - assert!(field.as_i128().is_none()); - assert!(field.as_float().is_none()); - assert!(field.as_boolean().is_none()); - assert!(field.as_string().is_none()); - assert!(field.as_text().is_none()); - assert!(field.as_binary().is_none()); - assert!(field.as_decimal().is_none()); - assert!(field.as_timestamp().is_none()); - assert!(field.as_date().is_some()); - assert!(field.as_json().is_none()); - assert!(field.as_point().is_none()); - assert!(field.as_duration().is_none()); - assert!(field.as_null().is_none()); - - let field = Field::Json(Vec::::new().into()); - assert!(field.as_uint().is_none()); - assert!(field.as_int().is_none()); - assert!(field.as_u128().is_none()); - assert!(field.as_i128().is_none()); - assert!(field.as_float().is_none()); - assert!(field.as_boolean().is_none()); - assert!(field.as_string().is_none()); - assert!(field.as_text().is_none()); - assert!(field.as_binary().is_none()); - assert!(field.as_decimal().is_none()); - assert!(field.as_timestamp().is_none()); - assert!(field.as_date().is_none()); - assert!(field.as_json().is_some()); - assert!(field.as_point().is_none()); - assert!(field.as_duration().is_none()); - assert!(field.as_null().is_none()); - - let field = Field::Point(DozerPoint::from((0.0, 0.0))); - assert!(field.as_uint().is_none()); - assert!(field.as_int().is_none()); - assert!(field.as_u128().is_none()); - assert!(field.as_i128().is_none()); - assert!(field.as_float().is_none()); - assert!(field.as_boolean().is_none()); - assert!(field.as_string().is_none()); - assert!(field.as_text().is_none()); - assert!(field.as_binary().is_none()); - assert!(field.as_decimal().is_none()); - assert!(field.as_timestamp().is_none()); - assert!(field.as_date().is_none()); - assert!(field.as_json().is_none()); - assert!(field.as_point().is_some()); - assert!(field.as_duration().is_none()); - assert!(field.as_null().is_none()); - - let field = Field::Duration(DozerDuration( - std::time::Duration::from_nanos(123_u64), - TimeUnit::Nanoseconds, - )); - assert!(field.as_uint().is_none()); - assert!(field.as_int().is_none()); - assert!(field.as_u128().is_none()); - assert!(field.as_i128().is_none()); - assert!(field.as_float().is_none()); - assert!(field.as_boolean().is_none()); - assert!(field.as_string().is_none()); - assert!(field.as_text().is_none()); - assert!(field.as_binary().is_none()); - assert!(field.as_decimal().is_none()); - assert!(field.as_timestamp().is_none()); - assert!(field.as_date().is_none()); - assert!(field.as_json().is_none()); - assert!(field.as_point().is_none()); - assert!(field.as_duration().is_some()); - assert!(field.as_null().is_none()); - - let field = Field::Null; - assert!(field.as_uint().is_none()); - assert!(field.as_int().is_none()); - assert!(field.as_u128().is_none()); - assert!(field.as_i128().is_none()); - assert!(field.as_float().is_none()); - assert!(field.as_boolean().is_none()); - assert!(field.as_string().is_none()); - assert!(field.as_text().is_none()); - assert!(field.as_binary().is_none()); - assert!(field.as_decimal().is_none()); - assert!(field.as_timestamp().is_none()); - assert!(field.as_date().is_none()); - assert!(field.as_json().is_none()); - assert!(field.as_point().is_none()); - assert!(field.as_null().is_some()); -} - -#[test] -fn test_to_conversion() { - let field = Field::UInt(1); - assert!(field.to_uint().is_some()); - assert!(field.to_int().is_some()); - assert!(field.to_u128().is_some()); - assert!(field.to_i128().is_some()); - assert!(field.to_float().is_some()); - assert!(field.to_boolean().is_some()); - assert!(field.to_binary().is_none()); - assert!(field.to_decimal().is_some()); - assert!(field.to_timestamp().is_none()); - assert!(field.to_date().is_none()); - assert!(field.to_json().is_some()); - assert!(field.to_point().is_none()); - assert!(field.to_duration().is_some()); - assert!(field.to_null().is_none()); - - let field = Field::Int(1); - assert!(field.to_uint().is_some()); - assert!(field.to_int().is_some()); - assert!(field.to_u128().is_some()); - assert!(field.to_i128().is_some()); - assert!(field.to_float().is_some()); - assert!(field.to_boolean().is_some()); - assert!(field.to_binary().is_none()); - assert!(field.to_decimal().is_some()); - assert!(field.to_timestamp().is_none()); - assert!(field.to_date().is_none()); - assert!(field.to_json().is_some()); - assert!(field.to_point().is_none()); - assert!(field.to_duration().is_some()); - assert!(field.to_null().is_none()); - - let field = Field::U128(1); - assert!(field.to_uint().is_some()); - assert!(field.to_int().is_some()); - assert!(field.to_u128().is_some()); - assert!(field.to_i128().is_some()); - assert!(field.to_float().is_some()); - assert!(field.to_boolean().is_some()); - assert!(field.to_binary().is_none()); - assert!(field.to_decimal().is_some()); - assert!(field.to_timestamp().is_none()); - assert!(field.to_date().is_none()); - assert!(field.to_json().is_some()); - assert!(field.to_point().is_none()); - assert!(field.to_duration().is_some()); - assert!(field.to_null().is_none()); - - let field = Field::I128(1); - assert!(field.to_uint().is_some()); - assert!(field.to_int().is_some()); - assert!(field.to_u128().is_some()); - assert!(field.to_i128().is_some()); - assert!(field.to_float().is_some()); - assert!(field.to_boolean().is_some()); - assert!(field.to_binary().is_none()); - assert!(field.to_decimal().is_some()); - assert!(field.to_timestamp().is_none()); - assert!(field.to_date().is_none()); - assert!(field.to_json().is_some()); - assert!(field.to_point().is_none()); - assert!(field.to_duration().is_some()); - assert!(field.to_null().is_none()); - - let field = Field::Float(OrderedFloat::from(1.0)); - assert!(field.to_uint().is_some()); - assert!(field.to_int().is_some()); - assert!(field.to_u128().is_some()); - assert!(field.to_i128().is_some()); - assert!(field.to_float().is_some()); - assert!(field.to_boolean().is_some()); - assert!(field.to_binary().is_none()); - assert!(field.to_decimal().is_some()); - assert!(field.to_timestamp().is_none()); - assert!(field.to_date().is_none()); - assert!(field.to_json().is_some()); - assert!(field.to_point().is_none()); - assert!(field.to_duration().is_none()); - assert!(field.to_null().is_none()); - - let field = Field::Boolean(true); - assert!(field.to_uint().is_none()); - assert!(field.to_int().is_none()); - assert!(field.to_u128().is_none()); - assert!(field.to_i128().is_none()); - assert!(field.to_float().is_none()); - assert!(field.to_boolean().is_some()); - assert!(field.to_binary().is_none()); - assert!(field.to_decimal().is_none()); - assert!(field.to_timestamp().is_none()); - assert!(field.to_date().is_none()); - assert!(field.to_json().is_some()); - assert!(field.to_point().is_none()); - assert!(field.to_duration().is_none()); - assert!(field.to_null().is_none()); - - let field = Field::String("".to_string()); - assert!(field.to_uint().is_none()); - assert!(field.to_int().is_none()); - assert!(field.to_u128().is_none()); - assert!(field.to_i128().is_none()); - assert!(field.to_float().is_none()); - assert!(field.to_boolean().is_none()); - assert!(field.to_binary().is_none()); - assert!(field.to_decimal().is_none()); - assert!(field.to_timestamp().is_none()); - assert!(field.to_date().is_none()); - assert!(field.to_json().is_some()); - assert!(field.to_point().is_none()); - assert!(field.to_duration().is_none()); - assert!(field.to_null().is_none()); - - let field = Field::Text("".to_string()); - assert!(field.to_uint().is_none()); - assert!(field.to_int().is_none()); - assert!(field.to_u128().is_none()); - assert!(field.to_i128().is_none()); - assert!(field.to_float().is_none()); - assert!(field.to_boolean().is_none()); - assert!(field.to_binary().is_none()); - assert!(field.to_decimal().is_none()); - assert!(field.to_timestamp().is_none()); - assert!(field.to_date().is_none()); - assert!(field.to_json().is_some()); - assert!(field.to_point().is_none()); - assert!(field.to_duration().is_none()); - assert!(field.to_null().is_none()); - - let field = Field::Binary(vec![]); - assert!(field.to_uint().is_none()); - assert!(field.to_int().is_none()); - assert!(field.to_u128().is_none()); - assert!(field.to_i128().is_none()); - assert!(field.to_float().is_none()); - assert!(field.to_boolean().is_none()); - assert!(field.to_binary().is_some()); - assert!(field.to_decimal().is_none()); - assert!(field.to_timestamp().is_none()); - assert!(field.to_date().is_none()); - assert!(field.to_json().is_none()); - assert!(field.to_point().is_none()); - assert!(field.to_duration().is_none()); - assert!(field.to_null().is_none()); - - let field = Field::Decimal(Decimal::from(1)); - assert!(field.to_uint().is_some()); - assert!(field.to_int().is_some()); - assert!(field.to_u128().is_some()); - assert!(field.to_i128().is_some()); - assert!(field.to_float().is_some()); - assert!(field.to_boolean().is_some()); - assert!(field.to_binary().is_none()); - assert!(field.to_decimal().is_some()); - assert!(field.to_timestamp().is_none()); - assert!(field.to_date().is_none()); - assert!(field.to_json().is_none()); - assert!(field.to_point().is_none()); - assert!(field.to_duration().is_none()); - assert!(field.to_null().is_none()); - - let field = Field::Timestamp(DateTime::from(Utc.timestamp_millis_opt(0).unwrap())); - assert!(field.to_uint().is_none()); - assert!(field.to_int().is_none()); - assert!(field.to_u128().is_none()); - assert!(field.to_i128().is_none()); - assert!(field.to_float().is_none()); - assert!(field.to_boolean().is_none()); - assert!(field.to_binary().is_none()); - assert!(field.to_decimal().is_none()); - assert!(field.to_timestamp().is_some()); - assert!(field.to_date().is_some()); - assert!(field.to_json().is_none()); - assert!(field.to_point().is_none()); - assert!(field.to_duration().is_none()); - assert!(field.to_null().is_none()); - - let field = Field::Date(NaiveDate::from_ymd_opt(1970, 1, 1).unwrap()); - assert!(field.to_uint().is_none()); - assert!(field.to_int().is_none()); - assert!(field.to_u128().is_none()); - assert!(field.to_i128().is_none()); - assert!(field.to_float().is_none()); - assert!(field.to_boolean().is_none()); - assert!(field.to_binary().is_none()); - assert!(field.to_decimal().is_none()); - assert!(field.to_timestamp().is_some()); - assert!(field.to_date().is_some()); - assert!(field.to_json().is_none()); - assert!(field.to_point().is_none()); - assert!(field.to_duration().is_none()); - assert!(field.to_null().is_none()); - - let field = Field::Json(Vec::::new().into()); - assert!(field.to_uint().is_none()); - assert!(field.to_int().is_none()); - assert!(field.to_u128().is_none()); - assert!(field.to_i128().is_none()); - assert!(field.to_float().is_none()); - assert!(field.to_boolean().is_none()); - assert!(field.to_binary().is_none()); - assert!(field.to_decimal().is_none()); - assert!(field.to_timestamp().is_none()); - assert!(field.to_date().is_none()); - assert!(field.to_json().is_some()); - assert!(field.to_point().is_none()); - assert!(field.to_duration().is_none()); - assert!(field.to_null().is_none()); - - let field = Field::Point(DozerPoint::from((0.0, 0.0))); - assert!(field.to_uint().is_none()); - assert!(field.to_int().is_none()); - assert!(field.to_u128().is_none()); - assert!(field.to_i128().is_none()); - assert!(field.to_float().is_none()); - assert!(field.to_boolean().is_none()); - assert!(field.to_binary().is_none()); - assert!(field.to_decimal().is_none()); - assert!(field.to_timestamp().is_none()); - assert!(field.to_date().is_none()); - assert!(field.to_json().is_none()); - assert!(field.to_point().is_some()); - assert!(field.to_duration().is_none()); - assert!(field.to_null().is_none()); - - let field = Field::Duration(DozerDuration( - std::time::Duration::from_nanos(123_u64), - TimeUnit::Nanoseconds, - )); - assert!(field.to_uint().is_none()); - assert!(field.to_int().is_none()); - assert!(field.to_u128().is_none()); - assert!(field.to_i128().is_none()); - assert!(field.to_float().is_none()); - assert!(field.to_boolean().is_none()); - assert!(field.to_binary().is_none()); - assert!(field.to_decimal().is_none()); - assert!(field.to_timestamp().is_none()); - assert!(field.to_date().is_none()); - assert!(field.to_json().is_none()); - assert!(field.to_point().is_none()); - assert!(field.to_duration().is_some()); - assert!(field.to_null().is_none()); - - let field = Field::Null; - assert!(field.to_uint().is_some()); - assert!(field.to_int().is_some()); - assert!(field.to_u128().is_some()); - assert!(field.to_i128().is_some()); - assert!(field.to_float().is_some()); - assert!(field.to_boolean().is_some()); - assert!(field.to_binary().is_none()); - assert!(field.to_decimal().is_some()); - assert!(field.to_timestamp().is_none()); - assert!(field.to_date().is_none()); - assert!(field.to_json().is_some()); - assert!(field.to_point().is_none()); - assert!(field.to_duration().is_some()); - assert!(field.to_null().is_some()); -} diff --git a/dozer-utils/Cargo.toml b/dozer-utils/Cargo.toml deleted file mode 100644 index e5dc786118..0000000000 --- a/dozer-utils/Cargo.toml +++ /dev/null @@ -1,10 +0,0 @@ -[package] -name = "dozer-utils" -version = "0.4.0" -edition = "2021" -license = "AGPL-3.0-or-later" - -# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html - -[dependencies] -dozer-types = { path = "../dozer-types" } diff --git a/dozer-utils/src/cleanup.rs b/dozer-utils/src/cleanup.rs deleted file mode 100644 index 0fa8817375..0000000000 --- a/dozer-utils/src/cleanup.rs +++ /dev/null @@ -1,54 +0,0 @@ -use std::process::{Child, Command}; - -use dozer_types::log::error; - -#[must_use] -pub enum Cleanup { - RemoveDirectory(String), - RemoveFile(String), - KillProcess(Child), - DockerCompose(String), -} - -impl Drop for Cleanup { - fn drop(&mut self) { - match self { - Cleanup::RemoveDirectory(dir) => { - if let Err(e) = std::fs::remove_dir_all(&dir) { - error!("Failed to remove directory {}: {}", dir, e); - } - } - Cleanup::RemoveFile(file) => { - if let Err(e) = std::fs::remove_file(&file) { - error!("Failed to remove file {}: {}", file, e); - } - } - Cleanup::KillProcess(child) => { - if let Err(e) = child.kill() { - error!("Failed to kill process {}: {}", child.id(), e); - } - } - Cleanup::DockerCompose(docker_compose_path) => { - match Command::new("docker") - .args(["compose", "-f", docker_compose_path, "down"]) - .status() - { - Ok(status) => { - if !status.success() { - error!( - "docker compose down with input file {} failed with status {}", - docker_compose_path, status - ); - } - } - Err(e) => { - error!( - "Failed to run docker compose down with input file {}: {}", - docker_compose_path, e - ); - } - } - } - } - } -} diff --git a/dozer-utils/src/lib.rs b/dozer-utils/src/lib.rs deleted file mode 100644 index aeb5a08457..0000000000 --- a/dozer-utils/src/lib.rs +++ /dev/null @@ -1,4 +0,0 @@ -mod cleanup; -pub mod process; - -pub use cleanup::Cleanup; diff --git a/dozer-utils/src/process.rs b/dozer-utils/src/process.rs deleted file mode 100644 index 8a68642ed7..0000000000 --- a/dozer-utils/src/process.rs +++ /dev/null @@ -1,56 +0,0 @@ -use std::{path::Path, process::Command}; - -use dozer_types::log::{error, info}; - -use crate::Cleanup; - -pub fn run_command(bin: &str, args: &[&str], on_error_command: Option<(&str, &[&str])>) { - let mut cmd = Command::new(bin); - cmd.args(args); - info!("Running command: {:?}", cmd); - let output = cmd - .output() - .unwrap_or_else(|e| panic!("Failed to run command {cmd:?}: {e}")); - if !output.status.success() { - let stderr = String::from_utf8_lossy(&output.stderr); - error!("{stderr}"); - if let Some((bin, args)) = on_error_command { - let mut cmd = Command::new(bin); - cmd.args(args); - let output = cmd - .output() - .unwrap_or_else(|e| panic!("Failed to run command {cmd:?}: {e}")); - let stdout = String::from_utf8_lossy(&output.stdout); - error!("{stdout}"); - } - panic!( - "Command {:?} failed with status {}, working directory {:?}", - cmd, - output.status, - std::env::current_dir().expect("Failed to get cwd") - ); - } - info!("Command done: {:?}", cmd); -} - -pub fn run_docker_compose(docker_compose_path: &Path, service_name: &str) -> Cleanup { - let docker_compose_path = docker_compose_path - .to_str() - .unwrap_or_else(|| panic!("Non-UFT8 path {docker_compose_path:?}")); - run_command( - "docker", - &[ - "compose", - "-f", - docker_compose_path, - "run", - "--build", - service_name, - ], - Some(( - "docker", - &["compose", "-f", docker_compose_path, "logs", "--no-color"], - )), - ); - Cleanup::DockerCompose(docker_compose_path.to_string()) -} diff --git a/images/dozer_live_screen1.png b/images/dozer_live_screen1.png deleted file mode 100644 index 763a40e3c2..0000000000 Binary files a/images/dozer_live_screen1.png and /dev/null differ diff --git a/images/dozer_live_screen2.png b/images/dozer_live_screen2.png deleted file mode 100644 index fbd7d0b672..0000000000 Binary files a/images/dozer_live_screen2.png and /dev/null differ diff --git a/images/supported_sources.png b/images/supported_sources.png deleted file mode 100644 index e1654f50aa..0000000000 Binary files a/images/supported_sources.png and /dev/null differ diff --git a/images/tools.png b/images/tools.png deleted file mode 100644 index 6f0d1f897c..0000000000 Binary files a/images/tools.png and /dev/null differ diff --git a/json_schemas/connections.json b/json_schemas/connections.json deleted file mode 100644 index c929085d1a..0000000000 --- a/json_schemas/connections.json +++ /dev/null @@ -1,873 +0,0 @@ -[ - { - "name": "postgres", - "schema": { - "$schema": "http://json-schema.org/draft-07/schema#", - "title": "PostgresConfig", - "description": "Configuration for a Postgres connection", - "examples": [ - { - "database": "postgres", - "host": "localhost", - "password": "postgres", - "port": 5432, - "schema": "public", - "user": "postgres" - } - ], - "type": "object", - "properties": { - "batch_size": { - "description": "The snapshot batch size", - "type": [ - "integer", - "null" - ], - "format": "uint32", - "minimum": 0.0 - }, - "connection_url": { - "description": "The connection url to use", - "type": [ - "string", - "null" - ] - }, - "database": { - "description": "The database to connect to (default: postgres)", - "type": [ - "string", - "null" - ] - }, - "host": { - "description": "The host to connect to (IP or DNS name)", - "type": [ - "string", - "null" - ] - }, - "password": { - "description": "The password to use for authentication", - "type": [ - "string", - "null" - ] - }, - "port": { - "description": "The port to connect to (default: 5432)", - "type": [ - "integer", - "null" - ], - "format": "uint32", - "minimum": 0.0 - }, - "schema": { - "description": "The schema of the tables", - "type": [ - "string", - "null" - ] - }, - "sslmode": { - "description": "The sslmode to use for the connection (disable, prefer, require)", - "type": [ - "string", - "null" - ] - }, - "user": { - "description": "The username to use for authentication", - "type": [ - "string", - "null" - ] - } - }, - "additionalProperties": false - } - }, - { - "name": "ethereum", - "schema": { - "$schema": "http://json-schema.org/draft-07/schema#", - "title": "EthConfig", - "examples": [ - { - "provider": { - "Log": { - "filter": { - "from_block": 0, - "to_block": null - }, - "wss_url": "" - } - } - } - ], - "type": "object", - "required": [ - "provider" - ], - "properties": { - "provider": { - "$ref": "#/definitions/EthProviderConfig" - } - }, - "definitions": { - "EthContract": { - "type": "object", - "required": [ - "abi", - "address", - "name" - ], - "properties": { - "abi": { - "type": "string" - }, - "address": { - "type": "string" - }, - "name": { - "type": "string" - } - } - }, - "EthFilter": { - "type": "object", - "properties": { - "addresses": { - "type": "array", - "items": { - "type": "string" - } - }, - "from_block": { - "type": [ - "integer", - "null" - ], - "format": "uint64", - "minimum": 0.0 - }, - "to_block": { - "type": [ - "integer", - "null" - ], - "format": "uint64", - "minimum": 0.0 - }, - "topics": { - "type": "array", - "items": { - "type": "string" - } - } - } - }, - "EthLogConfig": { - "type": "object", - "required": [ - "wss_url" - ], - "properties": { - "contracts": { - "type": "array", - "items": { - "$ref": "#/definitions/EthContract" - } - }, - "filter": { - "anyOf": [ - { - "$ref": "#/definitions/EthFilter" - }, - { - "type": "null" - } - ] - }, - "wss_url": { - "type": "string" - } - } - }, - "EthProviderConfig": { - "oneOf": [ - { - "type": "object", - "required": [ - "Log" - ], - "properties": { - "Log": { - "$ref": "#/definitions/EthLogConfig" - } - }, - "additionalProperties": false - }, - { - "type": "object", - "required": [ - "Trace" - ], - "properties": { - "Trace": { - "$ref": "#/definitions/EthTraceConfig" - } - }, - "additionalProperties": false - } - ] - }, - "EthTraceConfig": { - "type": "object", - "required": [ - "from_block", - "https_url" - ], - "properties": { - "batch_size": { - "type": [ - "integer", - "null" - ], - "format": "uint64", - "minimum": 0.0 - }, - "from_block": { - "type": "integer", - "format": "uint64", - "minimum": 0.0 - }, - "https_url": { - "type": "string" - }, - "to_block": { - "type": [ - "integer", - "null" - ], - "format": "uint64", - "minimum": 0.0 - } - } - } - } - } - }, - { - "name": "grpc", - "schema": { - "$schema": "http://json-schema.org/draft-07/schema#", - "title": "GrpcConfig", - "examples": [ - { - "adapter": "arrow", - "host": "localhost", - "port": 50051, - "schemas": { - "Path": "schema.json" - } - } - ], - "type": "object", - "required": [ - "schemas" - ], - "properties": { - "adapter": { - "type": [ - "string", - "null" - ] - }, - "host": { - "type": [ - "string", - "null" - ] - }, - "port": { - "type": [ - "integer", - "null" - ], - "format": "uint32", - "minimum": 0.0 - }, - "schemas": { - "$ref": "#/definitions/ConfigSchemas" - } - }, - "definitions": { - "ConfigSchemas": { - "oneOf": [ - { - "type": "object", - "required": [ - "Inline" - ], - "properties": { - "Inline": { - "type": "string" - } - }, - "additionalProperties": false - }, - { - "type": "object", - "required": [ - "Path" - ], - "properties": { - "Path": { - "type": "string" - } - }, - "additionalProperties": false - } - ] - } - } - } - }, - { - "name": "snowflake", - "schema": { - "$schema": "http://json-schema.org/draft-07/schema#", - "title": "SnowflakeConfig", - "examples": [ - { - "database": "database", - "driver": "SnowflakeDSIIDriver", - "password": "password", - "port": "443", - "role": "role", - "schema": "schema", - "server": "..snowflakecomputing.com", - "user": "bob", - "warehouse": "warehouse" - } - ], - "type": "object", - "required": [ - "database", - "password", - "port", - "role", - "schema", - "server", - "user", - "warehouse" - ], - "properties": { - "database": { - "type": "string" - }, - "driver": { - "type": [ - "string", - "null" - ] - }, - "password": { - "type": "string" - }, - "poll_interval_seconds": { - "type": "number", - "format": "double" - }, - "port": { - "type": "string" - }, - "role": { - "type": "string" - }, - "schema": { - "type": "string" - }, - "server": { - "type": "string" - }, - "user": { - "type": "string" - }, - "warehouse": { - "type": "string" - } - } - } - }, - { - "name": "kafka", - "schema": { - "$schema": "http://json-schema.org/draft-07/schema#", - "title": "KafkaConfig", - "examples": [ - { - "broker": "", - "schema_registry_url": "" - } - ], - "type": "object", - "required": [ - "broker" - ], - "properties": { - "broker": { - "type": "string" - }, - "schema_registry_url": { - "type": [ - "string", - "null" - ] - } - } - } - }, - { - "name": "s3", - "schema": { - "$schema": "http://json-schema.org/draft-07/schema#", - "title": "S3Storage", - "examples": [ - { - "details": { - "access_key_id": "", - "bucket_name": "", - "region": "", - "secret_access_key": "" - }, - "tables": [ - { - "config": { - "CSV": { - "extension": ".csv", - "path": "path/to/file" - } - }, - "name": "table_name" - } - ] - } - ], - "type": "object", - "required": [ - "details", - "tables" - ], - "properties": { - "details": { - "$ref": "#/definitions/S3Details" - }, - "tables": { - "type": "array", - "items": { - "$ref": "#/definitions/Table" - } - } - }, - "definitions": { - "CsvConfig": { - "type": "object", - "required": [ - "extension", - "path" - ], - "properties": { - "extension": { - "type": "string" - }, - "marker_extension": { - "type": [ - "string", - "null" - ] - }, - "path": { - "type": "string" - } - } - }, - "ParquetConfig": { - "type": "object", - "required": [ - "extension", - "path" - ], - "properties": { - "extension": { - "type": "string" - }, - "marker_extension": { - "type": [ - "string", - "null" - ] - }, - "path": { - "type": "string" - } - } - }, - "S3Details": { - "type": "object", - "required": [ - "access_key_id", - "bucket_name", - "region", - "secret_access_key" - ], - "properties": { - "access_key_id": { - "type": "string" - }, - "bucket_name": { - "type": "string" - }, - "region": { - "type": "string" - }, - "secret_access_key": { - "type": "string" - } - } - }, - "Table": { - "type": "object", - "required": [ - "config", - "name" - ], - "properties": { - "config": { - "$ref": "#/definitions/TableConfig" - }, - "name": { - "type": "string" - } - } - }, - "TableConfig": { - "oneOf": [ - { - "type": "object", - "required": [ - "CSV" - ], - "properties": { - "CSV": { - "$ref": "#/definitions/CsvConfig" - } - }, - "additionalProperties": false - }, - { - "type": "object", - "required": [ - "Parquet" - ], - "properties": { - "Parquet": { - "$ref": "#/definitions/ParquetConfig" - } - }, - "additionalProperties": false - } - ] - } - } - } - }, - { - "name": "local_storage", - "schema": { - "$schema": "http://json-schema.org/draft-07/schema#", - "title": "LocalStorage", - "examples": [ - { - "details": { - "path": "path" - }, - "tables": [ - { - "config": { - "CSV": { - "extension": ".csv", - "path": "path/to/table" - } - }, - "name": "table_name" - } - ] - } - ], - "type": "object", - "required": [ - "details", - "tables" - ], - "properties": { - "details": { - "$ref": "#/definitions/LocalDetails" - }, - "tables": { - "type": "array", - "items": { - "$ref": "#/definitions/Table" - } - } - }, - "definitions": { - "CsvConfig": { - "type": "object", - "required": [ - "extension", - "path" - ], - "properties": { - "extension": { - "type": "string" - }, - "marker_extension": { - "type": [ - "string", - "null" - ] - }, - "path": { - "type": "string" - } - } - }, - "LocalDetails": { - "type": "object", - "required": [ - "path" - ], - "properties": { - "path": { - "type": "string" - } - } - }, - "ParquetConfig": { - "type": "object", - "required": [ - "extension", - "path" - ], - "properties": { - "extension": { - "type": "string" - }, - "marker_extension": { - "type": [ - "string", - "null" - ] - }, - "path": { - "type": "string" - } - } - }, - "Table": { - "type": "object", - "required": [ - "config", - "name" - ], - "properties": { - "config": { - "$ref": "#/definitions/TableConfig" - }, - "name": { - "type": "string" - } - } - }, - "TableConfig": { - "oneOf": [ - { - "type": "object", - "required": [ - "CSV" - ], - "properties": { - "CSV": { - "$ref": "#/definitions/CsvConfig" - } - }, - "additionalProperties": false - }, - { - "type": "object", - "required": [ - "Parquet" - ], - "properties": { - "Parquet": { - "$ref": "#/definitions/ParquetConfig" - } - }, - "additionalProperties": false - } - ] - } - } - } - }, - { - "name": "deltalake", - "schema": { - "$schema": "http://json-schema.org/draft-07/schema#", - "title": "DeltaLakeConfig", - "examples": [ - { - "tables": [ - { - "name": "", - "path": "" - } - ] - } - ], - "type": "object", - "required": [ - "tables" - ], - "properties": { - "tables": { - "type": "array", - "items": { - "$ref": "#/definitions/DeltaTable" - } - } - }, - "definitions": { - "DeltaTable": { - "type": "object", - "required": [ - "name", - "path" - ], - "properties": { - "name": { - "type": "string" - }, - "path": { - "type": "string" - } - } - } - } - } - }, - { - "name": "mongodb", - "schema": { - "$schema": "http://json-schema.org/draft-07/schema#", - "title": "MongodbConfig", - "examples": [ - { - "connection_string": "mongodb://localhost:27017/db_name" - } - ], - "type": "object", - "required": [ - "connection_string" - ], - "properties": { - "connection_string": { - "type": "string" - } - } - } - }, - { - "name": "mysql", - "schema": { - "$schema": "http://json-schema.org/draft-07/schema#", - "title": "MySQLConfig", - "examples": [ - { - "server_id": 1, - "url": "mysql://root:1234@localhost:3306/db_name" - } - ], - "type": "object", - "required": [ - "url" - ], - "properties": { - "server_id": { - "type": [ - "integer", - "null" - ], - "format": "uint32", - "minimum": 0.0 - }, - "url": { - "type": "string" - } - } - } - }, - { - "name": "dozer", - "schema": { - "$schema": "http://json-schema.org/draft-07/schema#", - "title": "NestedDozerConfig", - "type": "object", - "required": [ - "url" - ], - "properties": { - "log_options": { - "$ref": "#/definitions/NestedDozerLogOptions" - }, - "url": { - "type": "string" - } - }, - "definitions": { - "NestedDozerLogOptions": { - "type": "object", - "properties": { - "batch_size": { - "type": [ - "integer", - "null" - ], - "format": "uint32", - "minimum": 0.0 - }, - "buffer_size": { - "type": [ - "integer", - "null" - ], - "format": "uint32", - "minimum": 0.0 - }, - "timeout_in_millis": { - "type": [ - "integer", - "null" - ], - "format": "uint32", - "minimum": 0.0 - } - } - } - } - } - } -] \ No newline at end of file diff --git a/json_schemas/dozer.json b/json_schemas/dozer.json deleted file mode 100644 index 3bac6f286e..0000000000 --- a/json_schemas/dozer.json +++ /dev/null @@ -1,2228 +0,0 @@ -{ - "$schema": "http://json-schema.org/draft-07/schema#", - "title": "Config", - "description": "The configuration for the app", - "type": "object", - "required": [ - "app_name", - "version" - ], - "properties": { - "api": { - "description": "Api server config related: port, host, etc", - "allOf": [ - { - "$ref": "#/definitions/ApiConfig" - } - ] - }, - "app": { - "description": "App runtime config: behaviour of pipeline and log", - "allOf": [ - { - "$ref": "#/definitions/AppConfig" - } - ] - }, - "app_name": { - "description": "name of the app", - "type": "string" - }, - "company_id": { - "description": "Unique application Id", - "default": "", - "type": "string" - }, - "connections": { - "description": "connections to databases: Eg: Postgres, Snowflake, etc", - "type": "array", - "items": { - "$ref": "#/definitions/Connection" - } - }, - "flags": { - "description": "flags to enable/disable features", - "allOf": [ - { - "$ref": "#/definitions/Flags" - } - ] - }, - "home_dir": { - "description": "directory for all process; Default: ./.dozer", - "type": [ - "string", - "null" - ] - }, - "id": { - "description": "Unique application Id", - "default": "", - "type": "string" - }, - "lambdas": { - "description": "Lambda functions.", - "type": "array", - "items": { - "$ref": "#/definitions/LambdaConfig" - } - }, - "sinks": { - "description": "sinks to output data to", - "type": "array", - "items": { - "$ref": "#/definitions/Sink" - } - }, - "sources": { - "description": "sources to ingest data related to particular connection", - "type": "array", - "items": { - "$ref": "#/definitions/Source" - } - }, - "sql": { - "description": "transformations to apply to source data in SQL format as multiple queries", - "type": [ - "string", - "null" - ] - }, - "telemetry": { - "description": "Instrument using Dozer", - "allOf": [ - { - "$ref": "#/definitions/TelemetryConfig" - } - ] - }, - "udfs": { - "description": "UDF specific configuration (eg. !Onnx)", - "type": "array", - "items": { - "$ref": "#/definitions/UdfConfig" - } - }, - "version": { - "type": "integer", - "format": "uint32", - "minimum": 0.0 - } - }, - "additionalProperties": false, - "definitions": { - "AerospikeConnection": { - "type": "object", - "required": [ - "hosts", - "namespace", - "sets" - ], - "properties": { - "batching": { - "default": false, - "type": "boolean" - }, - "hosts": { - "type": "string" - }, - "namespace": { - "type": "string" - }, - "replication": { - "default": { - "datacenter": "esp", - "server_address": "0.0.0.0", - "server_port": 5929 - }, - "allOf": [ - { - "$ref": "#/definitions/ReplicationSettings" - } - ] - }, - "schemas": { - "default": null, - "anyOf": [ - { - "$ref": "#/definitions/ConfigSchemas" - }, - { - "type": "null" - } - ] - }, - "sets": { - "type": "array", - "items": { - "type": "string" - } - } - } - }, - "AerospikeDenormalizations": { - "type": "object", - "required": [ - "columns", - "from_namespace", - "from_set", - "key" - ], - "properties": { - "columns": { - "type": "array", - "items": { - "$ref": "#/definitions/DenormColumn" - } - }, - "from_namespace": { - "type": "string" - }, - "from_set": { - "type": "string" - }, - "key": { - "$ref": "#/definitions/DenormKey" - } - }, - "additionalProperties": false - }, - "AerospikeSet": { - "type": "object", - "required": [ - "namespace", - "primary_key", - "set" - ], - "properties": { - "namespace": { - "type": "string" - }, - "primary_key": { - "type": "array", - "items": { - "type": "string" - } - }, - "set": { - "type": "string" - } - } - }, - "AerospikeSinkConfig": { - "type": "object", - "required": [ - "connection", - "metadata_namespace" - ], - "properties": { - "connection": { - "type": "string" - }, - "max_batch_duration_ms": { - "type": [ - "integer", - "null" - ], - "format": "uint64", - "minimum": 0.0 - }, - "metadata_namespace": { - "type": "string" - }, - "metadata_set": { - "default": null, - "type": [ - "string", - "null" - ] - }, - "n_threads": { - "type": [ - "integer", - "null" - ], - "format": "uint", - "minimum": 1.0 - }, - "preferred_batch_size": { - "type": [ - "integer", - "null" - ], - "format": "uint64", - "minimum": 0.0 - }, - "tables": { - "type": "array", - "items": { - "$ref": "#/definitions/AerospikeSinkTable" - } - } - }, - "additionalProperties": false - }, - "AerospikeSinkTable": { - "type": "object", - "required": [ - "namespace", - "set_name", - "source_table_name" - ], - "properties": { - "aggregate_by_pk": { - "default": false, - "type": "boolean" - }, - "denormalize": { - "type": "array", - "items": { - "$ref": "#/definitions/AerospikeDenormalizations" - } - }, - "namespace": { - "type": "string" - }, - "primary_key": { - "default": [], - "type": "array", - "items": { - "type": "string" - } - }, - "set_name": { - "type": "string" - }, - "source_table_name": { - "type": "string" - }, - "write_denormalized_to": { - "anyOf": [ - { - "$ref": "#/definitions/AerospikeSet" - }, - { - "type": "null" - } - ] - } - }, - "additionalProperties": false - }, - "ApiConfig": { - "type": "object", - "properties": { - "api_security": { - "description": "The security configuration for the API; Default: None", - "anyOf": [ - { - "$ref": "#/definitions/ApiSecurity" - }, - { - "type": "null" - } - ] - }, - "app_grpc": { - "$ref": "#/definitions/AppGrpcOptions" - }, - "default_max_num_records": { - "type": [ - "integer", - "null" - ], - "format": "uint", - "minimum": 0.0 - }, - "grpc": { - "$ref": "#/definitions/GrpcApiOptions" - }, - "pgwire": { - "$ref": "#/definitions/PgWireOptions" - }, - "rest": { - "$ref": "#/definitions/RestApiOptions" - } - }, - "additionalProperties": false - }, - "ApiSecurity": { - "description": "The security model option for the API", - "oneOf": [ - { - "description": "Initialize with a JWT_SECRET", - "type": "object", - "required": [ - "Jwt" - ], - "properties": { - "Jwt": { - "type": "string" - } - }, - "additionalProperties": false - } - ] - }, - "AppConfig": { - "type": "object", - "properties": { - "app_buffer_size": { - "description": "Pipeline buffer size", - "type": [ - "integer", - "null" - ], - "format": "uint32", - "minimum": 0.0 - }, - "error_threshold": { - "description": "How many errors we can tolerate before bringing down the app.", - "type": [ - "integer", - "null" - ], - "format": "uint32", - "minimum": 0.0 - }, - "event_hub_capacity": { - "description": "The event hub's queue capacity. Events that are not processed will be dropped.", - "type": [ - "integer", - "null" - ], - "format": "uint", - "minimum": 0.0 - } - }, - "additionalProperties": false - }, - "AppGrpcOptions": { - "type": "object", - "properties": { - "host": { - "type": [ - "string", - "null" - ] - }, - "port": { - "type": [ - "integer", - "null" - ], - "format": "uint32", - "minimum": 0.0 - } - }, - "additionalProperties": false - }, - "ClickhouseSinkConfig": { - "type": "object", - "required": [ - "options", - "sink_table_name", - "source_table_name" - ], - "properties": { - "create_table_options": { - "anyOf": [ - { - "$ref": "#/definitions/ClickhouseTableOptions" - }, - { - "type": "null" - } - ] - }, - "database": { - "default": "default", - "type": "string" - }, - "host": { - "default": "0.0.0.0", - "type": "string" - }, - "options": { - "type": "array", - "items": { - "type": "array", - "items": [ - { - "type": "string" - }, - { - "type": "string" - } - ], - "maxItems": 2, - "minItems": 2 - } - }, - "password": { - "default": null, - "type": [ - "string", - "null" - ] - }, - "port": { - "default": 9000, - "type": "integer", - "format": "uint16", - "minimum": 0.0 - }, - "scheme": { - "default": "tcp", - "type": "string" - }, - "sink_table_name": { - "type": "string" - }, - "source_table_name": { - "type": "string" - }, - "user": { - "default": "default", - "type": "string" - } - }, - "additionalProperties": false - }, - "ClickhouseTableOptions": { - "type": "object", - "properties": { - "cluster": { - "type": [ - "string", - "null" - ] - }, - "engine": { - "type": [ - "string", - "null" - ] - }, - "order_by": { - "type": [ - "array", - "null" - ], - "items": { - "type": "string" - } - }, - "partition_by": { - "type": [ - "string", - "null" - ] - }, - "primary_keys": { - "type": [ - "array", - "null" - ], - "items": { - "type": "string" - } - }, - "sample_by": { - "type": [ - "string", - "null" - ] - } - }, - "additionalProperties": false - }, - "ConfigSchemas": { - "oneOf": [ - { - "type": "object", - "required": [ - "Inline" - ], - "properties": { - "Inline": { - "type": "string" - } - }, - "additionalProperties": false - }, - { - "type": "object", - "required": [ - "Path" - ], - "properties": { - "Path": { - "type": "string" - } - }, - "additionalProperties": false - } - ] - }, - "Connection": { - "type": "object", - "required": [ - "config", - "name" - ], - "properties": { - "config": { - "$ref": "#/definitions/ConnectionConfig" - }, - "name": { - "type": "string" - } - }, - "additionalProperties": false - }, - "ConnectionConfig": { - "oneOf": [ - { - "description": "In yaml, present as tag: `!Postgres`", - "type": "object", - "required": [ - "Postgres" - ], - "properties": { - "Postgres": { - "$ref": "#/definitions/PostgresConfig" - } - }, - "additionalProperties": false - }, - { - "description": "In yaml, present as tag: `!Ethereum`", - "type": "object", - "required": [ - "Ethereum" - ], - "properties": { - "Ethereum": { - "$ref": "#/definitions/EthConfig" - } - }, - "additionalProperties": false - }, - { - "description": "In yaml, present as tag: `!Grpc`", - "type": "object", - "required": [ - "Grpc" - ], - "properties": { - "Grpc": { - "$ref": "#/definitions/GrpcConfig" - } - }, - "additionalProperties": false - }, - { - "description": "In yaml, present as tag: `!Snowflake`", - "type": "object", - "required": [ - "Snowflake" - ], - "properties": { - "Snowflake": { - "$ref": "#/definitions/SnowflakeConfig" - } - }, - "additionalProperties": false - }, - { - "description": "In yaml, present as tag: `!Kafka`", - "type": "object", - "required": [ - "Kafka" - ], - "properties": { - "Kafka": { - "$ref": "#/definitions/KafkaConfig" - } - }, - "additionalProperties": false - }, - { - "description": "In yaml, present as tag: `!ObjectStore`", - "type": "object", - "required": [ - "S3Storage" - ], - "properties": { - "S3Storage": { - "$ref": "#/definitions/S3Storage" - } - }, - "additionalProperties": false - }, - { - "description": "In yaml, present as tag: `!ObjectStore`", - "type": "object", - "required": [ - "LocalStorage" - ], - "properties": { - "LocalStorage": { - "$ref": "#/definitions/LocalStorage" - } - }, - "additionalProperties": false - }, - { - "description": "In yaml, present as tag\" `!DeltaLake`", - "type": "object", - "required": [ - "DeltaLake" - ], - "properties": { - "DeltaLake": { - "$ref": "#/definitions/DeltaLakeConfig" - } - }, - "additionalProperties": false - }, - { - "description": "In yaml, present as tag: `!MongoDB`", - "type": "object", - "required": [ - "MongoDB" - ], - "properties": { - "MongoDB": { - "$ref": "#/definitions/MongodbConfig" - } - }, - "additionalProperties": false - }, - { - "description": "In yaml, present as tag\" `!MySQL`", - "type": "object", - "required": [ - "MySQL" - ], - "properties": { - "MySQL": { - "$ref": "#/definitions/MySQLConfig" - } - }, - "additionalProperties": false - }, - { - "description": "In yaml, present as tag\" `!JavaScript`", - "type": "object", - "required": [ - "JavaScript" - ], - "properties": { - "JavaScript": { - "$ref": "#/definitions/JavaScriptConfig" - } - }, - "additionalProperties": false - }, - { - "description": "In yaml, present as tag\" `!Webhook`", - "type": "object", - "required": [ - "Webhook" - ], - "properties": { - "Webhook": { - "$ref": "#/definitions/WebhookConfig" - } - }, - "additionalProperties": false - }, - { - "type": "object", - "required": [ - "Oracle" - ], - "properties": { - "Oracle": { - "$ref": "#/definitions/OracleConfig" - } - }, - "additionalProperties": false - }, - { - "type": "object", - "required": [ - "Aerospike" - ], - "properties": { - "Aerospike": { - "$ref": "#/definitions/AerospikeConnection" - } - }, - "additionalProperties": false - } - ] - }, - "CsvConfig": { - "type": "object", - "required": [ - "extension", - "path" - ], - "properties": { - "extension": { - "type": "string" - }, - "marker_extension": { - "type": [ - "string", - "null" - ] - }, - "path": { - "type": "string" - } - } - }, - "DeltaLakeConfig": { - "examples": [ - { - "tables": [ - { - "name": "", - "path": "" - } - ] - } - ], - "type": "object", - "required": [ - "tables" - ], - "properties": { - "tables": { - "type": "array", - "items": { - "$ref": "#/definitions/DeltaTable" - } - } - } - }, - "DeltaTable": { - "type": "object", - "required": [ - "name", - "path" - ], - "properties": { - "name": { - "type": "string" - }, - "path": { - "type": "string" - } - } - }, - "DenormColumn": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "object", - "required": [ - "destination", - "source" - ], - "properties": { - "destination": { - "type": "string" - }, - "source": { - "type": "string" - } - } - } - ] - }, - "DenormKey": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "array", - "items": { - "type": "string" - } - } - ] - }, - "DummySinkConfig": { - "type": "object", - "required": [ - "table_name" - ], - "properties": { - "table_name": { - "type": "string" - } - }, - "additionalProperties": false - }, - "EnableProbabilisticOptimizations": { - "type": "object", - "properties": { - "in_aggregations": { - "description": "enable probabilistic optimizations in aggregations (SUM, COUNT, MIN, etc.); Default: false", - "type": [ - "boolean", - "null" - ] - }, - "in_joins": { - "description": "enable probabilistic optimizations in JOIN operations; Default: false", - "type": [ - "boolean", - "null" - ] - }, - "in_sets": { - "description": "enable probabilistic optimizations in set operations (UNION, EXCEPT, INTERSECT); Default: false", - "type": [ - "boolean", - "null" - ] - } - }, - "additionalProperties": false - }, - "EthConfig": { - "examples": [ - { - "provider": { - "Log": { - "filter": { - "from_block": 0, - "to_block": null - }, - "wss_url": "" - } - } - } - ], - "type": "object", - "required": [ - "provider" - ], - "properties": { - "provider": { - "$ref": "#/definitions/EthProviderConfig" - } - } - }, - "EthContract": { - "type": "object", - "required": [ - "abi", - "address", - "name" - ], - "properties": { - "abi": { - "type": "string" - }, - "address": { - "type": "string" - }, - "name": { - "type": "string" - } - } - }, - "EthFilter": { - "type": "object", - "properties": { - "addresses": { - "type": "array", - "items": { - "type": "string" - } - }, - "from_block": { - "type": [ - "integer", - "null" - ], - "format": "uint64", - "minimum": 0.0 - }, - "to_block": { - "type": [ - "integer", - "null" - ], - "format": "uint64", - "minimum": 0.0 - }, - "topics": { - "type": "array", - "items": { - "type": "string" - } - } - } - }, - "EthLogConfig": { - "type": "object", - "required": [ - "wss_url" - ], - "properties": { - "contracts": { - "type": "array", - "items": { - "$ref": "#/definitions/EthContract" - } - }, - "filter": { - "anyOf": [ - { - "$ref": "#/definitions/EthFilter" - }, - { - "type": "null" - } - ] - }, - "wss_url": { - "type": "string" - } - } - }, - "EthProviderConfig": { - "oneOf": [ - { - "type": "object", - "required": [ - "Log" - ], - "properties": { - "Log": { - "$ref": "#/definitions/EthLogConfig" - } - }, - "additionalProperties": false - }, - { - "type": "object", - "required": [ - "Trace" - ], - "properties": { - "Trace": { - "$ref": "#/definitions/EthTraceConfig" - } - }, - "additionalProperties": false - } - ] - }, - "EthTraceConfig": { - "type": "object", - "required": [ - "from_block", - "https_url" - ], - "properties": { - "batch_size": { - "type": [ - "integer", - "null" - ], - "format": "uint64", - "minimum": 0.0 - }, - "from_block": { - "type": "integer", - "format": "uint64", - "minimum": 0.0 - }, - "https_url": { - "type": "string" - }, - "to_block": { - "type": [ - "integer", - "null" - ], - "format": "uint64", - "minimum": 0.0 - } - } - }, - "Flags": { - "type": "object", - "properties": { - "authenticate_server_reflection": { - "description": "require authentication to access grpc server reflection service if true.; Default: false", - "type": [ - "boolean", - "null" - ] - }, - "dynamic": { - "description": "dynamic grpc enabled; Default: true", - "type": [ - "boolean", - "null" - ] - }, - "enable_app_checkpoints": { - "description": "app checkpoints can be used to resume execution of a query.; Default: false", - "type": [ - "boolean", - "null" - ] - }, - "enable_probabilistic_optimizations": { - "description": "probablistic optimizations reduce memory consumption at the expense of accuracy.", - "allOf": [ - { - "$ref": "#/definitions/EnableProbabilisticOptimizations" - } - ] - }, - "grpc_web": { - "description": "http1 + web support for grpc. This is required for browser clients.; Default: true", - "type": [ - "boolean", - "null" - ] - }, - "push_events": { - "description": "push events enabled.; Default: true", - "type": [ - "boolean", - "null" - ] - } - }, - "additionalProperties": false - }, - "GrpcApiOptions": { - "type": "object", - "properties": { - "cors": { - "type": [ - "boolean", - "null" - ] - }, - "enabled": { - "type": [ - "boolean", - "null" - ] - }, - "host": { - "type": [ - "string", - "null" - ] - }, - "port": { - "type": [ - "integer", - "null" - ], - "format": "uint16", - "minimum": 0.0 - }, - "web": { - "type": [ - "boolean", - "null" - ] - } - }, - "additionalProperties": false - }, - "GrpcConfig": { - "examples": [ - { - "adapter": "arrow", - "host": "localhost", - "port": 50051, - "schemas": { - "Path": "schema.json" - } - } - ], - "type": "object", - "required": [ - "schemas" - ], - "properties": { - "adapter": { - "type": [ - "string", - "null" - ] - }, - "host": { - "type": [ - "string", - "null" - ] - }, - "port": { - "type": [ - "integer", - "null" - ], - "format": "uint32", - "minimum": 0.0 - }, - "schemas": { - "$ref": "#/definitions/ConfigSchemas" - } - } - }, - "JavaScriptConfig": { - "type": "object", - "properties": { - "bootstrap_path": { - "type": [ - "string", - "null" - ] - } - } - }, - "JavaScriptConfig2": { - "type": "object", - "required": [ - "module" - ], - "properties": { - "module": { - "description": "path to the module file", - "type": "string" - } - }, - "additionalProperties": false - }, - "JavaScriptLambda": { - "type": "object", - "required": [ - "endpoint", - "module" - ], - "properties": { - "endpoint": { - "type": "string" - }, - "module": { - "type": "string" - } - }, - "additionalProperties": false - }, - "KafkaConfig": { - "examples": [ - { - "broker": "", - "schema_registry_url": "" - } - ], - "type": "object", - "required": [ - "broker" - ], - "properties": { - "broker": { - "type": "string" - }, - "schema_registry_url": { - "type": [ - "string", - "null" - ] - } - } - }, - "LambdaConfig": { - "oneOf": [ - { - "type": "object", - "required": [ - "JavaScript" - ], - "properties": { - "JavaScript": { - "$ref": "#/definitions/JavaScriptLambda" - } - }, - "additionalProperties": false - } - ] - }, - "LocalDetails": { - "type": "object", - "required": [ - "path" - ], - "properties": { - "path": { - "type": "string" - } - } - }, - "LocalStorage": { - "examples": [ - { - "details": { - "path": "path" - }, - "tables": [ - { - "config": { - "CSV": { - "extension": ".csv", - "path": "path/to/table" - } - }, - "name": "table_name" - } - ] - } - ], - "type": "object", - "required": [ - "details", - "tables" - ], - "properties": { - "details": { - "$ref": "#/definitions/LocalDetails" - }, - "tables": { - "type": "array", - "items": { - "$ref": "#/definitions/Table" - } - } - } - }, - "MongodbConfig": { - "examples": [ - { - "connection_string": "mongodb://localhost:27017/db_name" - } - ], - "type": "object", - "required": [ - "connection_string" - ], - "properties": { - "connection_string": { - "type": "string" - } - } - }, - "MySQLConfig": { - "examples": [ - { - "server_id": 1, - "url": "mysql://root:1234@localhost:3306/db_name" - } - ], - "type": "object", - "required": [ - "url" - ], - "properties": { - "server_id": { - "type": [ - "integer", - "null" - ], - "format": "uint32", - "minimum": 0.0 - }, - "url": { - "type": "string" - } - } - }, - "OnnxConfig": { - "type": "object", - "required": [ - "path" - ], - "properties": { - "path": { - "description": "path to the model file", - "type": "string" - } - }, - "additionalProperties": false - }, - "OracleConfig": { - "type": "object", - "required": [ - "host", - "password", - "port", - "replicator", - "sid", - "user" - ], - "properties": { - "batch_size": { - "description": "Batch size during snapshotting", - "type": [ - "integer", - "null" - ], - "format": "uint", - "minimum": 0.0 - }, - "host": { - "type": "string" - }, - "password": { - "type": "string" - }, - "pdb": { - "description": "Only needed if using pluggable database", - "type": [ - "string", - "null" - ] - }, - "port": { - "type": "integer", - "format": "uint16", - "minimum": 0.0 - }, - "replicator": { - "$ref": "#/definitions/OracleReplicator" - }, - "schemas": { - "description": "The schemas to consider when listing tables. If empty, will list all schemas, which can be slow.", - "type": "array", - "items": { - "type": "string" - } - }, - "sid": { - "type": "string" - }, - "user": { - "type": "string" - } - } - }, - "OracleReplicator": { - "oneOf": [ - { - "type": "string", - "enum": [ - "DozerLogReader" - ] - }, - { - "type": "object", - "required": [ - "LogMiner" - ], - "properties": { - "LogMiner": { - "type": "object", - "required": [ - "poll_interval_in_milliseconds" - ], - "properties": { - "poll_interval_in_milliseconds": { - "type": "integer", - "format": "uint64", - "minimum": 0.0 - } - } - } - }, - "additionalProperties": false - } - ] - }, - "OracleSinkConfig": { - "type": "object", - "required": [ - "connection", - "table_name" - ], - "properties": { - "connection": { - "type": "string" - }, - "owner": { - "default": null, - "type": [ - "string", - "null" - ] - }, - "table_name": { - "type": "string" - }, - "unique_key": { - "default": [], - "type": "array", - "items": { - "type": "string" - } - } - }, - "additionalProperties": false - }, - "ParquetConfig": { - "type": "object", - "required": [ - "extension", - "path" - ], - "properties": { - "extension": { - "type": "string" - }, - "marker_extension": { - "type": [ - "string", - "null" - ] - }, - "path": { - "type": "string" - } - } - }, - "PgWireOptions": { - "type": "object", - "properties": { - "enabled": { - "type": [ - "boolean", - "null" - ] - }, - "host": { - "type": [ - "string", - "null" - ] - }, - "port": { - "type": [ - "integer", - "null" - ], - "format": "uint32", - "minimum": 0.0 - } - }, - "additionalProperties": false - }, - "PostgresConfig": { - "description": "Configuration for a Postgres connection", - "examples": [ - { - "database": "postgres", - "host": "localhost", - "password": "postgres", - "port": 5432, - "schema": "public", - "user": "postgres" - } - ], - "type": "object", - "properties": { - "batch_size": { - "description": "The snapshot batch size", - "type": [ - "integer", - "null" - ], - "format": "uint32", - "minimum": 0.0 - }, - "connection_url": { - "description": "The connection url to use", - "type": [ - "string", - "null" - ] - }, - "database": { - "description": "The database to connect to (default: postgres)", - "type": [ - "string", - "null" - ] - }, - "host": { - "description": "The host to connect to (IP or DNS name)", - "type": [ - "string", - "null" - ] - }, - "password": { - "description": "The password to use for authentication", - "type": [ - "string", - "null" - ] - }, - "port": { - "description": "The port to connect to (default: 5432)", - "type": [ - "integer", - "null" - ], - "format": "uint32", - "minimum": 0.0 - }, - "schema": { - "description": "The schema of the tables", - "type": [ - "string", - "null" - ] - }, - "sslmode": { - "description": "The sslmode to use for the connection (disable, prefer, require)", - "type": [ - "string", - "null" - ] - }, - "user": { - "description": "The username to use for authentication", - "type": [ - "string", - "null" - ] - } - }, - "additionalProperties": false - }, - "PrometheusConfig": { - "type": "object", - "properties": { - "address": { - "default": "0.0.0.0:8089", - "type": "string" - } - }, - "additionalProperties": false - }, - "RefreshConfig": { - "type": "string", - "enum": [ - "RealTime" - ] - }, - "ReplicationSettings": { - "type": "object", - "properties": { - "datacenter": { - "default": "esp", - "type": "string" - }, - "server_address": { - "default": "0.0.0.0", - "type": "string" - }, - "server_port": { - "default": 5929, - "type": "integer", - "format": "uint32", - "minimum": 0.0 - } - } - }, - "RestApiOptions": { - "type": "object", - "properties": { - "cors": { - "type": [ - "boolean", - "null" - ] - }, - "enable_sql": { - "type": [ - "boolean", - "null" - ] - }, - "enabled": { - "type": [ - "boolean", - "null" - ] - }, - "host": { - "type": [ - "string", - "null" - ] - }, - "port": { - "type": [ - "integer", - "null" - ], - "format": "uint16", - "minimum": 0.0 - } - }, - "additionalProperties": false - }, - "S3Details": { - "type": "object", - "required": [ - "access_key_id", - "bucket_name", - "region", - "secret_access_key" - ], - "properties": { - "access_key_id": { - "type": "string" - }, - "bucket_name": { - "type": "string" - }, - "region": { - "type": "string" - }, - "secret_access_key": { - "type": "string" - } - } - }, - "S3Storage": { - "examples": [ - { - "details": { - "access_key_id": "", - "bucket_name": "", - "region": "", - "secret_access_key": "" - }, - "tables": [ - { - "config": { - "CSV": { - "extension": ".csv", - "path": "path/to/file" - } - }, - "name": "table_name" - } - ] - } - ], - "type": "object", - "required": [ - "details", - "tables" - ], - "properties": { - "details": { - "$ref": "#/definitions/S3Details" - }, - "tables": { - "type": "array", - "items": { - "$ref": "#/definitions/Table" - } - } - } - }, - "Sink": { - "type": "object", - "required": [ - "config", - "name" - ], - "properties": { - "config": { - "$ref": "#/definitions/SinkConfig" - }, - "name": { - "type": "string" - } - }, - "additionalProperties": false - }, - "SinkConfig": { - "oneOf": [ - { - "type": "object", - "required": [ - "Dummy" - ], - "properties": { - "Dummy": { - "$ref": "#/definitions/DummySinkConfig" - } - }, - "additionalProperties": false - }, - { - "type": "object", - "required": [ - "Aerospike" - ], - "properties": { - "Aerospike": { - "$ref": "#/definitions/AerospikeSinkConfig" - } - }, - "additionalProperties": false - }, - { - "type": "object", - "required": [ - "Clickhouse" - ], - "properties": { - "Clickhouse": { - "$ref": "#/definitions/ClickhouseSinkConfig" - } - }, - "additionalProperties": false - }, - { - "type": "object", - "required": [ - "Oracle" - ], - "properties": { - "Oracle": { - "$ref": "#/definitions/OracleSinkConfig" - } - }, - "additionalProperties": false - } - ] - }, - "SnowflakeConfig": { - "examples": [ - { - "database": "database", - "driver": "SnowflakeDSIIDriver", - "password": "password", - "port": "443", - "role": "role", - "schema": "schema", - "server": "..snowflakecomputing.com", - "user": "bob", - "warehouse": "warehouse" - } - ], - "type": "object", - "required": [ - "database", - "password", - "port", - "role", - "schema", - "server", - "user", - "warehouse" - ], - "properties": { - "database": { - "type": "string" - }, - "driver": { - "type": [ - "string", - "null" - ] - }, - "password": { - "type": "string" - }, - "poll_interval_seconds": { - "type": "number", - "format": "double" - }, - "port": { - "type": "string" - }, - "role": { - "type": "string" - }, - "schema": { - "type": "string" - }, - "server": { - "type": "string" - }, - "user": { - "type": "string" - }, - "warehouse": { - "type": "string" - } - } - }, - "Source": { - "type": "object", - "required": [ - "connection", - "name", - "table_name" - ], - "properties": { - "columns": { - "description": "list of columns gonna be used in the source table; Type: String[]", - "type": "array", - "items": { - "type": "string" - } - }, - "connection": { - "description": "reference to pre-defined connection name; Type: String", - "type": "string" - }, - "name": { - "description": "name of the source - to distinguish between multiple sources; Type: String", - "type": "string" - }, - "refresh_config": { - "description": "setting for how to refresh the data; Default: RealTime", - "allOf": [ - { - "$ref": "#/definitions/RefreshConfig" - } - ] - }, - "schema": { - "description": "name of schema source database; Type: String", - "type": [ - "string", - "null" - ] - }, - "table_name": { - "description": "name of the table in source database; Type: String", - "type": "string" - } - }, - "additionalProperties": false - }, - "Table": { - "type": "object", - "required": [ - "config", - "name" - ], - "properties": { - "config": { - "$ref": "#/definitions/TableConfig" - }, - "name": { - "type": "string" - } - } - }, - "TableConfig": { - "oneOf": [ - { - "type": "object", - "required": [ - "CSV" - ], - "properties": { - "CSV": { - "$ref": "#/definitions/CsvConfig" - } - }, - "additionalProperties": false - }, - { - "type": "object", - "required": [ - "Parquet" - ], - "properties": { - "Parquet": { - "$ref": "#/definitions/ParquetConfig" - } - }, - "additionalProperties": false - } - ] - }, - "TelemetryConfig": { - "type": "object", - "properties": { - "application_id": { - "default": 0, - "type": "integer", - "format": "uint", - "minimum": 0.0 - }, - "metrics": { - "anyOf": [ - { - "$ref": "#/definitions/TelemetryMetricsConfig" - }, - { - "type": "null" - } - ] - }, - "trace": { - "anyOf": [ - { - "$ref": "#/definitions/TelemetryTraceConfig" - }, - { - "type": "null" - } - ] - } - }, - "additionalProperties": false - }, - "TelemetryMetricsConfig": { - "oneOf": [ - { - "type": "object", - "required": [ - "Prometheus" - ], - "properties": { - "Prometheus": { - "$ref": "#/definitions/PrometheusConfig" - } - }, - "additionalProperties": false - } - ] - }, - "TelemetryTraceConfig": { - "oneOf": [ - { - "type": "object", - "required": [ - "XRay" - ], - "properties": { - "XRay": { - "$ref": "#/definitions/XRayConfig" - } - }, - "additionalProperties": false - } - ] - }, - "UdfConfig": { - "type": "object", - "required": [ - "config", - "name" - ], - "properties": { - "config": { - "description": "setting for what type of udf to use; Default: Onnx", - "allOf": [ - { - "$ref": "#/definitions/UdfType" - } - ] - }, - "name": { - "description": "name of the model function", - "type": "string" - } - }, - "additionalProperties": false - }, - "UdfType": { - "oneOf": [ - { - "type": "object", - "required": [ - "Onnx" - ], - "properties": { - "Onnx": { - "$ref": "#/definitions/OnnxConfig" - } - }, - "additionalProperties": false - }, - { - "type": "object", - "required": [ - "JavaScript" - ], - "properties": { - "JavaScript": { - "$ref": "#/definitions/JavaScriptConfig2" - } - }, - "additionalProperties": false - } - ] - }, - "WebhookConfig": { - "examples": [ - { - "endpoints": [ - { - "path": "/ingest", - "schema": { - "Inline": "\n {\n \"users\": {\n \"schema\": {\n \"fields\": [\n {\n \"name\": \"id\",\n \"typ\": \"Int\",\n \"nullable\": false\n },\n {\n \"name\": \"name\",\n \"typ\": \"String\",\n \"nullable\": true\n },\n {\n \"name\": \"json\",\n \"typ\": \"Json\",\n \"nullable\": true\n }\n ]\n }\n }\n }\n " - }, - "verbs": [ - "POST", - "DELETE" - ] - } - ], - "host": "localhost", - "port": 50059 - } - ], - "type": "object", - "required": [ - "endpoints" - ], - "properties": { - "endpoints": { - "type": "array", - "items": { - "$ref": "#/definitions/WebhookEndpoint" - } - }, - "host": { - "type": [ - "string", - "null" - ] - }, - "port": { - "type": [ - "integer", - "null" - ], - "format": "uint32", - "minimum": 0.0 - } - } - }, - "WebhookConfigSchemas": { - "oneOf": [ - { - "type": "object", - "required": [ - "Inline" - ], - "properties": { - "Inline": { - "type": "string" - } - }, - "additionalProperties": false - }, - { - "type": "object", - "required": [ - "Path" - ], - "properties": { - "Path": { - "type": "string" - } - }, - "additionalProperties": false - } - ] - }, - "WebhookEndpoint": { - "examples": [ - { - "path": "/ingest", - "schema": { - "Inline": "\n {\n \"users\": {\n \"schema\": {\n \"fields\": [\n {\n \"name\": \"id\",\n \"typ\": \"Int\",\n \"nullable\": false\n },\n {\n \"name\": \"name\",\n \"typ\": \"String\",\n \"nullable\": true\n },\n {\n \"name\": \"json\",\n \"typ\": \"Json\",\n \"nullable\": true\n }\n ]\n }\n }\n }\n " - }, - "verbs": [ - "POST", - "DELETE" - ] - } - ], - "type": "object", - "required": [ - "path", - "schema", - "verbs" - ], - "properties": { - "path": { - "type": "string" - }, - "schema": { - "$ref": "#/definitions/WebhookConfigSchemas" - }, - "verbs": { - "type": "array", - "items": { - "$ref": "#/definitions/WebhookVerb" - } - } - } - }, - "WebhookVerb": { - "examples": [ - "POST" - ], - "type": "string", - "enum": [ - "POST", - "PUT", - "DELETE" - ] - }, - "XRayConfig": { - "type": "object", - "required": [ - "endpoint", - "timeout_in_seconds" - ], - "properties": { - "endpoint": { - "type": "string" - }, - "timeout_in_seconds": { - "type": "integer", - "format": "uint64", - "minimum": 0.0 - } - }, - "additionalProperties": false - } - } -} \ No newline at end of file