diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c258922f..7616c566 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -441,25 +441,42 @@ jobs: if: always() && matrix.name == 'runtime' run: docker rm --force sensapp-ci-runtime || true - docker-bigquery-compile: - name: Experimental BigQuery Image Compile + # No Google Cloud project in CI: the BigQuery backend is compiled, linted and unit tested (the SQL + # it builds, the rows it sends, the errors it sorts). The live tests are run by hand, see + # docs/BIGQUERY.md. Not a requirement of the release. + bigquery-checks: + name: BigQuery Compile and Unit Tests runs-on: ubuntu-latest - timeout-minutes: 45 needs: quality-checks - if: github.event_name == 'workflow_dispatch' steps: - - uses: actions/checkout@v6 - - uses: docker/setup-buildx-action@v4 - - uses: docker/build-push-action@v7 + - name: Checkout repository + uses: actions/checkout@v6 + + - name: Setup Rust toolchain + uses: dtolnay/rust-toolchain@master with: - context: . - push: false - platforms: linux/amd64 - build-args: | - FEATURES=postgres,sqlite,timescaledb,duckdb,clickhouse,rrdcached,bigquery - NO_DEFAULT_FEATURES=true - cache-from: type=gha,scope=runtime-bigquery - cache-to: type=gha,scope=runtime-bigquery,mode=max + toolchain: stable + components: clippy + + - name: Cache Rust dependencies + uses: Swatinem/rust-cache@v2 + with: + key: bigquery-v1 + save-if: ${{ github.ref == 'refs/heads/main' || github.ref == 'refs/heads/develop' }} + cache-on-failure: true + + - name: Install cargo-make (prebuilt) + uses: taiki-e/install-action@v2 + with: + tool: cargo-make + + - name: Run BigQuery checks + run: cargo make check-bigquery + + # The backends of the Docker image plus BigQuery must compile together (DuckDB is left out: its + # library download is slow, and it does not interact with BigQuery) + - name: Lint the backends together + run: cargo clippy --all-targets --no-default-features --features postgres,sqlite,timescaledb,clickhouse,rrdcached,bigquery -- -D warnings # Disabled temporarily due to timeout issues # coverage: diff --git a/.gitignore b/.gitignore index 88a058e0..9b8be601 100644 --- a/.gitignore +++ b/.gitignore @@ -41,3 +41,7 @@ python/demo-dashboard/ bin/ target + +# Google Cloud service account keys (docs/BIGQUERY.md) +key.json +*-key.json diff --git a/Cargo.lock b/Cargo.lock index 40c075f3..2e7f9818 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -853,31 +853,6 @@ version = "1.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2af50177e190e07a26ab74f8b1efbfe2ef87da2116221318cb1c2e82baf7de06" -[[package]] -name = "big-decimal-byte-string-encoder" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a511cec22ed1f46a4f4cb4cfee83bbe41633539a446aef645af5479457712ccd" -dependencies = [ - "bigdecimal", - "num-bigint 0.4.8", - "once_cell", - "thiserror 1.0.69", -] - -[[package]] -name = "bigdecimal" -version = "0.4.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4d6867f1565b3aad85681f1015055b087fcfd840d6aeee6eee7f2da317603695" -dependencies = [ - "autocfg", - "libm", - "num-bigint 0.4.8", - "num-integer", - "num-traits", -] - [[package]] name = "bitflags" version = "2.13.2" @@ -896,18 +871,6 @@ dependencies = [ "no_std_io2", ] -[[package]] -name = "bitvec" -version = "1.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ddcec3d12c579d40898fe0a9a358a803c23e9c52ca3c425707f81c9436211837" -dependencies = [ - "funty", - "radium", - "tap", - "wyz", -] - [[package]] name = "blake3" version = "1.8.7" @@ -1062,7 +1025,7 @@ dependencies = [ "cached_proc_macro_types", "hashbrown 0.17.1", "parking_lot", - "thiserror 2.0.21", + "thiserror", "web-time", ] @@ -1171,7 +1134,7 @@ dependencies = [ "rustls", "serde", "serde_json", - "thiserror 2.0.21", + "thiserror", "time", "tokio", "tracing", @@ -1198,16 +1161,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3966eea13bfc826e24d471e49cb2095727261009be59d0ac5a6c0be0286974d1" dependencies = [ "bytes", - "thiserror 2.0.21", -] - -[[package]] -name = "clru" -version = "0.6.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "197fd99cb113a8d5d9b6376f3aa817f32c1078f2343b714fff7d2ca44fdf67d5" -dependencies = [ - "hashbrown 0.16.1", + "thiserror", ] [[package]] @@ -2067,12 +2021,6 @@ version = "1.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c" -[[package]] -name = "funty" -version = "2.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e6d5a32815ae3f33302d95fdcb2ce17862f8c65363dcfd29360480ba1001fc9c" - [[package]] name = "futures" version = "0.3.34" @@ -2193,7 +2141,7 @@ dependencies = [ "reqwest 0.12.28", "serde", "serde_json", - "thiserror 2.0.21", + "thiserror", "time", "tokio", "tokio-stream", @@ -2247,7 +2195,7 @@ dependencies = [ "rayon", "rstar", "serde", - "thiserror 2.0.21", + "thiserror", ] [[package]] @@ -3995,7 +3943,7 @@ dependencies = [ "rustc-hash", "rustls", "socket2", - "thiserror 2.0.21", + "thiserror", "tokio", "tracing", "web-time", @@ -4017,7 +3965,7 @@ dependencies = [ "rustls", "rustls-pki-types", "slab", - "thiserror 2.0.21", + "thiserror", "tinyvec", "tracing", "web-time", @@ -4058,12 +4006,6 @@ version = "6.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" -[[package]] -name = "radium" -version = "0.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc33ff2d4973d518d823d61aa239014831e521c75da58e3df4840d3f47749d09" - [[package]] name = "rand" version = "0.8.8" @@ -4356,7 +4298,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2ed7015424e281a765486545c5904767403747aa74117ccdd93b5cde2c19cf83" dependencies = [ "nom 8.0.0", - "thiserror 2.0.21", + "thiserror", "tokio", ] @@ -4521,7 +4463,7 @@ dependencies = [ "crc32c", "nom 8.0.0", "smallvec 2.0.0-alpha.12", - "thiserror 2.0.21", + "thiserror", "xxhash-rust", ] @@ -4628,13 +4570,10 @@ dependencies = [ "async-trait", "axum", "base64 0.22.1", - "big-decimal-byte-string-encoder", - "bigdecimal", "blake3", "byte-unit", "cached", "clickhouse", - "clru", "confique", "csv-async", "duckdb", @@ -4665,17 +4604,15 @@ dependencies = [ "serial_test", "simplify-polyline", "sindit-senml", - "sinteflake", "smallvec 1.16.2", "snap", "sqlx", "temp-env", - "thiserror 2.0.21", + "thiserror", "tokio", "tokio-stream", "tokio-test", "tokio-util", - "tonic", "tower", "tower-http", "tracing", @@ -4827,7 +4764,7 @@ dependencies = [ "rand 0.9.5", "serde", "serde_json", - "thiserror 2.0.21", + "thiserror", "time", "url", "uuid", @@ -5038,7 +4975,7 @@ checksum = "0d585997b0ac10be3c5ee635f1bab02d512760d14b7c468801ac8a01d9ae5f1d" dependencies = [ "num-bigint 0.4.8", "num-traits", - "thiserror 2.0.21", + "thiserror", "time", ] @@ -5063,29 +5000,9 @@ dependencies = [ "regex", "serde", "serde_json", - "thiserror 2.0.21", -] - -[[package]] -name = "sinteflake" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3d455901da270bc89ab9104afb1b08a57e25be9b3e833e1a5bfe5f1cc69ff07b" -dependencies = [ - "bitvec", - "once_cell", - "siphasher", - "thiserror 1.0.69", - "time", - "tokio", + "thiserror", ] -[[package]] -name = "siphasher" -version = "1.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "33f4fe9184a62d842c9ef383018f3306d8ba224fd9d836f56d7288308847c256" - [[package]] name = "slab" version = "0.4.12" @@ -5238,7 +5155,7 @@ dependencies = [ "serde_json", "sha2 0.10.9", "smallvec 1.16.2", - "thiserror 2.0.21", + "thiserror", "time", "tokio", "tokio-stream", @@ -5281,7 +5198,7 @@ dependencies = [ "sqlx-postgres", "sqlx-sqlite", "syn 2.0.119", - "thiserror 2.0.21", + "thiserror", "tokio", "url", ] @@ -5309,7 +5226,7 @@ dependencies = [ "sha1", "sha2 0.11.0", "sqlx-core", - "thiserror 2.0.21", + "thiserror", "time", "tracing", "uuid", @@ -5346,7 +5263,7 @@ dependencies = [ "smallvec 1.16.2", "sqlx-core", "stringprep", - "thiserror 2.0.21", + "thiserror", "time", "tracing", "uuid", @@ -5373,7 +5290,7 @@ dependencies = [ "regex", "serde", "sqlx-core", - "thiserror 2.0.21", + "thiserror", "time", "tracing", "url", @@ -5483,12 +5400,6 @@ dependencies = [ "syn 3.0.6", ] -[[package]] -name = "tap" -version = "1.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "55937e1799185b12863d447f42597ed69d9928686b8d88a1df17376a097d8369" - [[package]] name = "tar" version = "0.4.46" @@ -5522,33 +5433,13 @@ dependencies = [ "windows-sys 0.61.2", ] -[[package]] -name = "thiserror" -version = "1.0.69" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b6aaf5339b578ea85b50e080feb250a3e8ae8cfcdff9a461c9ec2904bc923f52" -dependencies = [ - "thiserror-impl 1.0.69", -] - [[package]] name = "thiserror" version = "2.0.21" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "09e52cb86a36cede5cb101bf8908837b3e4c6e5e59fe7fd85c23fb56200d189e" dependencies = [ - "thiserror-impl 2.0.21", -] - -[[package]] -name = "thiserror-impl" -version = "1.0.69" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4fee6c4efc90059e10f81e6d42c60a18f76588c3d74cb83a0b242a2b6c7504c1" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", + "thiserror-impl", ] [[package]] @@ -5779,11 +5670,9 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ac2a5518c70fa84342385732db33fb3f44bc4cc748936eb5833d2df34d6445ef" dependencies = [ "async-trait", - "axum", "base64 0.22.1", "bytes", "flate2", - "h2", "http 1.5.0", "http-body", "http-body-util", @@ -5793,7 +5682,6 @@ dependencies = [ "percent-encoding", "pin-project", "rustls-native-certs", - "socket2", "sync_wrapper", "tokio", "tokio-rustls", @@ -6511,15 +6399,6 @@ version = "0.6.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3ad82d2a33cdc9674dc7465672f271e096168fcdbe0f799d9e6db8c5892679dc" -[[package]] -name = "wyz" -version = "0.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "05f360fc0b24296329c78fda852a1e9ae82de9cf7b27dae4b7f62f118f77b9ed" -dependencies = [ - "tap", -] - [[package]] name = "xattr" version = "1.6.1" @@ -6578,7 +6457,7 @@ dependencies = [ "seahash", "serde", "serde_json", - "thiserror 2.0.21", + "thiserror", "time", "tokio", "url", diff --git a/Cargo.toml b/Cargo.toml index ddbc59a6..10e0462d 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -15,12 +15,7 @@ postgres = ["sqlx/postgres"] sqlite = ["sqlx/sqlite", "sqlx/regexp"] timescaledb = ["sqlx/postgres"] # Uses same driver as postgres duckdb = ["dep:duckdb"] -bigquery = [ - "dep:gcp-bigquery-client", - "dep:bigdecimal", - "dep:big-decimal-byte-string-encoder", - "dep:tonic", -] +bigquery = ["dep:gcp-bigquery-client"] rrdcached = ["dep:rrdcached-client"] clickhouse = ["dep:clickhouse"] # Convenience feature to enable all storage backends @@ -95,11 +90,6 @@ hybridmap = "0.1.3" sentry = { version = "0.47", features = ["anyhow", "tower"] } url = "2.5" gcp-bigquery-client = { version = "0.28", optional = true } -sinteflake = { version = "0.1", features = ["async"] } -tonic = { version = "0.14", optional = true } -bigdecimal = { version = "0.4", optional = true } -big-decimal-byte-string-encoder = { version = "0.1", optional = true } -clru = "0.6" utoipa = { version = "6", features = ["axum_extras"] } utoipa-scalar = { version = "0.4", features = ["axum"] } rrdcached-client = { version = "0.4", optional = true } diff --git a/Makefile.toml b/Makefile.toml index 0959d730..7098b3d1 100644 --- a/Makefile.toml +++ b/Makefile.toml @@ -23,6 +23,7 @@ TEST_DATABASE_URL_TIMESCALEDB = { value = "timescaledb://postgres:postgres@local TEST_DATABASE_URL_DUCKDB = { value = "duckdb://test.duckdb", condition = { env_not_set = ["TEST_DATABASE_URL_DUCKDB"] } } TEST_DATABASE_URL_CLICKHOUSE = { value = "clickhouse://default:password@localhost:8123/sensapp_test", condition = { env_not_set = ["TEST_DATABASE_URL_CLICKHOUSE"] } } TEST_DATABASE_URL_RRDCACHED = { value = "rrdcached://127.0.0.1:42217?preset=hoarder", condition = { env_not_set = ["TEST_DATABASE_URL_RRDCACHED"] } } +# BigQuery has no default: it needs a Google Cloud project (docs/BIGQUERY.md) # Environment profiles [env.ci] @@ -299,6 +300,15 @@ env = { "CARGO_MAKE_TASK_FEATURE" = "bigquery" } command = "cargo" args = ["test", "--features", "bigquery", "--no-default-features", "--lib", "--bins"] +# Runs the whole integration suite against a real dataset: it bills queries, see docs/BIGQUERY.md. +# Set TEST_DATABASE_URL_BIGQUERY=bigquery://key.json?project_id=P&dataset_id=D&max_bytes_billed=N +[tasks.test-bigquery-live] +private = false +description = "Integration tests against a real BigQuery dataset (TEST_DATABASE_URL_BIGQUERY, costs money)" +env = { "TEST_DATABASE_URL" = "${TEST_DATABASE_URL_BIGQUERY}" } +command = "cargo" +args = ["test", "--features", "bigquery", "--no-default-features", "--test", "integration"] + [tasks.clippy-bigquery] extend = "template-clippy" private = false diff --git a/TODO.md b/TODO.md index e3c2bd79..5bd95c12 100644 --- a/TODO.md +++ b/TODO.md @@ -42,7 +42,7 @@ Done: Prometheus scrape endpoint with HTTP request count, duration and in-flight These are valid tasks, but not the current focus. Most have a note in `ideas/`. -- [ ] Bring BigQuery back in sync with the current storage trait and query model (`ideas/bigquery-backend-reconciliation.md`) +- [x] BigQuery back in sync with the storage trait and query model, integration suite passing on a real dataset (`done/bigquery-back-in-sync.md`, `docs/BIGQUERY.md`) - [ ] Add benchmark tooling for storage backend comparison (a first script exists: `tests/perf/scale.sh`, writes and selector reads of thousands of series through the HTTP API) - [ ] Add research-specific comparison endpoints and reporting helpers - [ ] Add storage-space and latency comparison reports across backends diff --git a/docs/BACKENDS.md b/docs/BACKENDS.md index 225cf11a..ca393b2d 100644 --- a/docs/BACKENDS.md +++ b/docs/BACKENDS.md @@ -14,7 +14,7 @@ mature. This page says which ones to rely on, and what each one is for. | SQLite | **Maintained** | yes | Tests, demos, single-node edge devices, one writer | | DuckDB | Compatibility path, **less mature** | yes | Local analysis of a dataset, notebooks. Writes are bulk (3000 new series in 0.2 s) | | RRDCached | **Experimental**, good enough | yes (its own job, a real daemon) | Fixed-size round-robin storage of numeric monitoring-style data. No names, labels, types or deletion, reads are consolidated rows | -| BigQuery | **Experimental, parked** | compile only | Nothing yet: it has not been brought back in line with the current storage interface (see `ideas/bigquery-backend-reconciliation.md`) | +| BigQuery | **Experimental**, R&D | compile and unit tests; the integration suite is run by hand on a real dataset (passed on 4 Oct 2026) | Comparing a warehouse with the time series databases, SQL over sensor data next to other datasets. Every query is billed. [BIGQUERY.md](BIGQUERY.md) | "Maintained" means a regression in it fails CI, the backend-generic integration tests run on it, and bugs found on it are fixed. "Experimental" means it works for the basic ingest and query paths, may lag behind new @@ -25,12 +25,12 @@ features, and has no promise. | | ClickHouse | PostgreSQL | TimescaleDB | SQLite | DuckDB | RRDCached | BigQuery | | --- | --- | --- | --- | --- | --- | --- | --- | | Values of every type | yes | yes | yes | yes | yes | numeric data points | yes | -| Delete a series or its samples | yes | yes | yes | yes | yes | no | no | -| Aggregations (`step`) in the database | yes | yes | yes | yes | yes | no | no | +| Delete a series or its samples | yes | yes | yes | yes | yes | no | yes | +| Aggregations (`step`) in the database | yes | yes | yes | yes | yes | no (in SensApp, on the consolidated rows) | yes | | Remove duplicate samples (vacuum) | yes | yes | yes | yes | yes | no | no | -| Registration of many new series per write in bulk | yes | yes | yes | one by one, fast locally | one by one | no | no | -| A selector reads its series with a few queries | yes | yes | yes | yes | yes (numeric series; aggregated ones one at a time) | one series at a time | one series at a time | -| Prometheus remote read with a `step` aggregates its series with a few queries | yes | yes | yes | yes | one series at a time (each aggregation is a single query) | one series at a time | one series at a time | +| Registration of many new series per write in bulk | yes | yes | yes | one by one, fast locally | one by one | no | yes | +| A selector reads its series with a few queries | yes | yes | yes | yes | yes (numeric series; aggregated ones one at a time) | one series at a time | yes (numeric series) | +| Prometheus remote read with a `step` aggregates its series with a few queries | yes | yes | yes | yes | one series at a time (each aggregation is a single query) | one series at a time | yes | | Replication | not created by SensApp | the database's own | the database's own | no | no | no | managed | The last two rows are a consequence of how the code is written, not of the databases: a backend that reads @@ -60,7 +60,10 @@ or writes series one by one still works, it costs more round trips with many ser its UUID. It is tested against a real `rrdcached` with a test module of its own (the backend-generic suite does not apply). All the differences, the presets, the measured performance: [RRDCACHED.md](RRDCACHED.md). -- **BigQuery**: needs Google Cloud credentials and is only compiled in CI. +- **BigQuery**: a warehouse, not a time series database: every statement is a job of a few hundred milliseconds and + bills at least 10 MB. Writes are at least once and cannot be deduplicated (like ClickHouse), aggregation runs in + BigQuery. It needs a Google Cloud project, and CI only compiles it and runs its unit tests; the integration suite is + run by hand. Setup, cost and limits: [BIGQUERY.md](BIGQUERY.md). ## Features: production-oriented and research-oriented diff --git a/docs/BIGQUERY.md b/docs/BIGQUERY.md new file mode 100644 index 00000000..9024befc --- /dev/null +++ b/docs/BIGQUERY.md @@ -0,0 +1,162 @@ +# BigQuery Storage Backend + +SensApp can store its series in [BigQuery](https://cloud.google.com/bigquery). It exists for research and +development (comparing a warehouse with the time series databases, running SQL over sensor data next to other +datasets), not for production: it is **experimental**, kept simple on purpose, and the first thing to know is +that every query is billed. See [BACKENDS.md](BACKENDS.md) for how it compares to the others. + +It is tested by hand: CI compiles it and runs its unit tests, but has no Google Cloud project (see +[Testing](#testing)). The whole integration suite, backend-generic modules included, passed on a real dataset +on 4 October 2026. + +## Connection string + +``` +bigquery://[key.json]?project_id=P&dataset_id=D[&location=europe-north1][&max_bytes_billed=1000000000] +``` + +| Part | Meaning | +|---|---| +| `key.json` | Path of a service account key (`bigquery:///abs/path/key.json` for an absolute one). Without it SensApp uses the Application Default Credentials: `GOOGLE_APPLICATION_CREDENTIALS`, what `gcloud auth application-default login` stored, or the identity of the Google Cloud machine it runs on | +| `project_id` | The project that runs and pays for the statements, and holds the dataset | +| `dataset_id` | Created at startup if it does not exist (in `location`, `europe-north1` by default); the tables are created with `CREATE TABLE IF NOT EXISTS`. **Use a dataset that is only SensApp's** | +| `location` | Where the dataset is created. Statements run where the dataset is, unless this says otherwise | +| `max_bytes_billed` | A statement that would bill more than this fails and is not charged. A good safety net for experiments: a statement bills at least 10 MB per table it reads, so a few hundred MB (`500000000`) to a few GB is realistic | + +Unknown parameters are refused. The build needs the feature: `cargo build --features bigquery` (the container +image can include it, see the `FEATURES` build argument). + +## Schema + +Created by `src/storage/bigquery/migrations/init.sql`: `units`, `sensors`, `labels`, and one table per sample type +(`integer_values`, `numeric_values`, `float_values`, `string_values`, `boolean_values`, `location_values`, +`json_values`, `blob_values`). Sample tables are partitioned by month and clustered by `sensor_id`. + +- **Ids** are derived from the data, like ClickHouse's: the series id is the UUID folded to 64 bits, the unit id a + hash of its name (`storage::common`). BigQuery has no sequences and no unique keys, so this is what makes two + writers that register the same series at once insert identical rows instead of two different ids. The reads + collapse such rows. +- **Labels and strings are stored as text** in their rows, there are no dictionary tables. The datasets of the + first version of the backend (with `labels_name_dictionary` and friends) are refused at startup with a message: + use a new dataset. +- **Timestamps** are `TIMESTAMP`, written as microseconds since the epoch. BigQuery has a microsecond + precision and accepts years 1 to 9999: nanoseconds are rounded down, and a query window is clamped to the range. +- **Values**: floats and locations are `FLOAT64` and exact. `NUMERIC` keeps **9** decimal digits (rounded to even + when written), ClickHouse keeps 8. JSON is a `JSON` column: BigQuery normalizes it (the order of the keys is not + kept). Blobs are `BYTES`. + +## What is different + +| | | +|---|---| +| **Writes** | The Storage Write API, default stream: rows are queryable at once, delivery is *at least once*. A request that timed out and is retried can leave **duplicate samples**, BigQuery cannot refuse them, and SensApp cannot remove them: there is no deduplication at ingestion (`SENSAPP_DEDUPLICATE_ON_INGEST` makes the server refuse to start) and the vacuum does not remove duplicates (`duplicates_removed` is `null`). There is no transaction across tables: a failure can leave a sensor's labels without its samples, never the opposite for a series that is listed | +| **Latency and cost** | Every statement is a job: 0.3 to 1 s even for one row, and a read of a tiny table bills 10 MB. A write looks up the new series (cached for 2 minutes per instance), writes up to 11 tables, one append each. Prefer large batches. Query results are never served from BigQuery's cache, so a read sees what was just written | +| **Deleting** | `DELETE` statements (rows of the Storage Write API can be deleted). BigQuery allows 2 mutating statements at a time per table and queues 20: deleting many series quickly fails with "Too many DML statements". A write to a series that another instance deleted, within 2 minutes, is stored without its sensor and is not seen | +| **Aggregation** | In BigQuery, for numeric series: `step` and `aggregation` are a `GROUP BY` of buckets (`AVG`, `MIN`, `MAX`, `SUM`, `COUNT`, first and last value), so only the buckets come back. The statement still **scans** the window of the series, billed by the bytes of the three columns it reads (about 24 bytes a sample, less once compressed): a 6 hour average over 6 months of one series reads 6 months of that series, and the monthly partitions and the clustering on `sensor_id` keep it to that. Selectors, with or without a `step`, read their series with one statement per numeric type. An average of `NUMERIC` is rounded to 9 digits | +| **Matchers** | `=`, `!=`, `=~`, `!~` on names and labels, with RE2 (`REGEXP_CONTAINS`, anchored). As on ClickHouse, `!=` and `!~` also select the series that do not have the label | +| **Reading a large result** | All the pages of the result are read (10 MB each), and a statement that outlasts its first answer is waited for (up to 5 minutes, then the request fails as unavailable) | +| **Errors** | Overload, quota errors that say "retry" and network failures are `503`; a rejected statement or write is `500` | +| **Not implemented** | Vacuum (nothing to do), removing duplicates, deduplication at ingestion | + +## Setting up Google Cloud + +You need a project with billing enabled (the BigQuery sandbox does not allow the DML and the Storage Write API +SensApp uses). A throwaway project is the easiest way to keep the cost and the permissions contained. + +```bash +# 1. The project. The id is global, pick your own (or use an existing project and skip to 3) +PROJECT_ID=sensapp-test-$RANDOM +gcloud auth login +gcloud projects create "$PROJECT_ID" --name="SensApp test" # inside an organization: add --organization=ORG_ID or --folder=FOLDER_ID + +# 2. Billing (needs the Billing User role on the billing account) +gcloud billing accounts list +gcloud billing projects link "$PROJECT_ID" --billing-account=XXXXXX-XXXXXX-XXXXXX + +# 3. The APIs +gcloud config set project "$PROJECT_ID" +gcloud services enable bigquery.googleapis.com bigquerystorage.googleapis.com + +# 4. A service account that runs jobs and edits the data, and its key +gcloud iam service-accounts create sensapp-test --display-name="SensApp test" +SA="sensapp-test@$PROJECT_ID.iam.gserviceaccount.com" +gcloud projects add-iam-policy-binding "$PROJECT_ID" --member="serviceAccount:$SA" --role=roles/bigquery.jobUser +gcloud projects add-iam-policy-binding "$PROJECT_ID" --member="serviceAccount:$SA" --role=roles/bigquery.dataEditor +gcloud iam service-accounts keys create key.json --iam-account="$SA" + +# 5. The dataset the tests will wipe (SensApp creates it when it is missing, in europe-north1 by default) +bq --project_id="$PROJECT_ID" mk --dataset --location=europe-north1 "$PROJECT_ID:sensapp_test" + +# 6. A budget alert at 5 dollars (50%, 90% and 100% of it) +BILLING_ACCOUNT=XXXXXX-XXXXXX-XXXXXX +gcloud billing budgets create --billing-account="$BILLING_ACCOUNT" --display-name="SensApp test" \ + --budget-amount=5USD --filter-projects="projects/$PROJECT_ID" \ + --threshold-rule=percent=0.5 --threshold-rule=percent=0.9 --threshold-rule=percent=1.0 +``` + +Project-level roles keep the commands short; they are fine in a project that only exists for this. `bq` comes +with the Google Cloud CLI. + +If the key command fails with `iam.disableServiceAccountKeyCreation` or `iam.managed.disableServiceAccountKeyCreation` +(a policy of your organization, the default of the ones created since 2024), do not look for a way around it: run as +yourself. Skip the key and leave it out of the connection string: + +```bash +gcloud auth application-default login +unset GOOGLE_APPLICATION_CREDENTIALS +# TEST_DATABASE_URL='bigquery://?project_id=P&dataset_id=sensapp_test&max_bytes_billed=2000000000' +``` + +SensApp then reads `~/.config/gcloud/application_default_credentials.json` (or the file named by +`GOOGLE_APPLICATION_CREDENTIALS`). Only user credentials and service account keys are understood: an +`--impersonate-service-account` file is not. Your account needs `roles/bigquery.jobUser` on the project and +`roles/bigquery.dataEditor` on the dataset. A machine of Google Cloud (Compute Engine, Cloud Run, GKE) with a service +account attached needs neither a key nor a login. If a write fails with a 403 that talks about a quota project, try +`gcloud auth application-default set-quota-project P` and report it: the client does not send one itself. + +Limit what a mistake can cost before the first run: + +- In the console, **IAM & Admin → Quotas & System limits**, filter on "Query usage per day" (BigQuery API) and set a + custom limit for the project, for instance 20 GiB. This is a hard stop. +- The budget alert of step 6 only emails, it does not stop anything: the quota above and `max_bytes_billed` do. +- Keep `max_bytes_billed` in the connection string. + +Keep `key.json` out of the repository (`key.json` and `*-key.json` are in `.gitignore`). The paths in the +connection string are relative to where `cargo test` or `sensapp` runs, the root of the repository. + +Prices change, look at [the BigQuery pricing page](https://cloud.google.com/bigquery/pricing): at the time of +writing the first TiB of queries and 10 GiB of storage per month are free, then queries are billed per TiB +scanned, and the Storage Write API has a free monthly allowance. SensApp's tables are small in tests, and rows +that were just written (still in the streaming buffer) are billed 0 bytes. Measured on 4 October 2026: running +the whole integration suite, with partial and repeated runs, was 16 901 statements and 25 GiB billed, which is +less than a dollar at list price. + +## Testing + +Unit tests need nothing (the SQL SensApp builds, the rows it sends, the sorting of errors, the connection string): + +```bash +cargo make check-bigquery +``` + +The integration tests need the dataset. **They wipe it**: every test starts by deleting all the rows of all the +tables of the dataset, so use a dataset that holds nothing else, like `sensapp_test`. + +```bash +export TEST_DATABASE_URL_BIGQUERY='bigquery://key.json?project_id=my-sensapp-test&dataset_id=sensapp_test&max_bytes_billed=2000000000' + +# What is specific to BigQuery first: every type round trip, concurrent instances, paging, selectors, the cost cap +TEST_DATABASE_URL=$TEST_DATABASE_URL_BIGQUERY cargo test --no-default-features --features bigquery \ + --test integration bigquery_integration:: -- --nocapture + +# Then the backend-generic suites on BigQuery, a few modules at a time +TEST_DATABASE_URL=$TEST_DATABASE_URL_BIGQUERY cargo test --no-default-features --features bigquery \ + --test integration data_lifecycle:: advanced_backend_queries:: query_sensors_by_labels:: selector_reads:: + +# Everything: about 290 tests, a little under an hour (every test is serial and starts with a migration and a +# cleanup of 11 statements, a statement is about 2 seconds) +cargo make test-bigquery-live +``` + +Without a `bigquery:` `TEST_DATABASE_URL` the BigQuery-specific tests print that they are skipped and pass. +The deduplication tests that need a backend that can deduplicate skip themselves, as on ClickHouse. diff --git a/docs/CI.md b/docs/CI.md index 3a937201..2939b589 100644 --- a/docs/CI.md +++ b/docs/CI.md @@ -17,9 +17,10 @@ no npm audit findings. The regenerated client also passed a live API smoke test; see `ideas/remove-frontend-js-yaml-override.md` for the override follow-up. - Separate build, test, and Clippy checks for SQLite, PostgreSQL, ClickHouse, DuckDB, TimescaleDB, and RRDCached. The service-backed jobs use real database - containers. BigQuery remains an experimental compile-only Docker variant on - manual runs until an isolated CI dataset and credentials are - available; it does not block the ClickHouse release path. + containers. BigQuery is compiled, linted and unit tested on every run (`bigquery-checks`), + also together with the other backends of the Docker image. Its integration suite needs a + Google Cloud project, whose queries are billed, and is run by hand + ([BIGQUERY.md](BIGQUERY.md)). It does not block the ClickHouse release path. - Helm lint/template/package and Docker builds. The normal runtime image is actually started with ClickHouse, then `tests/clickhouse_container_smoke.py` checks readiness, publish, query, and service metrics. Live Python SDK tests diff --git a/docs/CONFIGURATION.md b/docs/CONFIGURATION.md index d054c0a1..ec613407 100644 --- a/docs/CONFIGURATION.md +++ b/docs/CONFIGURATION.md @@ -51,7 +51,6 @@ Details on what the limits protect against are in [HTTP_LIMITS.md](HTTP_LIMITS.m | --- | --- | --- | | `SENSAPP_SENSOR_SALT` | `sensapp` | Salt of the hash that turns a sensor's name, type, unit and labels into its UUID. **Changing it changes the UUID of every sensor**: existing data is no longer matched by new writes. Set it once per deployment, before ingesting. | | `SENSAPP_INFLUXDB_WITH_NUMERIC` | `false` | Store InfluxDB numbers as exact decimals instead of floats. Precise but slower. See [INFLUX_DB.md](INFLUX_DB.md). | -| `SENSAPP_INSTANCE_ID` | `0` | Instance number (`u16`) mixed into generated identifiers. Give each SensApp instance writing to the same database a different value. | ### Observability @@ -76,7 +75,7 @@ The scheme of `SENSAPP_STORAGE_CONNECTION_STRING` picks the backend. A backend o | `sqlite://path/to/sensapp.db` | SQLite | Single file, for small setups and tests. | | `duckdb://path/to/sensapp.db` | DuckDB | Experimental. | | `rrdcached://host:42217?preset=munin&heartbeat=3600`, `rrdcached+unix:///path/to.sock` | RRDCached | Experimental: numbers only, no names or labels. `preset` is `hoarder` (default) or `munin`, `heartbeat` is in seconds. See [RRDCACHED.md](RRDCACHED.md). | -| `bigquery://key.json?project_id=P&dataset_id=D` | BigQuery | Experimental, not in sync with the current storage interface. | +| `bigquery://key.json?project_id=P&dataset_id=D`, without `key.json` for the Application Default Credentials | BigQuery | Experimental, for research: every query is billed. Optional `location` and `max_bytes_billed`. See [BIGQUERY.md](BIGQUERY.md). | Backends create and migrate their own schema at startup. [BACKENDS.md](BACKENDS.md) says which ones are maintained and which are experimental, and what each is good for. diff --git a/docs/DATA_LIFECYCLE.md b/docs/DATA_LIFECYCLE.md index 43f15c7c..198697af 100644 --- a/docs/DATA_LIFECYCLE.md +++ b/docs/DATA_LIFECYCLE.md @@ -138,7 +138,7 @@ Two requests that write the same new sample at the same moment, on different ins | TimescaleDB | yes | yes | yes | Works on compressed chunks. `DELETE` on a compressed chunk decompresses what it touches first, which is slower. The vacuum decompresses only the chunks that hold duplicates, removes them and compresses those chunks again | | DuckDB | yes | yes | yes | Exact duplicates are found with `rowid` | | ClickHouse | yes | yes | yes | Lightweight `DELETE`: rows disappear from queries at once and are physically removed by later merges. The sample count is taken just before the delete | -| BigQuery | no | no | no | Returns `501 Not Implemented` | +| BigQuery | yes | yes | no | `DELETE` statements, which bill the bytes they scan. BigQuery runs 2 mutating statements at a time per table and queues 20: deleting many series at once fails. Duplicates cannot be removed: `501 Not Implemented` | | RRDCached | no | no | no | Returns `501 Not Implemented` | ## How other systems do it diff --git a/docs/PREPRODUCTION_RELEASE_PLAN.md b/docs/PREPRODUCTION_RELEASE_PLAN.md index 8564848d..1997ca64 100644 --- a/docs/PREPRODUCTION_RELEASE_PLAN.md +++ b/docs/PREPRODUCTION_RELEASE_PLAN.md @@ -4,8 +4,8 @@ Use the clean `sensapp-sep-26` clone of upstream `main`. The older local checkouts have been reviewed in `done/local-work-reconciliation-september-2026.md`; -none is a safe release base. The useful unfinished BigQuery work is recorded in -`ideas/bigquery-backend-reconciliation.md` and is outside the first ClickHouse +none is a safe release base. The BigQuery backend was rewritten on the current +storage interface (`docs/BIGQUERY.md`) and is outside the first ClickHouse release. ClickHouse already has migration, health, HTTP lifecycle, and query coverage. @@ -48,5 +48,5 @@ backend-only release could defer step 5. Ship the ClickHouse-backed pre-production build when all four steps have evidence. Treat PostgreSQL, SQLite, TimescaleDB, DuckDB, and RRDCached as tested compatibility paths rather than equal deployment promises for this release. -Keep BigQuery experimental until its older read work is ported and tested with a +Keep BigQuery experimental until its integration suite has been run against a real isolated dataset. diff --git a/done/bigquery-back-in-sync.md b/done/bigquery-back-in-sync.md new file mode 100644 index 00000000..b079aaed --- /dev/null +++ b/done/bigquery-back-in-sync.md @@ -0,0 +1,71 @@ +# BigQuery: back in sync with the storage interface + +Replaces `ideas/bigquery-backend-reconciliation.md` (removed). The old local checkout +`/Users/antoinep/work/sensapp-vibe-prom-read` was read for its ideas (cache lifespans, identifier +validation, parallel reads); nothing of it was copied, and it can be deleted once this task is done. + +## Goal + +BigQuery is an optional, R&D backend (like RRDCached): simple, documented limitations, cheap to test. +Bring it to the level of the other backends **without running anything against Google Cloud yet** +(queries cost money, the user will create the project), then say what to set up in GCP and run the +tests. + +## Audit of the old module (3 Oct 2026) + +`cargo check --features bigquery` passes, but only because the trait has default methods: + +| # | Finding | +|---|---------| +| 1 | `query_sensor_data` returns the sensor with **no samples** (a comment says so), ignores the window and the limit; `query_sensors_by_labels` is a `bail!`; `list_series` ignores filter, limit and bookmark and runs one labels query per sensor | +| 2 | Floats and locations are written as `f32` (the table says FLOAT64): precision loss | +| 3 | JSON values are written as `value.as_str().unwrap_or("")`: anything but a JSON string becomes `""` | +| 4 | `jobs.query` waits 10 s then answers `jobComplete=false`; `ResultSet::new_from_query_response` turns that into **zero rows**, silently. Only the first page (10 MB) of a result is read | +| 5 | The answer of the Storage Write API is checked for `row_errors` only, not for the `error` of the response; an append that is rejected is reported as a success | +| 6 | Every append takes the **write** lock of the whole client: all writes of the process are serialized | +| 7 | Sensor ids come from sinteflake after a read: two instances registering a sensor at once get two ids and duplicate rows. Caches are process-global statics, shared by every dataset | +| 8 | SQL built with `format!` and the sensor UUID of the caller in it (`WHERE s.uuid = '{}'`) | +| 9 | Timestamps are sent as ISO strings (nanoseconds are not valid for TIMESTAMP) | +| 10 | Dictionaries for label names, label values and strings: 3 more tables, a join per read, a lookup per write | + +## Done (3 Oct 2026) + +- [x] Id helpers shared with ClickHouse in `storage::common` +- [x] New schema and writes: ids from the data, no dictionaries, microsecond timestamps, doubles, NUMERIC as + text, checked answers, no global write lock, JSON as JSON +- [x] `client.rs`: waits for jobs, reads all pages, parameters, `maximum_bytes_billed`, error sorting +- [x] Reads: sensors and labels, 8 sample types, paginated listing, matchers, bulk selectors, latest sample, + deletes; aggregation in BigQuery (`GROUP BY` of buckets, single series and selectors), checked in the + integration tests against the local reference of `apply_query_options` +- [x] Unit tests (SQL builders, rows against the migration, connection string, errors): 27, no credentials needed +- [x] Integration tests: `tests/integration/bigquery_integration.rs` (round trip of every type, a unit and a + window, concurrent instances, paging, Prometheus matcher semantics, a result over one page, the cost + cap), BigQuery added to `data_lifecycle`, `advanced_backend_queries`, `deduplication` +- [x] Docs (`docs/BIGQUERY.md` with the Google Cloud setup), CI job `bigquery-checks`, `test-bigquery-live` +- [x] Dependencies: `gcp-bigquery-client` 0.28, `tonic`, `prost` were already the latest; `bigdecimal`, its + encoder, `clru`, `tonic` and `sinteflake` (with `SENSAPP_INSTANCE_ID`) are gone + +## Live run (4 Oct 2026, project `smartbuildinghub`, dataset in europe-north1, user credentials) + +All the integration tests that apply to BigQuery pass: the 9 of `bigquery_integration`, the 42 of the +`data_lifecycle`, `advanced_backend_queries`, `query_sensors_by_labels`, `selector_reads`, +`selector_aggregated`, `aggregated_windows` and `cross_series_reads` modules, and the 242 others (arrow, +ingestion, InfluxDB, Prometheus remote read and write, PromQL, JWT, exports, CRUD, publish robustness...). +Measured: 16 901 statements (3 415 `DELETE`, 9 639 `SELECT`), **25 GiB billed**, for these runs and the +partial or repeated ones. Time: 143 s for the 9, 1 300 s for the 42, 1 705 s for the 242. + +Found by the first live run and fixed (`994221d`): + +| # | Finding | +|---|---------| +| 1 | The gRPC client of the Storage Write API panicked: no default rustls provider when `ring` and `aws-lc-rs` are both compiled in (the server installs one, a test does not). `connect` installs it | +| 2 | `AVG` of BigQuery is not the exact sum over the count in the last bits (-5.5e-17 for integers that average to 0). Kept: it is faster, and the difference does not matter in a warehouse. The test compares with a tolerance | +| 3 | Rows still in the streaming buffer are billed 0 bytes, so `max_bytes_billed` cannot be tested on them: the test uses the metadata statement, which bills 10 MB at least | +| 4 | `TRUNCATE` is not faster than `DELETE ... WHERE TRUE` (about 2.5 s a statement either way), kept `DELETE` | + +Not a finding but worth knowing: doubles are exact on `gcp-bigquery-client` 0.28 (`0.1 + 0.2` and `f64::MAX` +round trip). The `f32` of the first backend was the `Float64` descriptor of 0.22 declared as a protobuf `float` +(lquerel/gcp-bigquery-client#106), fixed since 0.26. + +Not measured: latency of a small write and of a read in isolation, behaviour under real load, the Storage +Write API quotas. diff --git a/ideas/bigquery-backend-reconciliation.md b/ideas/bigquery-backend-reconciliation.md deleted file mode 100644 index 4458fabb..00000000 --- a/ideas/bigquery-backend-reconciliation.md +++ /dev/null @@ -1,31 +0,0 @@ -# Port the Local BigQuery Read Work - -## Source to preserve - -`/Users/antoinep/work/sensapp-vibe-prom-read` contains `storage-update` commit -`31d7a3d` and uncommitted changes in `src/storage/bigquery/`. It also contains -`tests/bigquery_integration.rs`. Do not reset or delete that checkout before the -port is complete. - -## Why a focused port is needed - -Current `main` uses paginated `list_series` and advanced query methods that the -old branch did not implement. BigQuery is currently marked experimental/deferred -in `TODO.md`, and the CI environment has no BigQuery credentials. Copying the -older module directly would replace newer interfaces and leave live behavior -unverified. - -## Proposed work - -1. Adapt the old matcher and typed sample queries to the current storage trait, - pagination contract, and current `gcp-bigquery-client` APIs. -2. Review the local identifier-validation patch against real Google Cloud project - and dataset naming rules before adopting it. Keep all dynamic values - parameterized; validate identifiers used in SQL. -3. Review bounded cache expiry and parallel queries for correctness under - concurrent writes, not just speed. -4. Port the old integration tests, add pagination and error-path cases, and run - them against an isolated real BigQuery dataset. Keep compile checks in CI even - when live credentials are unavailable. - -This is separate from the first ClickHouse pre-production release gate. diff --git a/settings.toml b/settings.toml index 59134069..8b2f5a24 100644 --- a/settings.toml +++ b/settings.toml @@ -6,7 +6,7 @@ storage_connection_string = "postgres://postgres:postgres@localhost:5432/sensapp # Alternative storage backends (uncomment one to use): # storage_connection_string = "sqlite://sensapp.db" # storage_connection_string = "duckdb://sensapp.db" -# storage_connection_string = "bigquery://key.json?project_id=PROJECT&dataset_id=DATASET" +# storage_connection_string = "bigquery://key.json?project_id=PROJECT&dataset_id=DATASET" # billed queries, see docs/BIGQUERY.md # storage_connection_string = "rrdcached://localhost:42217?preset=munin" # Network Configuration diff --git a/src/config/mod.rs b/src/config/mod.rs index 3d9700c9..f618ec85 100644 --- a/src/config/mod.rs +++ b/src/config/mod.rs @@ -9,9 +9,6 @@ const MAX_HTTP_BODY_LIMIT_BYTES: u64 = 128 * 1024 * 1024 * 1024; #[derive(Debug, Config)] pub struct SensAppConfig { - #[config(env = "SENSAPP_INSTANCE_ID", default = 0)] - pub instance_id: u16, - #[config(env = "SENSAPP_PORT", default = 3000)] pub port: u16, #[config(env = "SENSAPP_ENDPOINT", default = "127.0.0.1")] diff --git a/src/datamodel/typed_samples.rs b/src/datamodel/typed_samples.rs index 0ffa463c..0c38f30d 100644 --- a/src/datamodel/typed_samples.rs +++ b/src/datamodel/typed_samples.rs @@ -52,6 +52,20 @@ impl TypedSamples { } } + /// Keep the first `len` samples. + pub fn truncate(&mut self, len: usize) { + match self { + TypedSamples::Integer(vec) => vec.truncate(len), + TypedSamples::Numeric(vec) => vec.truncate(len), + TypedSamples::Float(vec) => vec.truncate(len), + TypedSamples::String(vec) => vec.truncate(len), + TypedSamples::Boolean(vec) => vec.truncate(len), + TypedSamples::Location(vec) => vec.truncate(len), + TypedSamples::Blob(vec) => vec.truncate(len), + TypedSamples::Json(vec) => vec.truncate(len), + } + } + pub fn is_empty(&self) -> bool { self.len() == 0 } diff --git a/src/main.rs b/src/main.rs index 4b924420..8f6535a9 100644 --- a/src/main.rs +++ b/src/main.rs @@ -55,11 +55,6 @@ async fn async_main() -> Result<()> { )) }); - sinteflake::set_instance_id(config.instance_id).context("Failed to set instance ID")?; - sinteflake::set_instance_id_async(config.instance_id) - .await - .context("Failed to set async instance ID")?; - // Initialize storage backend println!("🗄️ Connecting to storage..."); let storage = create_storage_from_connection_string(&config.storage_connection_string) diff --git a/src/storage/bigquery/README.md b/src/storage/bigquery/README.md index a89b2551..8d553672 100644 --- a/src/storage/bigquery/README.md +++ b/src/storage/bigquery/README.md @@ -1,26 +1,26 @@ -# BigQuery and SensApp - -We are usually not a fan of proprietary solutions, but BigQuery is worth a try. - -Infortunately, while it uses some SQL flavour to query data, it's not very standard nor straightforward to use. It has vendor lock-in. - -As the summer of 2024, it seems that we have many options to upload data to BigQuery: - -- A JSON/REST API. We upload JSON (potentially compressed with Gzip). -- A Storage Write API that requires some protocol buffer binary and schema over gRPC. -- Upload a static file to Google Cloud Storage and then load it into BigQuery as a job. -- Use some big data pipeline tool like Apache Beam or Apache Spark that can then be connected to BigQuery. - -The best option for now seems to be the Storage Write API with the `gcp-bigquery-client` crate. - -## No denormalisation - -BigQuery may benefit from data denormalisation, but we aren't doing it for now. It would be too different when comparing against the other databases. I'm also not convinced about the benefits of denormalisation for SensApp. - -## Transactions - -As far as I understood, BigQuery does not support transactions when ingesting data. - -## No sequential IDs - -BigQuery does not support sequential IDs. Querying the database to compute the next ID is not an option, as it's too slow and costly. Therefore, we use [sinteflake](https://crates.io/crates/sinteflake), a distributed unique ID generator inspired by Twitter's Snowflake. +# BigQuery backend + +User documentation: [docs/BIGQUERY.md](../../../docs/BIGQUERY.md) (connection string, differences, Google Cloud setup, +tests). Notes for whoever changes the code: + +| File | What | +|---|---| +| `connection.rs` | The connection string and the validation of the identifiers that end up in backticks | +| `client.rs` | Parameters, `run_query` (waits for the job, reads every page, DML counts), the sorting of errors | +| `rows.rs` | The protobuf rows of the Storage Write API and their descriptors, checked against `migrations/init.sql` by a test | +| `publishers.rs` | Registration of new series and the write of a batch (one append per table, answers checked) | +| `reads.rs` | Sensors, labels and samples, the SQL of the sets that collapse duplicate registrations | +| `matchers.rs` | Label matchers as parameterized SQL | +| `selector.rs` | `BulkSelectorBackend`: a selector with a few queries | +| `mod.rs` | The `StorageInstance` implementation | + +Rules the code relies on: + +- **Never put a value in a statement**: values go in `@parameters`. Only the project, dataset and table names + (validated, quoted) and numbers (`LIMIT`) are formatted into SQL. +- **Registration is idempotent**: ids come from the data, so concurrent writers insert identical rows. Reads of + `sensors`, `units` and `labels` must collapse duplicates (`GROUP BY`, `DISTINCT`), see `Dataset::sensors_set`. +- **A response of the Storage Write API is not a success until its `error` and `row_errors` are empty.** +- **A result is not read until the job is complete and every page is read** (`ResultSet::new_from_query_response` + returns zero rows for an unfinished job). +- The ids and the types of the columns are an on-disk format, like ClickHouse's. diff --git a/src/storage/bigquery/aggregation.rs b/src/storage/bigquery/aggregation.rs new file mode 100644 index 00000000..d8fa5a37 --- /dev/null +++ b/src/storage/bigquery/aggregation.rs @@ -0,0 +1,241 @@ +//! Aggregation in BigQuery: the buckets of `step` starting at the beginning of the window (at 0 without +//! one), as in the other backends and in `storage::common::apply_query_options`, which is the +//! reference the integration tests compare with. + +use super::BigQueryStorage; +use super::client::{int_array_param, int_param}; +use super::reads::{read_sample, sample_table, window}; +use crate::datamodel::{SensorType, TypedSamples}; +use crate::storage::Aggregation; +use crate::storage::selector::{AggregatedKind, AggregatedRead, aggregated_kind, empty_samples}; +use anyhow::{Context, Result, bail}; +use gcp_bigquery_client::model::query_parameter::QueryParameter; +use std::collections::HashMap; + +/// The aggregate of a bucket. `First` and `Last` take the value of the oldest and the newest sample. +/// +/// `AVG` is BigQuery's own: it is faster than a sum over a count, and floating point averages differ +/// from the exact one in the last bits (`-5.5e-17` for integers that average to 0). +fn value_expression(aggregation: Aggregation) -> &'static str { + match aggregation { + Aggregation::Avg => "AVG(value)", + Aggregation::Min => "MIN(value)", + Aggregation::Max => "MAX(value)", + Aggregation::Sum => "SUM(value)", + Aggregation::Count => "COUNT(*)", + Aggregation::First => "ANY_VALUE(value HAVING MIN timestamp)", + Aggregation::Last => "ANY_VALUE(value HAVING MAX timestamp)", + } +} + +/// The start of the bucket of a sample, in microseconds. A floor division, so that a sample before the +/// origin falls in the bucket before it, as `bucket_start` does (BigQuery's `DIV` truncates). +const BUCKET_EXPRESSION: &str = "@origin + (DIV(UNIX_MICROS(timestamp) - @origin, @step) \ + - IF(MOD(UNIX_MICROS(timestamp) - @origin, @step) < 0, 1, 0)) * @step"; + +pub fn kind_type(kind: AggregatedKind) -> SensorType { + match kind { + AggregatedKind::Integer => SensorType::Integer, + AggregatedKind::Float => SensorType::Float, + AggregatedKind::Numeric => SensorType::Numeric, + } +} + +/// The statement that aggregates the series of `@ids` (a numeric type), and its parameters other +/// than the ids: one row `sensor_id, bucket_us, value` per series and bucket, at most `limit`. +pub fn aggregated_sql( + table: &str, + sensor_type: SensorType, + read: &AggregatedRead, + limit: usize, +) -> Result<(String, Vec)> { + if !matches!( + sensor_type, + SensorType::Integer | SensorType::Numeric | SensorType::Float + ) { + bail!("{sensor_type} is not a numeric type"); + } + let step_us = read + .step_ms + .checked_mul(1000) + .context("step is too large")?; + let mut params = vec![ + int_param("origin", read.start_us.unwrap_or(0)), + int_param("step", step_us), + ]; + let conditions = window(read.start_us, read.end_us, &mut params); + let sql = format!( + "SELECT sensor_id, {BUCKET_EXPRESSION} AS bucket_us, {} AS value FROM {table} \ + WHERE sensor_id IN UNNEST(@ids){conditions} \ + GROUP BY sensor_id, bucket_us ORDER BY sensor_id, bucket_us LIMIT {}", + value_expression(read.aggregation), + limit.min(super::reads::MAX_LIMIT), + ); + Ok((sql, params)) +} + +impl BigQueryStorage { + /// The buckets of many series of one numeric type, with one statement, at most `limit` in total. + pub(super) async fn query_aggregated_of_many( + &self, + sensor_type: SensorType, + ids: &[i64], + read: &AggregatedRead, + limit: usize, + ) -> Result> { + let (sql, mut params) = aggregated_sql( + &self.table(sample_table(sensor_type)), + sensor_type, + read, + limit, + )?; + params.push(int_array_param("ids", ids)); + let kind = kind_type(aggregated_kind(sensor_type, read.aggregation)); + + let rows = self + .query_rows("aggregate samples", sql, params, |row| { + let sensor_id = super::client::required(row.get_i64(0), "sensor_id")?; + let mut bucket = empty_samples(kind); + read_sample(&mut bucket, row)?; + Ok((sensor_id, bucket)) + }) + .await?; + let mut by_sensor: HashMap = HashMap::new(); + for (sensor_id, bucket) in rows { + super::reads::append_samples( + by_sensor + .entry(sensor_id) + .or_insert_with(|| empty_samples(kind)), + bucket, + ); + } + Ok(by_sensor) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn read( + aggregation: Aggregation, + start_us: Option, + end_us: Option, + ) -> AggregatedRead { + AggregatedRead { + start_us, + end_us, + step_ms: 21_600_000, + aggregation, + } + } + + #[test] + fn six_hour_buckets_are_computed_in_the_database_with_parameters() { + let (sql, params) = aggregated_sql( + "`p.d.float_values`", + SensorType::Float, + &read(Aggregation::Avg, Some(1_000), Some(2_000)), + 500, + ) + .unwrap(); + assert!( + sql.contains("AVG(value) AS value FROM `p.d.float_values`"), + "{sql}" + ); + assert!(sql.contains("GROUP BY sensor_id, bucket_us"), "{sql}"); + assert!( + sql.contains("timestamp >= TIMESTAMP_MICROS(@start)"), + "{sql}" + ); + assert!(sql.contains("timestamp <= TIMESTAMP_MICROS(@end)"), "{sql}"); + assert!( + sql.ends_with("ORDER BY sensor_id, bucket_us LIMIT 500"), + "{sql}" + ); + let values: Vec<_> = params + .iter() + .map(|p| { + ( + p.name.clone().unwrap(), + p.parameter_value.as_ref().unwrap().value.clone().unwrap(), + ) + }) + .collect(); + assert_eq!( + values, + vec![ + ("origin".to_string(), "1000".to_string()), + ("step".to_string(), "21600000000".to_string()), + ("start".to_string(), "1000".to_string()), + ("end".to_string(), "2000".to_string()), + ] + ); + } + + #[test] + fn without_a_window_the_buckets_start_at_zero() { + let (sql, params) = aggregated_sql( + "`t`", + SensorType::Integer, + &read(Aggregation::Count, None, None), + 10, + ) + .unwrap(); + assert!(!sql.contains("@start") && !sql.contains("@end")); + assert!(sql.contains("COUNT(*) AS value")); + assert_eq!( + params[0].parameter_value.as_ref().unwrap().value.as_deref(), + Some("0") + ); + } + + #[test] + fn every_aggregation_has_an_expression() { + for (aggregation, expected) in [ + (Aggregation::Avg, "AVG(value)"), + (Aggregation::Min, "MIN(value)"), + (Aggregation::Max, "MAX(value)"), + (Aggregation::Sum, "SUM(value)"), + (Aggregation::Count, "COUNT(*)"), + (Aggregation::First, "HAVING MIN timestamp"), + (Aggregation::Last, "HAVING MAX timestamp"), + ] { + assert!(value_expression(aggregation).contains(expected)); + } + } + + #[test] + fn only_numeric_series_are_aggregated() { + assert!( + aggregated_sql( + "`t`", + SensorType::String, + &read(Aggregation::Count, None, None), + 1 + ) + .is_err() + ); + let huge = AggregatedRead { + step_ms: i64::MAX, + ..read(Aggregation::Avg, None, None) + }; + assert!(aggregated_sql("`t`", SensorType::Float, &huge, 1).is_err()); + } + + #[test] + fn the_type_of_the_buckets_follows_the_shared_rule() { + assert_eq!( + kind_type(aggregated_kind(SensorType::Integer, Aggregation::Avg)), + SensorType::Float + ); + assert_eq!( + kind_type(aggregated_kind(SensorType::Float, Aggregation::Count)), + SensorType::Integer + ); + assert_eq!( + kind_type(aggregated_kind(SensorType::Numeric, Aggregation::Sum)), + SensorType::Numeric + ); + } +} diff --git a/src/storage/bigquery/bigquery_labels_utilities.rs b/src/storage/bigquery/bigquery_labels_utilities.rs deleted file mode 100644 index 60a3db5a..00000000 --- a/src/storage/bigquery/bigquery_labels_utilities.rs +++ /dev/null @@ -1,380 +0,0 @@ -use std::{collections::HashSet, num::NonZeroUsize}; - -use crate::storage::StorageError; -use anyhow::Result; -use clru::CLruCache; -use gcp_bigquery_client::model::{ - query_parameter::QueryParameter, query_parameter_type::QueryParameterType, - query_parameter_value::QueryParameterValue, query_request::QueryRequest, - query_response::ResultSet, -}; -use hybridmap::HybridMap; -use once_cell::sync::Lazy; -use tokio::sync::Mutex; -use tracing::{debug, info}; - -use crate::datamodel::SensAppVec; - -use super::{ - BigQueryStorage, - bigquery_prost_structs::{ - LabelDescriptionDictionary as ProstLabelDescriptionDictionary, - LabelNameDictionary as ProstLabelNameDictionary, - }, - bigquery_table_descriptors::{ - LABELS_DESCRIPTION_DICTIONARY_DESCRIPTOR, LABELS_NAME_DICTIONARY_DESCRIPTOR, - }, - bigquery_utilities::publish_rows, -}; - -static LABELS_NAME_CACHE: Lazy>> = - Lazy::new(|| Mutex::new(CLruCache::new(NonZeroUsize::new(16384).unwrap()))); - -pub async fn get_or_create_labels_name_ids( - bqs: &BigQueryStorage, - labels: SensAppVec, -) -> Result> { - if labels.is_empty() { - return Ok(HybridMap::new()); - } - - let mut unknown_labels: HashSet = HashSet::new(); - let mut result = HybridMap::new(); - - { - let mut cache_guard = LABELS_NAME_CACHE.lock().await; - for label in labels.into_iter() { - match cache_guard.get(&label) { - Some(id) => { - result.insert(label, *id); - } - None => { - unknown_labels.insert(label); - } - }; - } - } - - debug!("BigQuery: Found {} known label names", result.len()); - debug!( - "BigQuery: Found {} unknown label names", - unknown_labels.len() - ); - - if unknown_labels.is_empty() { - return Ok(result); - } - - let just_the_labels = unknown_labels - .iter() - .cloned() - .collect::>(); - - let found_ids = get_existing_labels_name_ids(bqs, &just_the_labels).await?; - { - let mut cache_guard = LABELS_NAME_CACHE.lock().await; - for (label, id) in found_ids.iter() { - cache_guard.put(label.clone(), *id); - result.insert(label.clone(), *id); - } - } - - let labels_to_create = unknown_labels - .into_iter() - .filter(|label| found_ids.get(label).is_none()) - .collect::>(); - - if labels_to_create.is_empty() { - return Ok(result); - } - - info!( - "BigQuery: Creating {} new label names", - labels_to_create.len() - ); - - let new_ids = create_labels_name(bqs, labels_to_create).await?; - { - let mut cache_guard = LABELS_NAME_CACHE.lock().await; - for (label, id) in new_ids.iter() { - cache_guard.put(label.clone(), *id); - result.insert(label.clone(), *id); - } - } - - Ok(result) -} - -async fn get_existing_labels_name_ids( - bqs: &BigQueryStorage, - labels: &[String], -) -> Result> { - let mut query_request = QueryRequest::new( - r#" - SELECT id, name - FROM `{dataset_id}.labels_name_dictionary` - WHERE name IN UNNEST(@names) - "# - .replace("{dataset_id}", bqs.dataset_id()), - ); - - let query_parameter = QueryParameter { - name: Some("names".to_string()), - parameter_type: Some(QueryParameterType { - r#type: "ARRAY".to_string(), - struct_types: None, - array_type: Some(Box::new(QueryParameterType { - r#type: "STRING".to_string(), - struct_types: None, - array_type: None, - })), - }), - parameter_value: Some(QueryParameterValue { - value: None, - struct_values: None, - array_values: Some( - labels - .iter() - .map(|label| QueryParameterValue { - value: Some(label.clone()), - struct_values: None, - array_values: None, - }) - .collect(), - ), - }), - }; - query_request.query_parameters = Some(vec![query_parameter]); - - let result = bqs - .client() - .read() - .await - .job() - .query(bqs.project_id(), query_request) - .await?; - let mut result = ResultSet::new_from_query_response(result); - - let mut results_map = HybridMap::with_capacity(result.row_count()); - - while result.next_row() { - let id = result.get_i64(0)?.ok_or_else(|| { - anyhow::Error::from(StorageError::missing_field("label_name_id", None, None)) - })?; - let name = result.get_string(1)?.ok_or_else(|| { - anyhow::Error::from(StorageError::missing_field("label_name", None, None)) - })?; - - results_map.insert(name, id); - } - - Ok(results_map) -} - -async fn create_labels_name( - bqs: &BigQueryStorage, - labels: SensAppVec, -) -> Result> { - let mut map = HybridMap::with_capacity(labels.len()); - - for label in labels.as_ref() { - let id = sinteflake::next_id_with_hash_async(label.as_bytes()).await? as i64; - map.insert(label.clone(), id); - } - - let rows = labels - .into_iter() - .map(|label| ProstLabelNameDictionary { - id: *map - .get(&label) - .expect("Internal consistency error: Label missing from cached map"), - name: label, - }) - .collect::>(); - - publish_rows( - bqs, - "labels_name_dictionary", - &LABELS_NAME_DICTIONARY_DESCRIPTOR, - rows, - ) - .await?; - - Ok(map) -} - -static LABELS_DESCRIPTION_CACHE: Lazy>> = - Lazy::new(|| Mutex::new(CLruCache::new(NonZeroUsize::new(16384).unwrap()))); - -pub async fn get_or_create_labels_description_ids( - bqs: &BigQueryStorage, - labels: SensAppVec, -) -> Result> { - if labels.is_empty() { - return Ok(HybridMap::new()); - } - - let mut unknown_labels: HashSet = HashSet::new(); - let mut result = HybridMap::new(); - - { - let mut cache_guard = LABELS_DESCRIPTION_CACHE.lock().await; - for label in labels.into_iter() { - match cache_guard.get(&label) { - Some(id) => { - result.insert(label, *id); - } - None => { - unknown_labels.insert(label); - } - }; - } - } - - debug!("BigQuery: Found {} known label descriptions", result.len()); - debug!( - "BigQuery: Found {} unknown label descriptions", - unknown_labels.len() - ); - - if unknown_labels.is_empty() { - return Ok(result); - } - - let just_the_labels = unknown_labels - .iter() - .cloned() - .collect::>(); - - let found_ids = get_existing_labels_description_ids(bqs, &just_the_labels).await?; - { - let mut cache_guard = LABELS_DESCRIPTION_CACHE.lock().await; - for (label, id) in found_ids.iter() { - cache_guard.put(label.clone(), *id); - result.insert(label.clone(), *id); - } - } - - let labels_to_create = unknown_labels - .into_iter() - .filter(|label| found_ids.get(label).is_none()) - .collect::>(); - - if labels_to_create.is_empty() { - return Ok(result); - } - - info!( - "BigQuery: Creating {} new label descriptions", - labels_to_create.len() - ); - - let new_ids = create_labels_description(bqs, labels_to_create).await?; - { - let mut cache_guard = LABELS_DESCRIPTION_CACHE.lock().await; - for (label, id) in new_ids.iter() { - cache_guard.put(label.clone(), *id); - result.insert(label.clone(), *id); - } - } - - Ok(result) -} - -async fn get_existing_labels_description_ids( - bqs: &BigQueryStorage, - labels: &[String], -) -> Result> { - let mut query_request = QueryRequest::new( - r#" - SELECT id, description - FROM `{dataset_id}.labels_description_dictionary` - WHERE description IN UNNEST(@descriptions) - "# - .replace("{dataset_id}", bqs.dataset_id()), - ); - - let query_parameter = QueryParameter { - name: Some("descriptions".to_string()), - parameter_type: Some(QueryParameterType { - r#type: "ARRAY".to_string(), - struct_types: None, - array_type: Some(Box::new(QueryParameterType { - r#type: "STRING".to_string(), - struct_types: None, - array_type: None, - })), - }), - parameter_value: Some(QueryParameterValue { - value: None, - struct_values: None, - array_values: Some( - labels - .iter() - .map(|label| QueryParameterValue { - value: Some(label.clone()), - struct_values: None, - array_values: None, - }) - .collect(), - ), - }), - }; - query_request.query_parameters = Some(vec![query_parameter]); - - let result = bqs - .client() - .read() - .await - .job() - .query(bqs.project_id(), query_request) - .await?; - let mut result = ResultSet::new_from_query_response(result); - - let mut results_map = HybridMap::with_capacity(result.row_count()); - - while result.next_row() { - let id = result.get_i64(0)?.ok_or_else(|| { - anyhow::Error::from(StorageError::missing_field("label_name_id", None, None)) - })?; - let description = result.get_string(1)?.ok_or_else(|| { - anyhow::Error::from(StorageError::missing_field("label_description", None, None)) - })?; - - results_map.insert(description, id); - } - - Ok(results_map) -} - -async fn create_labels_description( - bqs: &BigQueryStorage, - labels: SensAppVec, -) -> Result> { - let mut map = HybridMap::with_capacity(labels.len()); - - for label in labels.as_ref() { - let id = sinteflake::next_id_with_hash_async(label.as_bytes()).await? as i64; - map.insert(label.clone(), id); - } - - let rows = labels - .into_iter() - .map(|label| ProstLabelDescriptionDictionary { - id: *map - .get(&label) - .expect("Internal consistency error: Label missing from cached map"), - description: label, - }) - .collect::>(); - - publish_rows( - bqs, - "labels_description_dictionary", - &LABELS_DESCRIPTION_DICTIONARY_DESCRIPTOR, - rows, - ) - .await?; - - Ok(map) -} diff --git a/src/storage/bigquery/bigquery_prost_structs.rs b/src/storage/bigquery/bigquery_prost_structs.rs deleted file mode 100644 index 6d563e53..00000000 --- a/src/storage/bigquery/bigquery_prost_structs.rs +++ /dev/null @@ -1,165 +0,0 @@ -use prost::Message; - -// Units table -#[derive(Clone, PartialEq, Message)] -pub struct Unit { - #[prost(int64, required, tag = "1")] - pub id: i64, - #[prost(string, required, tag = "2")] - pub name: String, - #[prost(string, optional, tag = "3")] - pub description: Option, -} - -// Sensors table -#[derive(Clone, PartialEq, Message)] -pub struct Sensor { - #[prost(int64, required, tag = "1")] - pub sensor_id: i64, - #[prost(string, required, tag = "2")] - pub uuid: String, - #[prost(string, required, tag = "3")] - pub name: String, - #[prost(string, required, tag = "4")] - pub r#type: String, - #[prost(int64, optional, tag = "5")] - pub unit: Option, -} - -// Labels name dictionary table -#[derive(Clone, PartialEq, Message)] -pub struct LabelNameDictionary { - #[prost(int64, required, tag = "1")] - pub id: i64, - #[prost(string, required, tag = "2")] - pub name: String, -} - -// Labels description dictionary table -#[derive(Clone, PartialEq, Message)] -pub struct LabelDescriptionDictionary { - #[prost(int64, required, tag = "1")] - pub id: i64, - #[prost(string, required, tag = "2")] - pub description: String, -} - -// Labels table -#[derive(Clone, PartialEq, Message)] -pub struct Label { - #[prost(int64, required, tag = "1")] - pub sensor_id: i64, - #[prost(int64, required, tag = "2")] - pub name: i64, - #[prost(int64, optional, tag = "3")] - pub description: Option, -} - -// Strings values dictionary table -#[derive(Clone, PartialEq, Message)] -pub struct StringValueDictionary { - #[prost(int64, required, tag = "1")] - pub id: i64, - #[prost(string, required, tag = "2")] - pub value: String, -} - -// Integer values table -#[derive(Clone, PartialEq, Message)] -pub struct IntegerValue { - #[prost(int64, required, tag = "1")] - pub sensor_id: i64, - #[prost(string, required, tag = "2")] - pub timestamp: String, - #[prost(int64, required, tag = "3")] - pub value: i64, -} - -// Numeric values table -#[derive(Clone, PartialEq, Message)] -pub struct NumericValue { - #[prost(int64, required, tag = "1")] - pub sensor_id: i64, - #[prost(string, required, tag = "2")] - pub timestamp: String, - #[prost(bytes, required, tag = "3")] - pub value: Vec, -} - -// Float values table -#[derive(Clone, PartialEq, Message)] -pub struct FloatValue { - #[prost(int64, required, tag = "1")] - pub sensor_id: i64, - #[prost(string, required, tag = "2")] - pub timestamp: String, - // /!\ This is currently *NOT* working! Only NULL are inserted instead. - // See https://github.com/lquerel/gcp-bigquery-client/issues/106 - //#[prost(double, tag = "3")] - //pub value: f64, - #[prost(float, required, tag = "3")] - pub value: f32, // reverting back to f32 for now, decimal/numeric value is heavily recommended instead -} - -// String values table -#[derive(Clone, PartialEq, Message)] -pub struct StringValue { - #[prost(int64, required, tag = "1")] - pub sensor_id: i64, - #[prost(string, required, tag = "2")] - pub timestamp: String, - #[prost(int64, required, tag = "3")] - pub value: i64, -} - -// Boolean values table -#[derive(Clone, PartialEq, Message)] -pub struct BooleanValue { - #[prost(int64, required, tag = "1")] - pub sensor_id: i64, - #[prost(string, required, tag = "2")] - pub timestamp: String, - #[prost(bool, required, tag = "3")] - pub value: bool, -} - -// Location values table -#[derive(Clone, PartialEq, Message)] -pub struct LocationValue { - #[prost(int64, required, tag = "1")] - pub sensor_id: i64, - #[prost(string, required, tag = "2")] - pub timestamp: String, - // Like for FloatValue, this is currently *NOT* working! Only NULL are inserted instead. - // We use f32 for now, which is more than enough for GPS coordinates by the way. - //#[prost(double, required, tag = "3")] - //pub latitude: f64, - //#[prost(double, required, tag = "4")] - //pub longitude: f64, - #[prost(float, required, tag = "3")] - pub latitude: f32, - #[prost(float, required, tag = "4")] - pub longitude: f32, -} - -// JSON values table -#[derive(Clone, PartialEq, Message)] -pub struct JsonValue { - #[prost(int64, required, tag = "1")] - pub sensor_id: i64, - #[prost(string, required, tag = "2")] - pub timestamp: String, - #[prost(string, required, tag = "3")] - pub value: String, // Using String to represent JSON -} - -// Blob values table -#[derive(Clone, PartialEq, Message)] -pub struct BlobValue { - #[prost(int64, required, tag = "1")] - pub sensor_id: i64, - #[prost(string, required, tag = "2")] - pub timestamp: String, - #[prost(bytes, tag = "3")] - pub value: Vec, -} diff --git a/src/storage/bigquery/bigquery_publishers.rs b/src/storage/bigquery/bigquery_publishers.rs deleted file mode 100644 index 87a25ebe..00000000 --- a/src/storage/bigquery/bigquery_publishers.rs +++ /dev/null @@ -1,326 +0,0 @@ -use anyhow::{Result, anyhow}; -use big_decimal_byte_string_encoder::encode_bigdecimal_to_bigquery_bytes; -use bigdecimal::BigDecimal; -use hybridmap::HybridMap; -use std::sync::Arc; -use tracing::{debug, info}; -use uuid::Uuid; - -use super::BigQueryStorage; -use super::bigquery_prost_structs::{BlobValue, IntegerValue, JsonValue, LocationValue}; -use super::bigquery_table_descriptors::{ - BLOB_VALUES_DESCRIPTOR, INTEGER_VALUES_DESCRIPTOR, JSON_VALUES_DESCRIPTOR, -}; -use super::bigquery_utilities::publish_rows; -use crate::datamodel::SensAppVec; -use crate::datamodel::{SensorType, TypedSamples, batch::Batch}; -use crate::storage::bigquery::bigquery_prost_structs::{ - BooleanValue, FloatValue, NumericValue, StringValue, -}; -use crate::storage::bigquery::bigquery_string_values_utilities::get_or_create_string_values_ids; -use crate::storage::bigquery::bigquery_table_descriptors::{ - BOOLEAN_VALUES_DESCRIPTOR, FLOAT_VALUES_DESCRIPTOR, LOCATION_VALUES_DESCRIPTOR, - NUMERIC_VALUES_DESCRIPTOR, STRING_VALUES_DESCRIPTOR, -}; - -pub async fn publish_integer_values( - bqs: &BigQueryStorage, - batch: Arc, - sensor_ids: Arc>, -) -> Result<()> { - let mut rows = vec![]; - for single_sensor_batch in batch.sensors.as_ref() { - if single_sensor_batch.sensor.sensor_type == SensorType::Integer { - { - let samples_guard = single_sensor_batch.samples.read().await; - if let TypedSamples::Integer(samples) = &*samples_guard { - for value in samples { - let sensor_id = sensor_ids - .get(&single_sensor_batch.sensor.uuid) - .ok_or(anyhow!("Sensor not found"))?; - let timestamp = value.datetime.to_isoformat(); - debug!( - "BigQuery: Publishing integer value with timestamp: {}", - timestamp - ); - rows.push(IntegerValue { - sensor_id: *sensor_id, - timestamp, - value: value.value, - }); - } - } else { - unreachable!("SensorType is Integer, but samples are not"); - } - } - } - } - - publish_rows(bqs, "integer_values", &INTEGER_VALUES_DESCRIPTOR, rows).await -} - -pub async fn publish_numeric_values( - bqs: &BigQueryStorage, - batch: Arc, - sensor_ids: Arc>, -) -> Result<()> { - let mut rows = vec![]; - for single_sensor_batch in batch.sensors.as_ref() { - if single_sensor_batch.sensor.sensor_type == SensorType::Numeric { - { - let samples_guard = single_sensor_batch.samples.read().await; - if let TypedSamples::Numeric(samples) = &*samples_guard { - for value in samples { - let sensor_id = sensor_ids - .get(&single_sensor_batch.sensor.uuid) - .ok_or(anyhow!("Sensor not found"))?; - let timestamp = value.datetime.to_isoformat(); - // Stupid conversion for now - let decimal_string = value.value.to_string(); - use std::str::FromStr; - let decimal_bigdecimal = - BigDecimal::from_str(&decimal_string)?.with_scale(9); - let value = encode_bigdecimal_to_bigquery_bytes(&decimal_bigdecimal)?; - rows.push(NumericValue { - sensor_id: *sensor_id, - timestamp, - value, - }); - } - } else { - unreachable!("SensorType is Numeric, but samples are not"); - } - } - } - } - - publish_rows(bqs, "numeric_values", &NUMERIC_VALUES_DESCRIPTOR, rows).await -} - -pub async fn publish_float_values( - bqs: &BigQueryStorage, - batch: Arc, - sensor_ids: Arc>, -) -> Result<()> { - let mut rows = vec![]; - for single_sensor_batch in batch.sensors.as_ref() { - if single_sensor_batch.sensor.sensor_type == SensorType::Float { - { - let samples_guard = single_sensor_batch.samples.read().await; - if let TypedSamples::Float(samples) = &*samples_guard { - for value in samples { - let sensor_id = sensor_ids - .get(&single_sensor_batch.sensor.uuid) - .ok_or(anyhow!("Sensor not found"))?; - let timestamp = value.datetime.to_isoformat(); - rows.push(FloatValue { - sensor_id: *sensor_id, - timestamp, - value: value.value as f32, // Stupid conversion for now - }); - } - } else { - unreachable!("SensorType is Float, but samples are not"); - } - } - } - } - - publish_rows(bqs, "float_values", &FLOAT_VALUES_DESCRIPTOR, rows).await -} - -pub async fn publish_string_values( - bqs: &BigQueryStorage, - batch: Arc, - sensor_ids: Arc>, -) -> Result<()> { - struct TmpStringValue { - sensor_id: i64, - timestamp: String, - value_string: String, - } - - let mut tmp_rows: SensAppVec = SensAppVec::new(); - - for single_sensor_batch in batch.sensors.as_ref() { - if single_sensor_batch.sensor.sensor_type == SensorType::String { - { - let samples_guard = single_sensor_batch.samples.read().await; - if let TypedSamples::String(samples) = &*samples_guard { - for value in samples { - let sensor_id = sensor_ids - .get(&single_sensor_batch.sensor.uuid) - .ok_or(anyhow!("Sensor not found"))?; - let timestamp = value.datetime.to_isoformat(); - tmp_rows.push(TmpStringValue { - sensor_id: *sensor_id, - timestamp, - value_string: value.value.clone(), - }); - } - } else { - unreachable!("SensorType is String, but samples are not"); - } - } - } - } - - if tmp_rows.is_empty() { - debug!("BigQuery: No string values to publish"); - return Ok(()); - } - - info!("BigQuery: Publishing {} string values", tmp_rows.len()); - - let only_string_values = tmp_rows - .iter() - .map(|row| row.value_string.clone()) - .collect::>(); - - let ids_map = get_or_create_string_values_ids(bqs, only_string_values).await?; - - let rows = tmp_rows - .into_iter() - .map(|row| { - let string_id = ids_map - .get(&row.value_string) - .expect("Internal consistency error: String value missing from cached map"); - StringValue { - sensor_id: row.sensor_id, - timestamp: row.timestamp, - value: *string_id, - } - }) - .collect::>(); - - publish_rows(bqs, "string_values", &STRING_VALUES_DESCRIPTOR, rows).await -} - -pub async fn publish_boolean_values( - bqs: &BigQueryStorage, - batch: Arc, - sensor_ids: Arc>, -) -> Result<()> { - let mut rows = vec![]; - for single_sensor_batch in batch.sensors.as_ref() { - if single_sensor_batch.sensor.sensor_type == SensorType::Boolean { - { - let samples_guard = single_sensor_batch.samples.read().await; - if let TypedSamples::Boolean(samples) = &*samples_guard { - for value in samples { - let sensor_id = sensor_ids - .get(&single_sensor_batch.sensor.uuid) - .ok_or(anyhow!("Sensor not found"))?; - let timestamp = value.datetime.to_isoformat(); - rows.push(BooleanValue { - sensor_id: *sensor_id, - timestamp, - value: value.value, - }); - } - } else { - unreachable!("SensorType is Boolean, but samples are not"); - } - } - } - } - - publish_rows(bqs, "boolean_values", &BOOLEAN_VALUES_DESCRIPTOR, rows).await -} - -pub async fn publish_location_values( - bqs: &BigQueryStorage, - batch: Arc, - sensor_ids: Arc>, -) -> Result<()> { - let mut rows = vec![]; - for single_sensor_batch in batch.sensors.as_ref() { - if single_sensor_batch.sensor.sensor_type == SensorType::Location { - { - let samples_guard = single_sensor_batch.samples.read().await; - if let TypedSamples::Location(samples) = &*samples_guard { - for value in samples { - let sensor_id = sensor_ids - .get(&single_sensor_batch.sensor.uuid) - .ok_or(anyhow!("Sensor not found"))?; - let timestamp = value.datetime.to_isoformat(); - rows.push(LocationValue { - sensor_id: *sensor_id, - timestamp, - latitude: value.value.y() as f32, - longitude: value.value.x() as f32, - }); - } - } else { - unreachable!("SensorType is Location, but samples are not"); - } - } - } - } - - publish_rows(bqs, "location_values", &LOCATION_VALUES_DESCRIPTOR, rows).await -} - -pub async fn publish_json_values( - bqs: &BigQueryStorage, - batch: Arc, - sensor_ids: Arc>, -) -> Result<()> { - let mut rows = vec![]; - for single_sensor_batch in batch.sensors.as_ref() { - if single_sensor_batch.sensor.sensor_type == SensorType::Json { - { - let samples_guard = single_sensor_batch.samples.read().await; - if let TypedSamples::Json(samples) = &*samples_guard { - for value in samples { - let sensor_id = sensor_ids - .get(&single_sensor_batch.sensor.uuid) - .ok_or(anyhow!("Sensor not found"))?; - let timestamp = value.datetime.to_isoformat(); - rows.push(JsonValue { - sensor_id: *sensor_id, - timestamp, - value: value.value.as_str().unwrap_or("").to_string(), - }); - } - } else { - unreachable!("SensorType is Json, but samples are not"); - } - } - } - } - - publish_rows(bqs, "json_values", &JSON_VALUES_DESCRIPTOR, rows).await -} - -pub async fn publish_blob_values( - bqs: &BigQueryStorage, - batch: Arc, - sensor_ids: Arc>, -) -> Result<()> { - let mut rows = vec![]; - for single_sensor_batch in batch.sensors.as_ref() { - if single_sensor_batch.sensor.sensor_type == SensorType::Blob { - { - let samples_guard = single_sensor_batch.samples.read().await; - if let TypedSamples::Blob(samples) = &*samples_guard { - for value in samples { - let sensor_id = sensor_ids - .get(&single_sensor_batch.sensor.uuid) - .ok_or(anyhow!("Sensor not found"))?; - let timestamp = value.datetime.to_isoformat(); - rows.push(BlobValue { - sensor_id: *sensor_id, - timestamp, - value: value.value.clone(), - }); - } - } else { - unreachable!("SensorType is Blob, but samples are not"); - } - } - } - } - - publish_rows(bqs, "blob_values", &BLOB_VALUES_DESCRIPTOR, rows).await -} diff --git a/src/storage/bigquery/bigquery_sensors_utilities.rs b/src/storage/bigquery/bigquery_sensors_utilities.rs deleted file mode 100644 index 0635d3d6..00000000 --- a/src/storage/bigquery/bigquery_sensors_utilities.rs +++ /dev/null @@ -1,262 +0,0 @@ -use crate::storage::StorageError; -use anyhow::Result; -use gcp_bigquery_client::model::{ - query_parameter::QueryParameter, query_parameter_type::QueryParameterType, - query_parameter_value::QueryParameterValue, query_request::QueryRequest, - query_response::ResultSet, -}; -use hybridmap::HybridMap; -use once_cell::sync::Lazy; -use std::sync::Arc; -use tokio::sync::RwLock; -use tracing::{debug, info}; -use uuid::Uuid; - -use super::{ - BigQueryStorage, - bigquery_labels_utilities::{ - get_or_create_labels_description_ids, get_or_create_labels_name_ids, - }, - bigquery_prost_structs::{Label as ProstLabel, Sensor as ProstSensor}, - bigquery_table_descriptors::LABELS_DESCRIPTOR, - bigquery_units_utilities::get_or_create_units_ids, - bigquery_utilities::publish_rows, -}; -use crate::{ - datamodel::{SensAppVec, Sensor, unit::Unit}, - storage::bigquery::bigquery_table_descriptors::SENSORS_DESCRIPTOR, -}; - -// We assume that the sensor ids are stable and never updated from BigQuery. -static SENSOR_ID_CACHE: Lazy>> = - Lazy::new(|| RwLock::new(HybridMap::new())); - -pub async fn get_sensor_ids_or_create_sensors( - bqs: &BigQueryStorage, - sensors: &[Arc], -) -> Result> { - let mut unknown_sensors: SensAppVec> = SensAppVec::new(); - - let mut result = HybridMap::new(); - - { - let sensor_ids_read = SENSOR_ID_CACHE.read().await; - for sensor in sensors { - let uuid = sensor.uuid; - match sensor_ids_read.get(&uuid) { - Some(id) => { - result.insert(uuid, *id); - } - None => { - unknown_sensors.push(sensor.clone()); - } - }; - } - } - - debug!("BigQuery: Found {} known sensors", result.len()); - debug!("BigQuery: Found {} unknown sensors", unknown_sensors.len()); - - if unknown_sensors.is_empty() { - return Ok(result); - } - - let just_the_uuids = unknown_sensors - .iter() - .map(|sensor| sensor.uuid) - .collect::>(); - - let found_ids = get_existing_sensors_ids_from_uuids(bqs, &just_the_uuids).await?; - { - let mut sensor_ids_write = SENSOR_ID_CACHE.write().await; - for (uuid, id) in found_ids.iter() { - sensor_ids_write.insert(*uuid, *id); - result.insert(*uuid, *id); - } - } - - let sensors_to_create = unknown_sensors - .into_iter() - .filter(|sensor| found_ids.get(&sensor.uuid).is_none()) - .collect::>(); - - if sensors_to_create.is_empty() { - return Ok(result); - } - - info!("BigQuery: Creating {} new sensors", sensors_to_create.len()); - - let new_ids = create_sensors(bqs, &sensors_to_create).await?; - { - let mut sensor_ids_write = SENSOR_ID_CACHE.write().await; - for (uuid, id) in new_ids.iter() { - sensor_ids_write.insert(*uuid, *id); - result.insert(*uuid, *id); - } - } - - Ok(result) -} - -async fn get_existing_sensors_ids_from_uuids( - bqs: &BigQueryStorage, - sensor_uuids: &[Uuid], -) -> Result> { - let mut query_request = QueryRequest::new( - r#" - SELECT uuid, sensor_id - FROM `{dataset_id}.sensors` - WHERE uuid IN UNNEST(@sensor_uuids) - "# - .replace("{dataset_id}", bqs.dataset_id()), - ); - - let query_parameter = QueryParameter { - name: Some("sensor_uuids".to_string()), - parameter_type: Some(QueryParameterType { - r#type: "ARRAY".to_string(), - struct_types: None, - array_type: Some(Box::new(QueryParameterType { - r#type: "STRING".to_string(), - struct_types: None, - array_type: None, - })), - }), - parameter_value: Some(QueryParameterValue { - value: None, - struct_values: None, - array_values: Some( - sensor_uuids - .iter() - .map(|uuid| QueryParameterValue { - value: Some(uuid.to_string()), - struct_values: None, - array_values: None, - }) - .collect(), - ), - }), - }; - - query_request.query_parameters = Some(vec![query_parameter]); - - let result = bqs - .client() - .read() - .await - .job() - .query(bqs.project_id(), query_request) - .await?; - let mut result = ResultSet::new_from_query_response(result); - - let mut results_map = HybridMap::with_capacity(result.row_count()); - - while result.next_row() { - let uuid = result - .get_string(0)? - .ok_or_else(|| anyhow::Error::from(StorageError::missing_field("uuid", None, None)))?; - let sensor_id = result.get_i64(1)?.ok_or_else(|| { - let parsed_uuid = Uuid::parse_str(&uuid).ok(); - anyhow::Error::from(StorageError::missing_field("sensor_id", parsed_uuid, None)) - })?; - debug!( - "BigQuery: Found existing sensor {} with id: {}", - uuid, sensor_id - ); - results_map.insert(Uuid::parse_str(&uuid)?, sensor_id); - } - - Ok(results_map) -} - -async fn create_sensors( - bqs: &BigQueryStorage, - sensors: &[Arc], -) -> Result> { - sinteflake::update_time_async().await?; - - let mut map = HybridMap::with_capacity(sensors.len()); - let mut units: SensAppVec = SensAppVec::new(); - let mut labels_names: SensAppVec = SensAppVec::new(); - let mut labels_descriptions: SensAppVec = SensAppVec::new(); - for sensor in sensors { - let sensor_id = sinteflake::next_id_with_hash_async(sensor.uuid.as_bytes()).await? as i64; - map.insert(sensor.uuid, sensor_id); - if let Some(unit) = &sensor.unit { - units.push(unit.clone()); - } - for label in sensor.labels.iter() { - labels_names.push(label.0.clone()); - labels_descriptions.push(label.1.clone()); - } - } - - let units_map = get_or_create_units_ids(bqs, units).await?; - let labels_names_map = get_or_create_labels_name_ids(bqs, labels_names).await?; - let labels_descriptions_map = - get_or_create_labels_description_ids(bqs, labels_descriptions).await?; - - // create the sensors - let rows = sensors - .iter() - .map(|sensor| { - let sensor_id = map.get(&sensor.uuid).ok_or_else(|| { - anyhow::Error::from(StorageError::missing_field( - "sensor_id", - Some(sensor.uuid), - Some(&sensor.name), - )) - })?; - let unit = sensor - .unit - .as_ref() - .and_then(|unit| units_map.get(&unit.name).copied()); - Ok(ProstSensor { - sensor_id: *sensor_id, - uuid: sensor.uuid.to_string(), - name: sensor.name.clone(), - r#type: sensor.sensor_type.to_string(), - unit, - }) - }) - .collect::>>()?; - - publish_rows(bqs, "sensors", &SENSORS_DESCRIPTOR, rows).await?; - - // create the labels - let mut labels_rows = Vec::new(); - for sensor in sensors { - let sensor_id = map.get(&sensor.uuid).ok_or_else(|| { - anyhow::Error::from(StorageError::missing_field( - "sensor_id", - Some(sensor.uuid), - Some(&sensor.name), - )) - })?; - for (name, description) in sensor.labels.iter() { - let name_id = labels_names_map.get(name).ok_or_else(|| { - anyhow::Error::from(StorageError::missing_field( - "label_name_id", - Some(sensor.uuid), - Some(&sensor.name), - )) - })?; - let description_id = labels_descriptions_map.get(description).ok_or_else(|| { - anyhow::Error::from(StorageError::missing_field( - "label_description_id", - Some(sensor.uuid), - Some(&sensor.name), - )) - })?; - labels_rows.push(ProstLabel { - sensor_id: *sensor_id, - name: *name_id, - description: Some(*description_id), - }); - } - } - - publish_rows(bqs, "labels", &LABELS_DESCRIPTOR, labels_rows).await?; - - Ok(map) -} diff --git a/src/storage/bigquery/bigquery_string_values_utilities.rs b/src/storage/bigquery/bigquery_string_values_utilities.rs deleted file mode 100644 index 243957c2..00000000 --- a/src/storage/bigquery/bigquery_string_values_utilities.rs +++ /dev/null @@ -1,195 +0,0 @@ -use std::{collections::HashSet, num::NonZeroUsize}; - -use crate::storage::StorageError; -use anyhow::Result; -use clru::CLruCache; -use gcp_bigquery_client::model::{ - query_parameter::QueryParameter, query_parameter_type::QueryParameterType, - query_parameter_value::QueryParameterValue, query_request::QueryRequest, - query_response::ResultSet, -}; -use hybridmap::HybridMap; -use once_cell::sync::Lazy; -use tokio::sync::Mutex; -use tracing::{debug, info}; - -use crate::datamodel::SensAppVec; - -use super::{ - BigQueryStorage, bigquery_prost_structs::StringValueDictionary, - bigquery_table_descriptors::STRINGS_VALUES_DICTIONARY_DESCRIPTOR, - bigquery_utilities::publish_rows, -}; - -static STRING_VALUES_CACHE: Lazy>> = - Lazy::new(|| Mutex::new(CLruCache::new(NonZeroUsize::new(32768).unwrap()))); - -pub async fn get_or_create_string_values_ids( - bqs: &BigQueryStorage, - strings: SensAppVec, -) -> Result> { - let mut unknown_string_values: HashSet = HashSet::new(); - - let mut result = HybridMap::new(); - - { - let mut cache_guard = STRING_VALUES_CACHE.lock().await; - for string_value in strings.into_iter() { - match cache_guard.get(&string_value) { - Some(id) => { - result.insert(string_value, *id); - } - None => { - unknown_string_values.insert(string_value); - } - }; - } - } - - debug!("BigQuery: Found {} known string values", result.len()); - debug!( - "BigQuery: Found {} unknown string values", - unknown_string_values.len() - ); - - if unknown_string_values.is_empty() { - return Ok(result); - } - - let just_the_values = unknown_string_values - .iter() - .cloned() - .collect::>(); - - let found_ids = get_existing_string_values_ids(bqs, &just_the_values).await?; - { - let mut cache_guard = STRING_VALUES_CACHE.lock().await; - for (value, id) in found_ids.iter() { - cache_guard.put(value.clone(), *id); - result.insert(value.clone(), *id); - } - } - - let values_to_create = unknown_string_values - .into_iter() - .filter(|value| found_ids.get(value).is_none()) - .collect::>(); - - if values_to_create.is_empty() { - return Ok(result); - } - - info!( - "BigQuery: Creating {} new string values", - values_to_create.len() - ); - - let new_ids = create_string_values(bqs, values_to_create).await?; - { - let mut cache_guard = STRING_VALUES_CACHE.lock().await; - for (value, id) in new_ids.into_iter() { - cache_guard.put(value.clone(), id); - result.insert(value, id); - } - } - - Ok(result) -} - -async fn get_existing_string_values_ids( - bqs: &BigQueryStorage, - string_values: &[String], -) -> Result> { - let mut query_request = QueryRequest::new( - r#" - SELECT id, value - FROM `{dataset_id}.strings_values_dictionary` - WHERE value IN UNNEST(@values) - "# - .replace("{dataset_id}", bqs.dataset_id()), - ); - - let query_parameter = QueryParameter { - name: Some("values".to_string()), - parameter_type: Some(QueryParameterType { - r#type: "ARRAY".to_string(), - struct_types: None, - array_type: Some(Box::new(QueryParameterType { - r#type: "STRING".to_string(), - struct_types: None, - array_type: None, - })), - }), - parameter_value: Some(QueryParameterValue { - value: None, - struct_values: None, - array_values: Some( - string_values - .iter() - .map(|string_value| QueryParameterValue { - value: Some(string_value.clone()), - struct_values: None, - array_values: None, - }) - .collect(), - ), - }), - }; - query_request.query_parameters = Some(vec![query_parameter]); - - let result = bqs - .client() - .read() - .await - .job() - .query(bqs.project_id(), query_request) - .await?; - let mut result = ResultSet::new_from_query_response(result); - - let mut results_map = HybridMap::with_capacity(result.row_count()); - - while result.next_row() { - let id = result.get_i64(0)?.ok_or_else(|| { - anyhow::Error::from(StorageError::missing_field("string_value_id", None, None)) - })?; - let value = result.get_string(1)?.ok_or_else(|| { - anyhow::Error::from(StorageError::missing_field("string_value", None, None)) - })?; - - results_map.insert(value, id); - } - - Ok(results_map) -} - -async fn create_string_values( - bqs: &BigQueryStorage, - string_values: SensAppVec, -) -> Result> { - let mut map = HybridMap::with_capacity(string_values.len()); - - for string_value in string_values.as_ref() { - let id = sinteflake::next_id_with_hash_async(string_value.as_bytes()).await? as i64; - map.insert(string_value.clone(), id); - } - - let rows = string_values - .into_iter() - .map(|string_value| StringValueDictionary { - id: *map - .get(&string_value) - .expect("Internal consistency error: String value missing from cached map"), - value: string_value, - }) - .collect::>(); - - publish_rows( - bqs, - "strings_values_dictionary", - &STRINGS_VALUES_DICTIONARY_DESCRIPTOR, - rows, - ) - .await?; - - Ok(map) -} diff --git a/src/storage/bigquery/bigquery_table_descriptors.rs b/src/storage/bigquery/bigquery_table_descriptors.rs deleted file mode 100644 index 2d5876db..00000000 --- a/src/storage/bigquery/bigquery_table_descriptors.rs +++ /dev/null @@ -1,126 +0,0 @@ -use gcp_bigquery_client::storage::{ColumnMode, ColumnType, FieldDescriptor, TableDescriptor}; -use once_cell::sync::Lazy; - -fn field(name: &str, number: u32, typ: ColumnType) -> FieldDescriptor { - FieldDescriptor { - name: name.to_string(), - number, - typ, - mode: ColumnMode::Nullable, - } -} - -pub static UNITS_DESCRIPTOR: Lazy = Lazy::new(|| TableDescriptor { - field_descriptors: vec![ - field("id", 1, ColumnType::Int64), - field("name", 2, ColumnType::String), - field("description", 3, ColumnType::String), - ], -}); - -pub static SENSORS_DESCRIPTOR: Lazy = Lazy::new(|| TableDescriptor { - field_descriptors: vec![ - field("sensor_id", 1, ColumnType::Int64), - field("uuid", 2, ColumnType::String), - field("name", 3, ColumnType::String), - field("type", 4, ColumnType::String), - field("unit", 5, ColumnType::Int64), - ], -}); - -pub static LABELS_NAME_DICTIONARY_DESCRIPTOR: Lazy = - Lazy::new(|| TableDescriptor { - field_descriptors: vec![ - field("id", 1, ColumnType::Int64), - field("name", 2, ColumnType::String), - ], - }); - -pub static LABELS_DESCRIPTION_DICTIONARY_DESCRIPTOR: Lazy = - Lazy::new(|| TableDescriptor { - field_descriptors: vec![ - field("id", 1, ColumnType::Int64), - field("description", 2, ColumnType::String), - ], - }); - -pub static LABELS_DESCRIPTOR: Lazy = Lazy::new(|| TableDescriptor { - field_descriptors: vec![ - field("sensor_id", 1, ColumnType::Int64), - field("name", 2, ColumnType::Int64), - field("description", 3, ColumnType::Int64), - ], -}); - -pub static STRINGS_VALUES_DICTIONARY_DESCRIPTOR: Lazy = - Lazy::new(|| TableDescriptor { - field_descriptors: vec![ - field("id", 1, ColumnType::Int64), - field("value", 2, ColumnType::String), - ], - }); - -pub static INTEGER_VALUES_DESCRIPTOR: Lazy = Lazy::new(|| TableDescriptor { - field_descriptors: vec![ - field("sensor_id", 1, ColumnType::Int64), - field("timestamp", 2, ColumnType::String), - field("value", 3, ColumnType::Int64), - ], -}); - -pub static NUMERIC_VALUES_DESCRIPTOR: Lazy = Lazy::new(|| TableDescriptor { - field_descriptors: vec![ - field("sensor_id", 1, ColumnType::Int64), - field("timestamp", 2, ColumnType::String), - field("value", 3, ColumnType::Bytes), - ], -}); - -pub static FLOAT_VALUES_DESCRIPTOR: Lazy = Lazy::new(|| TableDescriptor { - field_descriptors: vec![ - field("sensor_id", 1, ColumnType::Int64), - field("timestamp", 2, ColumnType::String), - field("value", 3, ColumnType::Double), - ], -}); - -pub static STRING_VALUES_DESCRIPTOR: Lazy = Lazy::new(|| TableDescriptor { - field_descriptors: vec![ - field("sensor_id", 1, ColumnType::Int64), - field("timestamp", 2, ColumnType::String), - field("value", 3, ColumnType::Int64), - ], -}); - -pub static BOOLEAN_VALUES_DESCRIPTOR: Lazy = Lazy::new(|| TableDescriptor { - field_descriptors: vec![ - field("sensor_id", 1, ColumnType::Int64), - field("timestamp", 2, ColumnType::String), - field("value", 3, ColumnType::Bool), - ], -}); - -pub static LOCATION_VALUES_DESCRIPTOR: Lazy = Lazy::new(|| TableDescriptor { - field_descriptors: vec![ - field("sensor_id", 1, ColumnType::Int64), - field("timestamp", 2, ColumnType::String), - field("latitude", 3, ColumnType::Double), - field("longitude", 4, ColumnType::Double), - ], -}); - -pub static JSON_VALUES_DESCRIPTOR: Lazy = Lazy::new(|| TableDescriptor { - field_descriptors: vec![ - field("sensor_id", 1, ColumnType::Int64), - field("timestamp", 2, ColumnType::String), - field("value", 3, ColumnType::String), - ], -}); - -pub static BLOB_VALUES_DESCRIPTOR: Lazy = Lazy::new(|| TableDescriptor { - field_descriptors: vec![ - field("sensor_id", 1, ColumnType::Int64), - field("timestamp", 2, ColumnType::String), - field("value", 3, ColumnType::Bytes), - ], -}); diff --git a/src/storage/bigquery/bigquery_units_utilities.rs b/src/storage/bigquery/bigquery_units_utilities.rs deleted file mode 100644 index b9a824e1..00000000 --- a/src/storage/bigquery/bigquery_units_utilities.rs +++ /dev/null @@ -1,187 +0,0 @@ -use std::{collections::HashMap, num::NonZeroUsize}; - -use crate::storage::StorageError; -use anyhow::Result; -use clru::CLruCache; -use gcp_bigquery_client::model::{ - query_parameter::QueryParameter, query_parameter_type::QueryParameterType, - query_parameter_value::QueryParameterValue, query_request::QueryRequest, - query_response::ResultSet, -}; -use hybridmap::HybridMap; -use once_cell::sync::Lazy; -use tokio::sync::Mutex; -use tracing::{debug, info}; - -use crate::datamodel::{SensAppVec, unit::Unit}; - -use super::{ - BigQueryStorage, bigquery_prost_structs::Unit as ProstUnit, - bigquery_table_descriptors::UNITS_DESCRIPTOR, bigquery_utilities::publish_rows, -}; - -static UNITS_CACHE: Lazy>> = - Lazy::new(|| Mutex::new(CLruCache::new(NonZeroUsize::new(16384).unwrap()))); - -pub async fn get_or_create_units_ids( - bqs: &BigQueryStorage, - units: SensAppVec, -) -> Result> { - if units.is_empty() { - return Ok(HybridMap::new()); - } - - let mut unknown_units: HashMap = HashMap::new(); - - let mut result = HybridMap::new(); - - { - let mut cache_guard = UNITS_CACHE.lock().await; - for unit in units.into_iter() { - match cache_guard.get(&unit.name) { - Some(id) => { - result.insert(unit.name, *id); - } - None => { - unknown_units.insert(unit.name.clone(), unit); - } - }; - } - } - - debug!("BigQuery: Found {} known units", result.len()); - debug!("BigQuery: Found {} unknown units", unknown_units.len()); - - if unknown_units.is_empty() { - return Ok(result); - } - - let just_the_units = unknown_units - .values() - .cloned() - .collect::>(); - - let found_ids = get_existing_units_ids(bqs, &just_the_units).await?; - { - let mut cache_guard = UNITS_CACHE.lock().await; - for (unit, id) in found_ids.iter() { - cache_guard.put(unit.clone(), *id); - result.insert(unit.clone(), *id); - } - } - - let units_to_create = unknown_units - .into_values() - .filter(|unit| found_ids.get(&unit.name).is_none()) - .collect::>(); - - if units_to_create.is_empty() { - return Ok(result); - } - - info!("BigQuery: Creating {} new units", units_to_create.len()); - - let new_ids = create_units(bqs, units_to_create).await?; - { - let mut cache_guard = UNITS_CACHE.lock().await; - for (unit, id) in new_ids.iter() { - cache_guard.put(unit.clone(), *id); - result.insert(unit.clone(), *id); - } - } - - Ok(result) -} - -async fn get_existing_units_ids( - bqs: &BigQueryStorage, - units: &[Unit], -) -> Result> { - let mut query_request = QueryRequest::new( - r#" - SELECT id, name - FROM `{dataset_id}.units` - WHERE name IN UNNEST(@names) - "# - .replace("{dataset_id}", bqs.dataset_id()), - ); - - let query_parameter = QueryParameter { - name: Some("names".to_string()), - parameter_type: Some(QueryParameterType { - r#type: "ARRAY".to_string(), - struct_types: None, - array_type: Some(Box::new(QueryParameterType { - r#type: "STRING".to_string(), - struct_types: None, - array_type: None, - })), - }), - parameter_value: Some(QueryParameterValue { - value: None, - struct_values: None, - array_values: Some( - units - .iter() - .map(|unit| QueryParameterValue { - value: Some(unit.name.clone()), - struct_values: None, - array_values: None, - }) - .collect(), - ), - }), - }; - query_request.query_parameters = Some(vec![query_parameter]); - - let result = bqs - .client() - .read() - .await - .job() - .query(bqs.project_id(), query_request) - .await?; - let mut result = ResultSet::new_from_query_response(result); - - let mut results_map = HybridMap::with_capacity(result.row_count()); - - while result.next_row() { - let id = result.get_i64(0)?.ok_or_else(|| { - anyhow::Error::from(StorageError::missing_field("unit_id", None, None)) - })?; - let name = result.get_string(1)?.ok_or_else(|| { - anyhow::Error::from(StorageError::missing_field("unit_name", None, None)) - })?; - - results_map.insert(name, id); - } - - Ok(results_map) -} - -async fn create_units( - bqs: &BigQueryStorage, - units: SensAppVec, -) -> Result> { - let mut map = HybridMap::with_capacity(units.len()); - - for unit in units.as_ref() { - let id = sinteflake::next_id_with_hash_async(unit.name.as_bytes()).await? as i64; - map.insert(unit.name.clone(), id); - } - - let rows = units - .into_iter() - .map(|unit| ProstUnit { - id: *map - .get(&unit.name) - .expect("Internal consistency error: Unit missing from cached map"), - name: unit.name, - description: unit.description, - }) - .collect::>(); - - publish_rows(bqs, "units", &UNITS_DESCRIPTOR, rows).await?; - - Ok(map) -} diff --git a/src/storage/bigquery/bigquery_utilities.rs b/src/storage/bigquery/bigquery_utilities.rs deleted file mode 100644 index 7bc64184..00000000 --- a/src/storage/bigquery/bigquery_utilities.rs +++ /dev/null @@ -1,125 +0,0 @@ -use anyhow::{Result, bail}; -use gcp_bigquery_client::{ - google::cloud::bigquery::storage::v1::AppendRowsResponse, - storage::{StorageApi, TableDescriptor}, -}; -use tokio_stream::StreamExt; -use tonic::Streaming; - -use super::BigQueryStorage; -use tracing::{debug, error}; - -pub async fn publish_rows( - bqs: &BigQueryStorage, - table_name: &'static str, - table_descriptor: &TableDescriptor, - rows: Vec, -) -> Result<()> { - if rows.is_empty() { - debug!("BigQuery: No {} rows to publish", table_name); - return Ok(()); - } - - debug!("BigQuery: Publishing {} rows to {}", rows.len(), table_name); - - let stream_name = bqs.new_stream_name(table_name.to_string()); - let trace_id = create_trace_id(table_name); - let (rows, _) = StorageApi::create_rows(table_descriptor, &rows, 9 * 1024 * 1024); - - let streaming = bqs - .client() - .write() - .await - .storage_mut() - .append_rows(&stream_name, rows, trace_id) - .await?; - - check_streaming(streaming).await?; - - Ok(()) -} - -async fn check_streaming(mut streaming: Streaming) -> Result<()> { - while let Some(response) = streaming.next().await { - let response = response?; - if !response.row_errors.is_empty() { - for error in response.row_errors { - error!("BigQuery Row Error: {:?}", error); - } - bail!("Failed to publish rows"); - } - } - - Ok(()) -} - -pub fn create_trace_id(context: &str) -> String { - format!("sensapp-{}-{}", context, uuid::Uuid::new_v4()) -} - -// pub fn convert_sensapp_timestamp_to_prost_timestamp( -// timestamp: SensAppDateTime, -// ) -> Result { -// let unix_seconds = timestamp.to_unix_seconds(); -// let seconds = unix_seconds -// .trunc() -// .to_i64() -// .ok_or_else(|| anyhow!("Failed to convert seconds"))?; - -// let nanos = (unix_seconds.fract() * 1_000_000_000.0) -// .to_i32() -// .ok_or_else(|| anyhow!("Failed to convert nanos"))?; - -// Ok(prost_types::Timestamp { seconds, nanos }) -// } - -#[cfg(test)] -mod tests { - // use super::*; - // use crate::datamodel::sensapp_datetime::SensAppDateTimeExt; - // use prost_types::Timestamp; - // - // #[test] - // fn test_convert_sensapp_timestamp_to_prost_timestamp() { - // // Test case 1: Simple whole second - // let sensapp_time = SensAppDateTime::from_unix_seconds_i64(1625097600); // 2021-07-01 00:00:00 UTC - // let result = convert_sensapp_timestamp_to_prost_timestamp(sensapp_time).unwrap(); - // assert_eq!( - // result, - // Timestamp { - // seconds: 1625097600, - // nanos: 0 - // } - // ); - - // // Test case 2: With fractional seconds - // let sensapp_time = SensAppDateTime::from_unix_seconds_i64(1625097600) - // + hifitime::Duration::from_milliseconds(500.0); - // let result = convert_sensapp_timestamp_to_prost_timestamp(sensapp_time).unwrap(); - // assert_eq!( - // result, - // Timestamp { - // seconds: 1625097600, - // nanos: 500_000_000 - // } - // ); - - // // Test case 3: Current time - // let sensapp_time = SensAppDateTime::now().unwrap(); - // let result = convert_sensapp_timestamp_to_prost_timestamp(sensapp_time).unwrap(); - // assert!(result.seconds > 0); - // assert!(result.nanos >= 0 && result.nanos < 1_000_000_000); - - // // Test case 4: Edge case - christmas 2124 - // const CHRISTMAS_UNIX_TIMESTAMP: i64 = 4_890_758_400; - // let max_time = SensAppDateTime::from_unix_seconds_i64(CHRISTMAS_UNIX_TIMESTAMP); - // let result = convert_sensapp_timestamp_to_prost_timestamp(max_time).unwrap(); - // assert_eq!(result.seconds, CHRISTMAS_UNIX_TIMESTAMP); - // assert_eq!(result.nanos, 0); - - // // Test case: Time before Unix epoch - // let before_epoch = SensAppDateTime::from_unix_seconds_i64(-1); - // let result = convert_sensapp_timestamp_to_prost_timestamp(before_epoch).unwrap(); - // assert_eq!(result.seconds, -1); - // } -} diff --git a/src/storage/bigquery/client.rs b/src/storage/bigquery/client.rs new file mode 100644 index 00000000..aa49c5f0 --- /dev/null +++ b/src/storage/bigquery/client.rs @@ -0,0 +1,375 @@ +//! Running statements: query parameters, jobs that outlast the first answer, result pages, DML counts +//! and the sorting of errors. +//! +//! `jobs.query` answers after 10 seconds even when the job is not done (`jobComplete: false`, no +//! rows), and `ResultSet::new_from_query_response` turns that answer into an empty result. Every +//! statement of SensApp goes through [`BigQueryStorage::run_query`], which waits for the job and +//! reads all the pages. + +use super::BigQueryStorage; +use crate::storage::StorageError; +use anyhow::{Result, anyhow}; +use gcp_bigquery_client::{ + error::BQError, + model::{ + get_query_results_parameters::GetQueryResultsParameters, query_parameter::QueryParameter, + query_parameter_type::QueryParameterType, query_parameter_value::QueryParameterValue, + query_request::QueryRequest, query_response::ResultSet, + }, +}; +use std::time::{Duration, Instant}; + +/// How long one request waits for a job (the service default is 10 seconds) +const WAIT_MS: i32 = 20_000; +/// How long a statement may run in total before the request gives up on it +const DEADLINE: Duration = Duration::from_secs(300); + +fn parameter( + name: &str, + parameter_type: QueryParameterType, + value: QueryParameterValue, +) -> QueryParameter { + QueryParameter { + name: Some(name.to_string()), + parameter_type: Some(parameter_type), + parameter_value: Some(value), + } +} + +fn scalar_type(name: &str) -> QueryParameterType { + QueryParameterType { + r#type: name.to_string(), + struct_types: None, + array_type: None, + } +} + +fn scalar_value(value: String) -> QueryParameterValue { + QueryParameterValue { + value: Some(value), + struct_values: None, + array_values: None, + } +} + +fn array_type(element: &str) -> QueryParameterType { + QueryParameterType { + r#type: "ARRAY".to_string(), + struct_types: None, + array_type: Some(Box::new(scalar_type(element))), + } +} + +fn array_value(values: impl Iterator) -> QueryParameterValue { + QueryParameterValue { + value: None, + struct_values: None, + array_values: Some(values.map(scalar_value).collect()), + } +} + +pub fn string_param(name: &str, value: &str) -> QueryParameter { + parameter(name, scalar_type("STRING"), scalar_value(value.to_string())) +} + +pub fn int_param(name: &str, value: i64) -> QueryParameter { + parameter(name, scalar_type("INT64"), scalar_value(value.to_string())) +} + +pub fn string_array_param(name: &str, values: &[&str]) -> QueryParameter { + parameter( + name, + array_type("STRING"), + array_value(values.iter().map(|value| value.to_string())), + ) +} + +pub fn int_array_param(name: &str, values: &[i64]) -> QueryParameter { + parameter( + name, + array_type("INT64"), + array_value(values.iter().map(i64::to_string)), + ) +} + +/// Whether a failure is worth retrying later: the service or the network, not the statement. +pub fn is_transient(error: &BQError) -> bool { + match error { + BQError::RequestError(error) => error.is_timeout() || error.is_connect(), + BQError::ResponseError { error } => { + let reason = error + .error + .errors + .first() + .and_then(|details| details.get("reason")) + .map(String::as_str); + matches!(error.error.code, 408 | 429 | 500 | 502 | 503 | 504) + || (error.error.code == 403 + && matches!(reason, Some("rateLimitExceeded" | "backendError"))) + } + BQError::TonicStatusError(status) => grpc_code_is_transient(status.code() as i32), + BQError::TonicTransportError(_) | BQError::ConnectionPoolError(_) => true, + _ => false, + } +} + +/// `google.rpc.Code` numbers, shared by gRPC statuses and the `error` of an append response: +/// CANCELLED is left out, the others are DEADLINE_EXCEEDED, RESOURCE_EXHAUSTED, ABORTED, INTERNAL +/// and UNAVAILABLE. +pub fn grpc_code_is_transient(code: i32) -> bool { + matches!(code, 4 | 8 | 10 | 13 | 14) +} + +pub fn map_error(operation: &str, error: BQError) -> anyhow::Error { + if is_transient(&error) { + StorageError::Unavailable(format!("{operation}: {error}")).into() + } else { + StorageError::OperationFailed { + operation: format!("BigQuery {operation}"), + details: error.to_string(), + } + .into() + } +} + +pub struct QueryOutput { + pub pages: Vec, + /// Rows changed by a DML statement + pub affected_rows: u64, +} + +impl BigQueryStorage { + /// The fully qualified, quoted name of a table of the dataset. + pub(super) fn table(&self, name: &str) -> String { + self.dataset.table(name) + } + + pub(super) async fn run_query( + &self, + operation: &str, + sql: String, + params: Vec, + ) -> Result { + let mut request = QueryRequest::new(sql); + if !params.is_empty() { + request.query_parameters = Some(params); + request.parameter_mode = Some("NAMED".to_string()); + } + request.location = self.location.clone(); + request.maximum_bytes_billed = self.max_bytes_billed.map(|bytes| bytes.to_string()); + request.timeout_ms = Some(WAIT_MS); + // A result is never served from the cache of an earlier identical query: a read must see + // what a write just stored + request.use_query_cache = Some(false); + + let job = self.client.job(); + let started = Instant::now(); + let mut response = job + .query(&self.dataset.project_id, request) + .await + .map_err(|error| map_error(operation, error))?; + + // The job outlasted the wait: poll it until it is done + while !response.job_complete.unwrap_or(false) { + if started.elapsed() > DEADLINE { + return Err(StorageError::Unavailable(format!( + "{operation}: the statement did not finish in {} seconds", + DEADLINE.as_secs() + )) + .into()); + } + let reference = response + .job_reference + .as_ref() + .ok_or_else(|| anyhow!("BigQuery gave no job reference for {operation}"))?; + let job_id = reference + .job_id + .as_ref() + .ok_or_else(|| anyhow!("BigQuery gave no job id for {operation}"))?; + let results = job + .get_query_results( + &self.dataset.project_id, + job_id, + GetQueryResultsParameters { + location: reference.location.clone(), + timeout_ms: Some(WAIT_MS), + ..Default::default() + }, + ) + .await + .map_err(|error| map_error(operation, error))?; + response = results.into(); + } + + let affected_rows = response + .num_dml_affected_rows + .as_deref() + .and_then(|rows| rows.parse().ok()) + .unwrap_or(0); + let reference = response.job_reference.clone(); + let mut next_page = response.page_token.clone(); + let mut pages = vec![ResultSet::new_from_query_response(response)]; + + while let Some(page_token) = next_page.take() { + let reference = reference + .as_ref() + .ok_or_else(|| anyhow!("BigQuery gave no job reference for {operation}"))?; + let job_id = reference + .job_id + .as_ref() + .ok_or_else(|| anyhow!("BigQuery gave no job id for {operation}"))?; + let results = job + .get_query_results( + &self.dataset.project_id, + job_id, + GetQueryResultsParameters { + location: reference.location.clone(), + page_token: Some(page_token), + timeout_ms: Some(WAIT_MS), + ..Default::default() + }, + ) + .await + .map_err(|error| map_error(operation, error))?; + if !results.job_complete.unwrap_or(false) { + return Err(StorageError::Unavailable(format!( + "{operation}: BigQuery lost the rest of the result" + )) + .into()); + } + next_page = results.page_token.clone(); + pages.push(ResultSet::new_from_get_query_results_response(results)); + } + + Ok(QueryOutput { + pages, + affected_rows, + }) + } + + /// Run a query and convert each row of each page. + pub(super) async fn query_rows( + &self, + operation: &str, + sql: String, + params: Vec, + mut read: impl FnMut(&ResultSet) -> Result, + ) -> Result> { + let output = self.run_query(operation, sql, params).await?; + let mut rows = Vec::new(); + for mut page in output.pages { + while page.next_row() { + rows.push(read(&page)?); + } + } + Ok(rows) + } + + /// Run a statement that returns no rows, and say how many rows it changed. + pub(super) async fn execute( + &self, + operation: &str, + sql: String, + params: Vec, + ) -> Result { + Ok(self.run_query(operation, sql, params).await?.affected_rows) + } +} + +/// The value of a column that is never null in SensApp's tables. +pub fn required(value: std::result::Result, BQError>, column: &str) -> Result { + value?.ok_or_else(|| StorageError::missing_field(column, None, None).into()) +} + +#[cfg(test)] +mod tests { + use super::*; + use gcp_bigquery_client::error::{NestedResponseError, ResponseError}; + use std::collections::HashMap; + + fn http_error(code: i64, reason: &str) -> BQError { + BQError::ResponseError { + error: ResponseError { + error: NestedResponseError { + code, + errors: vec![HashMap::from([("reason".to_string(), reason.to_string())])], + message: "message".to_string(), + status: String::new(), + }, + }, + } + } + + #[test] + fn the_parameters_have_the_json_shape_of_the_api() { + let request = { + let mut request = QueryRequest::new("SELECT 1"); + request.query_parameters = Some(vec![ + string_param("name", "°C"), + int_param("id", -5), + int_array_param("ids", &[1, 2]), + ]); + serde_json::to_value(request).unwrap() + }; + let parameters = &request["queryParameters"]; + assert_eq!(parameters[0]["name"], "name"); + assert_eq!(parameters[0]["parameterType"]["type"], "STRING"); + assert_eq!(parameters[0]["parameterValue"]["value"], "°C"); + assert_eq!(parameters[1]["parameterType"]["type"], "INT64"); + assert_eq!(parameters[1]["parameterValue"]["value"], "-5"); + assert_eq!(parameters[2]["parameterType"]["type"], "ARRAY"); + assert_eq!(parameters[2]["parameterType"]["arrayType"]["type"], "INT64"); + assert_eq!( + parameters[2]["parameterValue"]["arrayValues"][1]["value"], + "2" + ); + assert_eq!(request["useLegacySql"], false); + } + + #[test] + fn an_empty_array_is_still_an_array() { + let value = serde_json::to_value(int_array_param("ids", &[])).unwrap(); + assert_eq!( + value["parameterValue"]["arrayValues"], + serde_json::json!([]) + ); + } + + #[test] + fn overload_and_outages_are_transient_bad_statements_are_not() { + for code in [408, 429, 500, 502, 503, 504] { + assert!(is_transient(&http_error(code, "x")), "{code}"); + } + assert!(is_transient(&http_error(403, "rateLimitExceeded"))); + assert!(!is_transient(&http_error(403, "accessDenied"))); + assert!(!is_transient(&http_error(400, "invalidQuery"))); + assert!(!is_transient(&http_error(404, "notFound"))); + assert!(!is_transient(&BQError::NoToken)); + assert!(is_transient(&BQError::ConnectionPoolError("down".into()))); + } + + #[test] + fn grpc_codes() { + for code in [4, 8, 10, 13, 14] { + assert!(grpc_code_is_transient(code), "{code}"); + } + // INVALID_ARGUMENT, NOT_FOUND, PERMISSION_DENIED + for code in [3, 5, 7] { + assert!(!grpc_code_is_transient(code), "{code}"); + } + } + + #[test] + fn errors_are_sorted_for_the_http_layer() { + let unavailable = map_error("read", http_error(503, "backendError")); + assert!(matches!( + unavailable.downcast_ref::(), + Some(StorageError::Unavailable(_)) + )); + let failed = map_error("read", http_error(400, "invalidQuery")); + assert!(matches!( + failed.downcast_ref::(), + Some(StorageError::OperationFailed { .. }) + )); + } +} diff --git a/src/storage/bigquery/connection.rs b/src/storage/bigquery/connection.rs new file mode 100644 index 00000000..d5f55952 --- /dev/null +++ b/src/storage/bigquery/connection.rs @@ -0,0 +1,200 @@ +//! The connection string: `bigquery://[key.json]?project_id=P&dataset_id=D[&location=L][&max_bytes_billed=N]`. +//! +//! Without a key file the client uses the Application Default Credentials: the +//! `GOOGLE_APPLICATION_CREDENTIALS` file, what `gcloud auth application-default login` stored, or the +//! metadata server of Google Cloud. + +use crate::storage::StorageError; +use anyhow::{Result, bail}; + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct ConnectionInfo { + /// Service account key file, `None` for the Application Default Credentials + pub credentials_file: Option, + pub project_id: String, + pub dataset_id: String, + /// Where the dataset is created, and where the queries run when it is given + pub location: Option, + /// Queries that would bill more bytes than this fail without being charged + pub max_bytes_billed: Option, +} + +/// Project ids are lowercase letters, digits and hyphens (domain-scoped ones also have `.` and `:`). +/// What matters here is that the id can sit inside backticks in a statement, so no quote, backtick, +/// space or other punctuation. +pub fn validate_project_id(project_id: &str) -> Result<()> { + let valid = !project_id.is_empty() + && project_id.len() <= 128 + && project_id + .chars() + .all(|c| c.is_ascii_alphanumeric() || matches!(c, '-' | '_' | '.' | ':')); + if !valid { + return Err(StorageError::Configuration(format!( + "invalid BigQuery project_id '{project_id}': letters, digits, '-', '_', '.' and ':' only" + )) + .into()); + } + Ok(()) +} + +/// Dataset ids are letters, digits and underscores, up to 1024 characters. +pub fn validate_dataset_id(dataset_id: &str) -> Result<()> { + let valid = !dataset_id.is_empty() + && dataset_id.len() <= 1024 + && dataset_id + .chars() + .all(|c| c.is_ascii_alphanumeric() || c == '_'); + if !valid { + return Err(StorageError::Configuration(format!( + "invalid BigQuery dataset_id '{dataset_id}': letters, digits and '_' only" + )) + .into()); + } + Ok(()) +} + +pub fn parse_connection_string(connection_string: &str) -> Result { + let Some(rest) = connection_string.strip_prefix("bigquery:") else { + bail!("Invalid scheme in connection string, expected bigquery://"); + }; + // `bigquery://key.json`, `bigquery:///absolute/key.json`, `bigquery://?...` (no key file) + let rest = rest.strip_prefix("//").unwrap_or(rest); + let (path, query) = rest.split_once('?').unwrap_or((rest, "")); + + let credentials_file = if path.is_empty() { + None + } else { + Some(urlencoding::decode(path)?.into_owned()) + }; + + let mut project_id = None; + let mut dataset_id = None; + let mut location = None; + let mut max_bytes_billed = None; + for (key, value) in url::form_urlencoded::parse(query.as_bytes()) { + match key.as_ref() { + "project_id" => project_id = Some(value.into_owned()), + "dataset_id" => dataset_id = Some(value.into_owned()), + "location" => location = Some(value.into_owned()), + "max_bytes_billed" => { + let bytes: i64 = + value + .parse() + .ok() + .filter(|bytes| *bytes > 0) + .ok_or_else(|| { + StorageError::Configuration(format!( + "invalid max_bytes_billed '{value}': a number of bytes above 0" + )) + })?; + max_bytes_billed = Some(bytes); + } + other => bail!( + "Unknown parameter '{other}' in the BigQuery connection string \ + (known: project_id, dataset_id, location, max_bytes_billed)" + ), + } + } + + let project_id = project_id + .filter(|id| !id.is_empty()) + .ok_or_else(|| StorageError::Configuration("project_id is required".to_string()))?; + let dataset_id = dataset_id + .filter(|id| !id.is_empty()) + .ok_or_else(|| StorageError::Configuration("dataset_id is required".to_string()))?; + validate_project_id(&project_id)?; + validate_dataset_id(&dataset_id)?; + if let Some(location) = &location + && (location.is_empty() + || !location + .chars() + .all(|c| c.is_ascii_alphanumeric() || c == '-')) + { + return Err(StorageError::Configuration(format!("invalid location '{location}'")).into()); + } + + Ok(ConnectionInfo { + credentials_file, + project_id, + dataset_id, + location, + max_bytes_billed, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn parses_a_key_file_and_the_ids() { + let info = parse_connection_string( + "bigquery://key.json?project_id=my-project-123&dataset_id=sensapp_test", + ) + .unwrap(); + assert_eq!(info.credentials_file.as_deref(), Some("key.json")); + assert_eq!(info.project_id, "my-project-123"); + assert_eq!(info.dataset_id, "sensapp_test"); + assert_eq!(info.location, None); + assert_eq!(info.max_bytes_billed, None); + } + + #[test] + fn absolute_and_encoded_key_paths() { + let info = + parse_connection_string("bigquery:///home/me/my%20key.json?project_id=p&dataset_id=d") + .unwrap(); + assert_eq!( + info.credentials_file.as_deref(), + Some("/home/me/my key.json") + ); + } + + #[test] + fn no_key_file_means_application_default_credentials() { + for url in [ + "bigquery://?project_id=p&dataset_id=d", + "bigquery:?project_id=p&dataset_id=d", + ] { + assert_eq!(parse_connection_string(url).unwrap().credentials_file, None); + } + } + + #[test] + fn location_and_cost_cap() { + let info = parse_connection_string( + "bigquery://?project_id=p&dataset_id=d&location=europe-north1&max_bytes_billed=1000000000", + ) + .unwrap(); + assert_eq!(info.location.as_deref(), Some("europe-north1")); + assert_eq!(info.max_bytes_billed, Some(1_000_000_000)); + } + + #[test] + fn missing_or_unknown_parameters_are_refused() { + assert!(parse_connection_string("bigquery://key.json?dataset_id=d").is_err()); + assert!(parse_connection_string("bigquery://key.json?project_id=p").is_err()); + assert!(parse_connection_string("bigquery://?project_id=p&dataset_id=d&datset=x").is_err()); + assert!(parse_connection_string("postgres://?project_id=p&dataset_id=d").is_err()); + assert!( + parse_connection_string("bigquery://?project_id=p&dataset_id=d&max_bytes_billed=0") + .is_err() + ); + } + + #[test] + fn identifiers_that_could_leave_the_backticks_are_refused() { + for bad in ["p`; DROP", "p q", "p'", "", "p/../x"] { + assert!(validate_project_id(bad).is_err(), "{bad}"); + } + for bad in ["d`x", "d-x", "d.x", "", "d x"] { + assert!(validate_dataset_id(bad).is_err(), "{bad}"); + } + assert!(validate_project_id("example.com:my-project").is_ok()); + assert!(validate_dataset_id("sensapp_2026").is_ok()); + assert!( + parse_connection_string("bigquery://?project_id=p&dataset_id=d%60%3B").is_err(), + "the encoded backtick is decoded before the check" + ); + } +} diff --git a/src/storage/bigquery/matchers.rs b/src/storage/bigquery/matchers.rs new file mode 100644 index 00000000..e5d3edb9 --- /dev/null +++ b/src/storage/bigquery/matchers.rs @@ -0,0 +1,138 @@ +//! Label matchers as SQL conditions on the deduplicated sensors (`s`) and the `labels` table. +//! +//! The semantics are the ones of the ClickHouse backend: `!=` and `!~` also select the series that +//! do not have the label. Regex patterns are anchored by `LabelMatcher` and run by RE2, like +//! Prometheus' own. + +use super::BigQueryStorage; +use super::client::string_param; +use crate::datamodel::Sensor; +use crate::storage::{LabelMatcher, MatcherType}; +use anyhow::Result; +use gcp_bigquery_client::model::query_parameter::QueryParameter; + +/// The `WHERE` conditions and parameters (`m0`, `m1`, ...) of a selector. +pub fn matcher_conditions( + name_matchers: &[&LabelMatcher], + label_matchers: &[&LabelMatcher], + numeric_only: bool, + labels_table: &str, +) -> (Vec, Vec) { + let mut conditions = Vec::new(); + let mut params = Vec::new(); + let mut parameter = |value: &str| { + let name = format!("m{}", params.len()); + params.push(string_param(&name, value)); + format!("@{name}") + }; + + if numeric_only { + conditions.push("s.type IN ('Integer', 'Numeric', 'Float')".to_string()); + } + + for matcher in name_matchers { + let value = parameter(&matcher.value); + conditions.push(match matcher.matcher_type { + MatcherType::Equal => format!("s.name = {value}"), + MatcherType::NotEqual => format!("s.name != {value}"), + MatcherType::RegexMatch => format!("REGEXP_CONTAINS(s.name, {value})"), + MatcherType::RegexNotMatch => format!("NOT REGEXP_CONTAINS(s.name, {value})"), + }); + } + + for matcher in label_matchers { + let name = parameter(&matcher.name); + let value = parameter(&matcher.value); + let (membership, test) = match matcher.matcher_type { + MatcherType::Equal => ("IN", format!("l.description = {value}")), + MatcherType::NotEqual => ("NOT IN", format!("l.description = {value}")), + MatcherType::RegexMatch => ("IN", format!("REGEXP_CONTAINS(l.description, {value})")), + MatcherType::RegexNotMatch => { + ("NOT IN", format!("REGEXP_CONTAINS(l.description, {value})")) + } + }; + conditions.push(format!( + "s.sensor_id {membership} (SELECT l.sensor_id FROM {labels_table} l \ + WHERE l.name = {name} AND {test})" + )); + } + + (conditions, params) +} + +impl BigQueryStorage { + /// The series matching all the matchers, by increasing id, with their labels and units. + pub(super) async fn find_sensors_by_matchers( + &self, + matchers: &[LabelMatcher], + numeric_only: bool, + limit: Option, + ) -> Result> { + let (name_matchers, label_matchers): (Vec<_>, Vec<_>) = matchers + .iter() + .partition(|matcher| matcher.is_name_matcher()); + let (conditions, params) = matcher_conditions( + &name_matchers, + &label_matchers, + numeric_only, + &self.table("labels"), + ); + let mut tail = "ORDER BY s.sensor_id".to_string(); + if let Some(limit) = limit { + // A number, never text from a caller + tail.push_str(&format!(" LIMIT {limit}")); + } + self.read_sensors(self.dataset.sensors_sql(&conditions, &tail), params) + .await + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn conditions(matchers: &[LabelMatcher], numeric_only: bool) -> (Vec, usize) { + let (names, labels): (Vec<_>, Vec<_>) = matchers + .iter() + .partition(|matcher| matcher.is_name_matcher()); + let (conditions, params) = + matcher_conditions(&names, &labels, numeric_only, "`p.d.labels`"); + (conditions, params.len()) + } + + #[test] + fn names_and_labels_become_parameters_never_text() { + let (conditions, parameters) = conditions( + &[ + LabelMatcher::eq("__name__", "cpu'; DROP TABLE x; --"), + LabelMatcher::eq("site", "oslo"), + LabelMatcher::neq("env", "prod"), + ], + false, + ); + assert_eq!(parameters, 5); + assert_eq!(conditions[0], "s.name = @m0"); + assert!( + conditions[1].starts_with("s.sensor_id IN (SELECT l.sensor_id FROM `p.d.labels` l") + ); + assert!(conditions[1].contains("l.name = @m1 AND l.description = @m2")); + assert!(conditions[2].starts_with("s.sensor_id NOT IN")); + assert!(conditions.iter().all(|c| !c.contains("DROP"))); + } + + #[test] + fn regexes_use_re2_and_numeric_only_filters_the_type() { + let (conditions, parameters) = conditions( + &[ + LabelMatcher::regex("__name__", "cpu.*"), + LabelMatcher::not_regex("site", "o.*"), + ], + true, + ); + assert_eq!(parameters, 3); + assert_eq!(conditions[0], "s.type IN ('Integer', 'Numeric', 'Float')"); + assert_eq!(conditions[1], "REGEXP_CONTAINS(s.name, @m0)"); + assert!(conditions[2].starts_with("s.sensor_id NOT IN")); + assert!(conditions[2].contains("REGEXP_CONTAINS(l.description, @m2)")); + } +} diff --git a/src/storage/bigquery/migrations/20240223133248_init.sql b/src/storage/bigquery/migrations/20240223133248_init.sql deleted file mode 100644 index ef70aeb9..00000000 --- a/src/storage/bigquery/migrations/20240223133248_init.sql +++ /dev/null @@ -1,151 +0,0 @@ --- Create the 'units' table -CREATE TABLE IF NOT EXISTS `{dataset_id}.units` ( - id INT64 NOT NULL, - name STRING NOT NULL, - description STRING -); - --- Create the 'sensors' table -CREATE TABLE IF NOT EXISTS `{dataset_id}.sensors` ( - sensor_id INT64 NOT NULL, - uuid STRING NOT NULL, - name STRING NOT NULL, - type STRING NOT NULL, - unit INT64 -); - --- Create the 'labels_name_dictionary' table -CREATE TABLE IF NOT EXISTS `{dataset_id}.labels_name_dictionary` ( - id INT64 NOT NULL, - name STRING NOT NULL -); - --- Create the 'labels_description_dictionary' table -CREATE TABLE IF NOT EXISTS `{dataset_id}.labels_description_dictionary` ( - id INT64 NOT NULL, - description STRING NOT NULL -); - --- Create the 'labels' table -CREATE TABLE IF NOT EXISTS `{dataset_id}.labels` ( - sensor_id INT64 NOT NULL, - name INT64 NOT NULL, - description INT64 -); - --- Create the 'strings_values_dictionary' table -CREATE TABLE IF NOT EXISTS `{dataset_id}.strings_values_dictionary` ( - id INT64 NOT NULL, - value STRING NOT NULL -); - --- Create the 'integer_values' table -CREATE TABLE IF NOT EXISTS `{dataset_id}.integer_values` ( - sensor_id INT64 NOT NULL, - timestamp TIMESTAMP NOT NULL, - value INT64 NOT NULL -) -PARTITION BY DATE(timestamp) -CLUSTER BY sensor_id; - --- Create the 'numeric_values' table -CREATE TABLE IF NOT EXISTS `{dataset_id}.numeric_values` ( - sensor_id INT64 NOT NULL, - timestamp TIMESTAMP NOT NULL, - value NUMERIC NOT NULL -) -PARTITION BY DATE(timestamp) -CLUSTER BY sensor_id; - --- Create the 'float_values' table -CREATE TABLE IF NOT EXISTS `{dataset_id}.float_values` ( - sensor_id INT64 NOT NULL, - timestamp TIMESTAMP NOT NULL, - value FLOAT64 NOT NULL -) -PARTITION BY DATE(timestamp) -CLUSTER BY sensor_id; - --- Create the 'string_values' table -CREATE TABLE IF NOT EXISTS `{dataset_id}.string_values` ( - sensor_id INT64 NOT NULL, - timestamp TIMESTAMP NOT NULL, - value INT64 NOT NULL -) -PARTITION BY DATE(timestamp) -CLUSTER BY sensor_id; - --- Create the 'boolean_values' table -CREATE TABLE IF NOT EXISTS `{dataset_id}.boolean_values` ( - sensor_id INT64 NOT NULL, - timestamp TIMESTAMP NOT NULL, - value BOOL NOT NULL -) -PARTITION BY DATE(timestamp) -CLUSTER BY sensor_id; - --- Create the 'location_values' table -CREATE TABLE IF NOT EXISTS `{dataset_id}.location_values` ( - sensor_id INT64 NOT NULL, - timestamp TIMESTAMP NOT NULL, - latitude FLOAT64 NOT NULL, - longitude FLOAT64 NOT NULL -) -PARTITION BY DATE(timestamp) -CLUSTER BY sensor_id; - --- Create the 'json_values' table -CREATE TABLE IF NOT EXISTS `{dataset_id}.json_values` ( - sensor_id INT64 NOT NULL, - timestamp TIMESTAMP NOT NULL, - value JSON NOT NULL -) -PARTITION BY DATE(timestamp) -CLUSTER BY sensor_id; - --- Create the 'blob_values' table -CREATE TABLE IF NOT EXISTS `{dataset_id}.blob_values` ( - sensor_id INT64 NOT NULL, - timestamp TIMESTAMP NOT NULL, - value BYTES NOT NULL -) -PARTITION BY DATE(timestamp) -CLUSTER BY sensor_id; - - - -CREATE OR REPLACE VIEW `{dataset_id}.sensor_numeric_values` AS -SELECT - s.sensor_id, - s.uuid, - s.name AS sensor_name, - s.type AS sensor_type, - s.unit, - nv.timestamp, - nv.value AS numeric_value -FROM - `{dataset_id}.sensors` s -JOIN - `{dataset_id}.numeric_values` nv -ON - s.sensor_id = nv.sensor_id; - -CREATE OR REPLACE VIEW `{dataset_id}.sensor_labels_view` AS -SELECT - s.sensor_id, - s.uuid, - s.name AS sensor_name, - s.type, - s.unit, - ( - SELECT JSON_OBJECT( - ARRAY_AGG(lnd.name), - ARRAY_AGG(ldd.description) - ) - FROM `{dataset_id}.labels` l - LEFT JOIN `{dataset_id}.labels_name_dictionary` lnd ON l.name = lnd.id - LEFT JOIN `{dataset_id}.labels_description_dictionary` ldd ON l.description = ldd.id - WHERE l.sensor_id = s.sensor_id - ) AS labels -FROM - `{dataset_id}.sensors` s; diff --git a/src/storage/bigquery/migrations/init.sql b/src/storage/bigquery/migrations/init.sql new file mode 100644 index 00000000..0051a77c --- /dev/null +++ b/src/storage/bigquery/migrations/init.sql @@ -0,0 +1,97 @@ +-- SensApp schema for BigQuery. `{dataset}` is replaced by `project.dataset`. +-- +-- No sequential ids and no unique keys in BigQuery: sensor and unit ids are derived from the UUID and +-- the name (see `storage::common`), so two writers registering the same sensor at once insert +-- identical rows, and the reads collapse them. +-- +-- Samples are stored once per write: a retry after a timeout can leave a duplicate. BigQuery cannot +-- refuse it and SensApp does not remove it. + +CREATE TABLE IF NOT EXISTS `{dataset}.units` ( + id INT64 NOT NULL, + name STRING NOT NULL, + description STRING +); + +CREATE TABLE IF NOT EXISTS `{dataset}.sensors` ( + sensor_id INT64 NOT NULL, + uuid STRING NOT NULL, + name STRING NOT NULL, + type STRING NOT NULL, + unit INT64 +) +CLUSTER BY sensor_id; + +CREATE TABLE IF NOT EXISTS `{dataset}.labels` ( + sensor_id INT64 NOT NULL, + name STRING NOT NULL, + description STRING NOT NULL +) +CLUSTER BY sensor_id; + +-- Monthly partitions: BigQuery refuses more than 10 000 partitions in a table, and a daily partition +-- would stop at 27 years of distinct days. +CREATE TABLE IF NOT EXISTS `{dataset}.integer_values` ( + sensor_id INT64 NOT NULL, + timestamp TIMESTAMP NOT NULL, + value INT64 NOT NULL +) +PARTITION BY TIMESTAMP_TRUNC(timestamp, MONTH) +CLUSTER BY sensor_id; + +CREATE TABLE IF NOT EXISTS `{dataset}.numeric_values` ( + sensor_id INT64 NOT NULL, + timestamp TIMESTAMP NOT NULL, + value NUMERIC NOT NULL +) +PARTITION BY TIMESTAMP_TRUNC(timestamp, MONTH) +CLUSTER BY sensor_id; + +CREATE TABLE IF NOT EXISTS `{dataset}.float_values` ( + sensor_id INT64 NOT NULL, + timestamp TIMESTAMP NOT NULL, + value FLOAT64 NOT NULL +) +PARTITION BY TIMESTAMP_TRUNC(timestamp, MONTH) +CLUSTER BY sensor_id; + +CREATE TABLE IF NOT EXISTS `{dataset}.string_values` ( + sensor_id INT64 NOT NULL, + timestamp TIMESTAMP NOT NULL, + value STRING NOT NULL +) +PARTITION BY TIMESTAMP_TRUNC(timestamp, MONTH) +CLUSTER BY sensor_id; + +CREATE TABLE IF NOT EXISTS `{dataset}.boolean_values` ( + sensor_id INT64 NOT NULL, + timestamp TIMESTAMP NOT NULL, + value BOOL NOT NULL +) +PARTITION BY TIMESTAMP_TRUNC(timestamp, MONTH) +CLUSTER BY sensor_id; + +CREATE TABLE IF NOT EXISTS `{dataset}.location_values` ( + sensor_id INT64 NOT NULL, + timestamp TIMESTAMP NOT NULL, + latitude FLOAT64 NOT NULL, + longitude FLOAT64 NOT NULL +) +PARTITION BY TIMESTAMP_TRUNC(timestamp, MONTH) +CLUSTER BY sensor_id; + +CREATE TABLE IF NOT EXISTS `{dataset}.json_values` ( + sensor_id INT64 NOT NULL, + timestamp TIMESTAMP NOT NULL, + value JSON NOT NULL +) +PARTITION BY TIMESTAMP_TRUNC(timestamp, MONTH) +CLUSTER BY sensor_id; + +CREATE TABLE IF NOT EXISTS `{dataset}.blob_values` ( + sensor_id INT64 NOT NULL, + timestamp TIMESTAMP NOT NULL, + value BYTES NOT NULL +) +PARTITION BY TIMESTAMP_TRUNC(timestamp, MONTH) +CLUSTER BY sensor_id; diff --git a/src/storage/bigquery/mod.rs b/src/storage/bigquery/mod.rs index 366ba27d..46fdc5f4 100644 --- a/src/storage/bigquery/mod.rs +++ b/src/storage/bigquery/mod.rs @@ -1,548 +1,610 @@ +//! BigQuery: an optional backend for research and development. See `README.md` for what it does +//! differently, and `docs/BIGQUERY.md` for the limits and the setup. + use crate::{ - datamodel::{SensAppDateTime, SensorData}, - storage::StorageInstance, + datamodel::{Metric, SensAppDateTime, SensorData, SensorType, batch::Batch, unit::Unit}, + storage::{ + DEFAULT_LIST_SERIES_LIMIT, DEFAULT_QUERY_LIMIT, LabelMatcher, ListSeriesResult, + MAX_LIST_SERIES_LIMIT, SensorDataQueryOptions, StorageError, StorageInstance, + selector::{AggregatedRead, SelectorRead, aggregated_kind, empty_samples}, + }, }; -use anyhow::{Context, Result, bail}; +use anyhow::{Context, Result}; use async_trait::async_trait; -use bigquery_publishers::{ - publish_blob_values, publish_boolean_values, publish_float_values, publish_integer_values, - publish_json_values, publish_location_values, publish_numeric_values, publish_string_values, -}; -use bigquery_sensors_utilities::get_sensor_ids_or_create_sensors; -use futures::future::try_join_all; -use gcp_bigquery_client::{ - error::BQError, - model::{dataset::Dataset, query_request::QueryRequest, query_response::ResultSet}, - storage::StreamName, +use client::{int_param, required, string_param}; +use futures::{StreamExt, stream}; +use gcp_bigquery_client::{Client, error::BQError, model::dataset::Dataset as BqDataset}; +use reads::{Dataset, sample_table}; +use std::{ + collections::HashMap, + str::FromStr, + sync::{Arc, Mutex}, + time::{Duration, Instant}, }; -use once_cell::sync::Lazy; -use regex::Regex; -use std::{future::Future, pin::Pin, str::FromStr, sync::Arc}; -use tokio::sync::RwLock; use tracing::{debug, info}; -use url::Url; - -mod bigquery_labels_utilities; -mod bigquery_prost_structs; -mod bigquery_publishers; -mod bigquery_sensors_utilities; -mod bigquery_string_values_utilities; -mod bigquery_table_descriptors; -mod bigquery_units_utilities; -mod bigquery_utilities; - -pub struct BigQueryStorage { - client: Arc>, - project_id: String, - - dataset_id: String, +mod aggregation; +mod client; +mod connection; +mod matchers; +mod publishers; +mod reads; +mod rows; +mod selector; + +/// Where a dataset is created when the connection string does not say +const DEFAULT_LOCATION: &str = "europe-north1"; +/// A series that was registered is not looked up again for this long +const REGISTERED_LIFESPAN: Duration = Duration::from_secs(120); +const REGISTERED_CACHE_SIZE: usize = 65_536; +const HEALTH_CHECK_TIMEOUT: Duration = Duration::from_secs(15); +/// Series read at the same time by `query_sensors_by_labels` +const READ_CONCURRENCY: usize = 8; + +/// Tables of the schema before the dictionaries were dropped +const LEGACY_TABLES: [&str; 4] = [ + "labels_name_dictionary", + "labels_description_dictionary", + "strings_values_dictionary", + "sensor_labels_view", +]; + +/// Ids with the time they were seen, forgotten after `lifespan` and when there are too many. +struct RegisteredIds { + seen: HashMap, + lifespan: Duration, + capacity: usize, } -impl std::fmt::Debug for BigQueryStorage { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.debug_struct("BigQueryStorage") - .field("project_id", &self.project_id) - .field("dataset_id", &self.dataset_id) - .finish() +impl RegisteredIds { + fn new(lifespan: Duration, capacity: usize) -> Self { + Self { + seen: HashMap::new(), + lifespan, + capacity, + } } -} -fn parse_connection_string(connection_string: &str) -> Result<(String, String, String)> { - let url = Url::parse(connection_string)?; - if url.scheme() != "bigquery" { - bail!("Invalid scheme in connection string: {}", url.scheme()); + fn contains(&self, id: i64) -> bool { + self.seen + .get(&id) + .is_some_and(|seen| seen.elapsed() < self.lifespan) } - static URL_PARSE_REX: Lazy = - Lazy::new(|| Regex::new(r"^bigquery://?(.*?)(\?|$)").expect("Failed to compile regex")); - - let gcp_sa_key = URL_PARSE_REX - .captures(connection_string) - .map(|caps| caps.get(1).expect("Failed to get capture").as_str()) - .expect("Failed to get capture") - .to_string(); - - let mut project_id = String::new(); - let mut dataset_id = String::new(); - - for (key, value) in url.query_pairs() { - match key.as_ref() { - "project_id" => project_id = value.into_owned(), - "dataset_id" => dataset_id = value.into_owned(), - _ => {} // Ignore unknown parameters + fn insert(&mut self, id: i64) { + if self.seen.len() >= self.capacity { + let lifespan = self.lifespan; + self.seen.retain(|_, seen| seen.elapsed() < lifespan); + if self.seen.len() >= self.capacity { + self.seen.clear(); + } } + self.seen.insert(id, Instant::now()); } - if project_id.is_empty() { - bail!("project_id is required in connection string"); + fn remove(&mut self, id: i64) { + self.seen.remove(&id); } - if dataset_id.is_empty() { - bail!("dataset_id is required in connection string"); + + #[cfg(any(test, feature = "test-utils"))] + fn clear(&mut self) { + self.seen.clear(); } +} + +pub struct BigQueryStorage { + client: Client, + dataset: Dataset, + /// Where the dataset is created, and where the statements run when it is given + location: Option, + max_bytes_billed: Option, + /// Ids of the series known to be stored, so that a write does not look them up: a series + /// deleted by another instance is written to for up to `REGISTERED_LIFESPAN` after the delete. + registered: Mutex, +} - Ok((gcp_sa_key, project_id, dataset_id)) +impl std::fmt::Debug for BigQueryStorage { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("BigQueryStorage") + .field("project_id", &self.dataset.project_id) + .field("dataset_id", &self.dataset.dataset_id) + .finish() + } } impl BigQueryStorage { pub async fn connect(connection_string: &str) -> Result { - let (gcp_sa_key, project_id, dataset_id) = parse_connection_string(connection_string)?; - + let info = connection::parse_connection_string(connection_string)?; + // The gRPC client of the Storage Write API builds its TLS configuration from the default + // provider of rustls, and refuses to guess when both `ring` and `aws-lc-rs` are compiled in + // (the REST client does not care). The server installs it at startup; a library user or a + // test did not. `Err` only says that one is installed already. + let _ = rustls::crypto::aws_lc_rs::default_provider().install_default(); info!( "Connecting to BigQuery with project_id: {}, dataset_id: {}", - project_id, dataset_id + info.project_id, info.dataset_id ); - debug!("Using service account key file: {}", gcp_sa_key); - let client = Arc::new(RwLock::new( - gcp_bigquery_client::Client::from_service_account_key_file(&gcp_sa_key).await?, - )); - + let client = match &info.credentials_file { + Some(file) => Client::from_service_account_key_file(file) + .await + .with_context(|| format!("Failed to use the service account key {file}"))?, + None => Client::from_application_default_credentials() + .await + .context("Failed to find Application Default Credentials")?, + }; Ok(Self { client, - project_id, - dataset_id, + dataset: Dataset { + project_id: info.project_id, + dataset_id: info.dataset_id, + }, + location: info.location, + max_bytes_billed: info.max_bytes_billed, + registered: Mutex::new(RegisteredIds::new( + REGISTERED_LIFESPAN, + REGISTERED_CACHE_SIZE, + )), }) } - pub fn client(&self) -> Arc> { - self.client.clone() + fn registered(&self) -> std::sync::MutexGuard<'_, RegisteredIds> { + self.registered.lock().unwrap_or_else(|e| e.into_inner()) } - pub fn project_id(&self) -> &str { - &self.project_id + fn is_registered(&self, sensor_id: i64) -> bool { + self.registered().contains(sensor_id) } - pub fn dataset_id(&self) -> &str { - &self.dataset_id + fn remember_registered(&self, sensor_id: i64) { + self.registered().insert(sensor_id); } - pub fn new_stream_name(&self, table: String) -> StreamName { - StreamName::new_default(self.project_id.clone(), self.dataset_id.clone(), table) + fn forget_registered(&self, sensor_id: i64) { + self.registered().remove(sensor_id); } - fn parse_sensor_type(sensor_type: &str) -> Result { - crate::datamodel::SensorType::from_str(sensor_type).map_err(anyhow::Error::msg) + async fn reject_legacy_schema(&self) -> Result<()> { + let sql = format!( + "SELECT table_name FROM `{}.{}.INFORMATION_SCHEMA.TABLES` \ + WHERE table_name IN UNNEST(@names)", + self.dataset.project_id, self.dataset.dataset_id + ); + let params = vec![client::string_array_param("names", &LEGACY_TABLES)]; + let found = self + .query_rows("look for the legacy schema", sql, params, |row| { + required(row.get_string(0), "table_name") + }) + .await?; + if !found.is_empty() { + return Err(StorageError::Configuration(format!( + "the BigQuery dataset {} has the tables of the previous SensApp schema ({}); \ + it cannot be migrated, use a new dataset or drop the old one", + self.dataset.dataset_id, + found.join(", ") + )) + .into()); + } + Ok(()) } } #[async_trait] impl StorageInstance for BigQueryStorage { async fn create_or_migrate(&self) -> Result<()> { + let dataset = &self.dataset; match self .client - .read() - .await .dataset() - .get(&self.project_id, &self.dataset_id) + .get(&dataset.project_id, &dataset.dataset_id) .await { - Ok(_) => { - debug!("BigQuery dataset already exists"); - } + Ok(_) => debug!("BigQuery dataset already exists"), Err(BQError::ResponseError { error }) if error.error.code == 404 => { info!("BigQuery dataset does not exist, creating it"); - let dataset = - Dataset::new(&self.project_id, &self.dataset_id).location("europe-north1"); - self.client.read().await.dataset().create(dataset).await?; - } - Err(e) => { - return Err(e.into()); + let location = self.location.as_deref().unwrap_or(DEFAULT_LOCATION); + self.client + .dataset() + .create( + BqDataset::new(&dataset.project_id, &dataset.dataset_id).location(location), + ) + .await + .map_err(|e| client::map_error("create the dataset", e))?; } - } - // client.dataset().create(dataset).await.unwrap(); - - const INIT_SQL: &str = include_str!("./migrations/20240223133248_init.sql"); - - let parametrized_init_sql = INIT_SQL - .replace("{project_id}", &self.project_id) - .replace("{dataset_id}", &self.dataset_id); - - let rs = self - .client - .read() - .await - .job() - .query(&self.project_id, QueryRequest::new(parametrized_init_sql)) - .await?; - - if let Some(total_rows) = rs - .total_rows - .as_deref() - .and_then(|value| value.parse::().ok()) - && total_rows > 0 - { - bail!("BigQuery should not return any rows on the schema creation query"); + Err(e) => return Err(client::map_error("look for the dataset", e)), } + self.reject_legacy_schema().await?; + let sql = include_str!("migrations/init.sql").replace( + "{dataset}", + &format!("{}.{}", dataset.project_id, dataset.dataset_id), + ); + self.execute("create the tables", sql, Vec::new()).await?; Ok(()) } - async fn publish(&self, batch: Arc) -> Result<()> { - let sensors = batch - .sensors - .iter() - .map(|sensor_batch| sensor_batch.sensor.clone()) - .collect::>(); - debug!("BigQuery: Publishing batch with {} sensors", sensors.len()); - let sensor_ids = Arc::new(get_sensor_ids_or_create_sensors(self, &sensors).await?); - - let futures: Vec> + Send>>> = vec![ - Box::pin(publish_integer_values( - self, - batch.clone(), - sensor_ids.clone(), - )), - Box::pin(publish_numeric_values( - self, - batch.clone(), - sensor_ids.clone(), - )), - Box::pin(publish_float_values( - self, - batch.clone(), - sensor_ids.clone(), - )), - Box::pin(publish_string_values( - self, - batch.clone(), - sensor_ids.clone(), - )), - Box::pin(publish_boolean_values( - self, - batch.clone(), - sensor_ids.clone(), - )), - Box::pin(publish_location_values( - self, - batch.clone(), - sensor_ids.clone(), - )), - Box::pin(publish_json_values(self, batch.clone(), sensor_ids.clone())), - Box::pin(publish_blob_values(self, batch.clone(), sensor_ids.clone())), - ]; - debug!("BigQuery: Waiting for all publishers to finish"); - try_join_all(futures).await?; - Ok(()) + async fn publish(&self, batch: Arc) -> Result<()> { + self.publish_batch(&batch).await } async fn vacuum(&self) -> Result<()> { - // Implement vacuum logic here + // Nothing to do: BigQuery compacts its storage by itself Ok(()) } - async fn delete_series(&self, _sensor_uuid: &str) -> Result { - Err(crate::storage::StorageError::Unsupported( - "deleting series is not implemented for BigQuery".to_string(), - ) - .into()) + async fn delete_series(&self, sensor_uuid: &str) -> Result { + let Some((id, sensor)) = self.get_sensor_metadata(sensor_uuid).await? else { + return Ok(false); + }; + // One script, the sensor last: a failure in the middle leaves a series that can be + // deleted again. The id is a parameter of the whole script. + let sql: String = [sample_table(sensor.sensor_type), "labels", "sensors"] + .iter() + .map(|table| format!("DELETE FROM {} WHERE sensor_id = @id;\n", self.table(table))) + .collect(); + self.execute("delete a series", sql, vec![int_param("id", id)]) + .await?; + self.forget_registered(id); + Ok(true) } async fn delete_series_samples( &self, - _sensor_uuid: &str, - _start_time: SensAppDateTime, - _end_time: SensAppDateTime, + sensor_uuid: &str, + start_time: SensAppDateTime, + end_time: SensAppDateTime, ) -> Result> { - Err(crate::storage::StorageError::Unsupported( - "deleting samples is not implemented for BigQuery".to_string(), - ) - .into()) + let Some((id, sensor)) = self.get_sensor_metadata(sensor_uuid).await? else { + return Ok(None); + }; + let micros = |time: &SensAppDateTime| { + reads::clamp_micros(crate::storage::common::datetime_to_micros(time)) + }; + let sql = format!( + "DELETE FROM {} WHERE sensor_id = @id \ + AND timestamp BETWEEN TIMESTAMP_MICROS(@start) AND TIMESTAMP_MICROS(@end)", + self.table(sample_table(sensor.sensor_type)) + ); + let params = vec![ + int_param("id", id), + int_param("start", micros(&start_time)), + int_param("end", micros(&end_time)), + ]; + let deleted = self.execute("delete samples", sql, params).await?; + Ok(Some(deleted)) } async fn list_series( &self, - _metric_filter: Option<&str>, - _limit: Option, - _bookmark: Option<&str>, - ) -> Result { - // TODO: Implement pagination for BigQuery backend - // TODO: Implement metric_filter support for BigQuery backend - // For now, ignore limit, bookmark, and metric_filter parameters and return all results - use crate::datamodel::{Sensor, sensapp_vec::SensAppLabels, unit::Unit}; - use gcp_bigquery_client::model::query_request::QueryRequest; - use smallvec::smallvec; - use std::str::FromStr; - use uuid::Uuid; - - let query = format!( - r#" - SELECT s.sensor_id, s.uuid AS sensor_uuid, s.name AS sensor_name, s.type AS sensor_type, u.name AS unit_name, u.description AS unit_description - FROM `{}.{}.sensors` s - LEFT JOIN `{}.{}.units` u ON s.unit = u.id - ORDER BY s.uuid ASC - "#, - self.project_id, self.dataset_id, self.project_id, self.dataset_id - ); - - let rs = self - .client - .read() - .await - .job() - .query(&self.project_id, QueryRequest::new(query)) - .await?; - let mut rs = ResultSet::new_from_query_response(rs); - - let mut sensors = Vec::new(); - - while rs.next_row() { - let sensor_id = rs - .get_i64_by_name("sensor_id")? - .context("BigQuery row missing sensor_id")?; - let sensor_uuid = Uuid::from_str( - &rs.get_string_by_name("sensor_uuid")? - .context("BigQuery row missing sensor_uuid")?, - )?; - let sensor_name = rs - .get_string_by_name("sensor_name")? - .context("BigQuery row missing sensor_name")?; - let sensor_type = Self::parse_sensor_type( - &rs.get_string_by_name("sensor_type")? - .context("BigQuery row missing sensor_type")?, - )?; - let unit_name = rs.get_string_by_name("unit_name")?; - let unit_description = rs.get_string_by_name("unit_description")?; - let unit = unit_name.map(|name| Unit::new(name, unit_description)); - - // Query labels for this sensor - let labels_query = format!( - r#" - SELECT lnd.name as label_name, ldd.description as label_value - FROM `{}.{}.labels` l - JOIN `{}.{}.labels_name_dictionary` lnd ON l.name = lnd.id - JOIN `{}.{}.labels_description_dictionary` ldd ON l.description = ldd.id - WHERE l.sensor_id = {} - "#, - self.project_id, - self.dataset_id, - self.project_id, - self.dataset_id, - self.project_id, - self.dataset_id, - sensor_id - ); - - let labels_rs = self - .client - .read() - .await - .job() - .query(&self.project_id, QueryRequest::new(labels_query)) - .await?; - let mut labels_rs = ResultSet::new_from_query_response(labels_rs); - - let mut labels: SensAppLabels = smallvec![]; - while labels_rs.next_row() { - let label_name = labels_rs - .get_string_by_name("label_name")? - .context("BigQuery row missing label_name")?; - let label_value = labels_rs - .get_string_by_name("label_value")? - .context("BigQuery row missing label_value")?; - labels.push((label_name, label_value)); - } - - let sensor = Sensor::new(sensor_uuid, sensor_name, sensor_type, unit, Some(labels)); - - sensors.push(sensor); + metric_filter: Option<&str>, + limit: Option, + bookmark: Option<&str>, + ) -> Result { + let after: Option = bookmark + .map(|bookmark| { + bookmark.parse().map_err(|error| { + anyhow::Error::from(StorageError::invalid_data_format( + &format!("Invalid bookmark format: {error}"), + None, + None, + )) + }) + }) + .transpose()?; + let limit = limit + .unwrap_or(DEFAULT_LIST_SERIES_LIMIT) + .min(MAX_LIST_SERIES_LIMIT); + + let mut conditions = Vec::new(); + let mut params = Vec::new(); + if let Some(metric) = metric_filter { + conditions.push("s.name = @metric".to_string()); + params.push(string_param("metric", metric)); + } + if let Some(after) = after { + conditions.push("s.sensor_id > @bookmark".to_string()); + params.push(int_param("bookmark", after)); } + // One more than the page, to know whether there is a next one + let tail = format!("ORDER BY s.sensor_id LIMIT {}", limit.saturating_add(1)); + let mut found = self + .read_sensors(self.dataset.sensors_sql(&conditions, &tail), params) + .await?; - Ok(crate::storage::ListSeriesResult { - series: sensors, - bookmark: None, + let has_more = found.len() > limit; + found.truncate(limit); + let bookmark = has_more + .then(|| found.last().map(|(id, _)| id.to_string())) + .flatten(); + Ok(ListSeriesResult { + series: found.into_iter().map(|(_, sensor)| sensor).collect(), + bookmark, }) } - async fn list_metrics(&self) -> Result> { - use crate::datamodel::{Metric, unit::Unit}; - - let query = format!( - r#" - SELECT s.name AS metric_name, s.type AS sensor_type, u.name AS unit_name, u.description AS unit_description, COUNT(*) AS series_count - FROM `{}.{}.sensors` s - LEFT JOIN `{}.{}.units` u ON s.unit = u.id - GROUP BY s.name, s.type, u.name, u.description - ORDER BY s.name ASC - "#, - self.project_id, self.dataset_id, self.project_id, self.dataset_id + async fn list_metrics(&self) -> Result> { + let sql = format!( + "SELECT s.name AS metric_name, s.type AS sensor_type, \ + u.name AS unit_name, u.description AS unit_description, COUNT(*) AS series_count \ + FROM {} s LEFT JOIN {} u ON s.unit = u.id \ + GROUP BY s.name, s.type, u.name, u.description \ + ORDER BY s.name, s.type, u.name", + self.dataset.sensors_set(), + self.dataset.units_set() ); - - let rs = self - .client - .read() - .await - .job() - .query(&self.project_id, QueryRequest::new(query)) - .await?; - let mut rs = ResultSet::new_from_query_response(rs); - let mut metrics = Vec::new(); - - while rs.next_row() { - let metric_name = rs - .get_string_by_name("metric_name")? - .context("BigQuery row missing metric_name")?; - let sensor_type = Self::parse_sensor_type( - &rs.get_string_by_name("sensor_type")? - .context("BigQuery row missing sensor_type")?, - )?; - let unit_name = rs.get_string_by_name("unit_name")?; - let unit_description = rs.get_string_by_name("unit_description")?; - let unit = unit_name.map(|name| Unit::new(name, unit_description)); - let series_count = rs - .get_i64_by_name("series_count")? - .context("BigQuery row missing series_count")?; - - metrics.push(Metric::new( - metric_name, + self.query_rows("list metrics", sql, Vec::new(), |row| { + let name = required(row.get_string_by_name("metric_name"), "metric_name")?; + let sensor_type = required(row.get_string_by_name("sensor_type"), "sensor_type")?; + let sensor_type = SensorType::from_str(&sensor_type).map_err(|error| { + StorageError::invalid_data_format( + &format!("Failed to parse sensor type '{sensor_type}': {error}"), + None, + Some(&name), + ) + })?; + let unit = row + .get_string_by_name("unit_name")? + .map(|unit_name| -> Result { + Ok(Unit::new( + unit_name, + row.get_string_by_name("unit_description")?, + )) + }) + .transpose()?; + let series_count = required(row.get_i64_by_name("series_count"), "series_count")?; + Ok(Metric::new( + name, sensor_type, unit, series_count, Vec::new(), - )); - } - - Ok(metrics) + )) + }) + .await } async fn query_sensor_data( &self, sensor_uuid: &str, - _start_time: Option, - _end_time: Option, - _limit: Option, - ) -> Result> { - use crate::datamodel::{Sensor, SensorData, sensapp_vec::SensAppLabels, unit::Unit}; - use gcp_bigquery_client::model::query_request::QueryRequest; - use smallvec::smallvec; - - // Query sensor metadata by UUID - let sensor_query = format!( - r#" - SELECT s.sensor_id, s.uuid AS sensor_uuid, s.name AS sensor_name, s.type AS sensor_type, u.name AS unit_name, u.description AS unit_description - FROM `{}.{}.sensors` s - LEFT JOIN `{}.{}.units` u ON s.unit = u.id - WHERE s.uuid = '{}' - "#, - self.project_id, self.dataset_id, self.project_id, self.dataset_id, sensor_uuid - ); - - let sensor_rs = self - .client - .read() - .await - .job() - .query(&self.project_id, QueryRequest::new(sensor_query)) + start_time: Option, + end_time: Option, + limit: Option, + ) -> Result> { + let Some((id, sensor)) = self.get_sensor_metadata(sensor_uuid).await? else { + return Ok(None); + }; + let samples = self + .query_samples_by_type( + id, + sensor.sensor_type, + start_time, + end_time, + limit.unwrap_or(DEFAULT_QUERY_LIMIT), + false, + ) .await?; - let mut sensor_rs = ResultSet::new_from_query_response(sensor_rs); - if !sensor_rs.next_row() { + Ok(Some(SensorData::new(sensor, samples))) + } + + /// Aggregations run in BigQuery for the numeric series; the `limit` counts the buckets. The + /// other types have no aggregation, as on the other backends. + async fn query_sensor_data_advanced( + &self, + sensor_uuid: &str, + options: &SensorDataQueryOptions, + ) -> Result> { + options.validate()?; + let (Some(step_ms), Some(aggregation)) = (options.step_ms, options.aggregation) else { + let raw = self + .query_sensor_data( + sensor_uuid, + options.start_time, + options.end_time, + options.limit, + ) + .await?; + return raw + .map(|data| crate::storage::common::apply_query_options(data, options)) + .transpose(); + }; + + let Some((id, mut sensor)) = self.get_sensor_metadata(sensor_uuid).await? else { return Ok(None); + }; + if !matches!( + sensor.sensor_type, + SensorType::Integer | SensorType::Numeric | SensorType::Float + ) { + anyhow::bail!("aggregation is only supported for numeric series"); + } + let micros = |time: &SensAppDateTime| crate::storage::common::datetime_to_micros(time); + let read = AggregatedRead { + start_us: options.start_time.as_ref().map(micros), + end_us: options.end_time.as_ref().map(micros), + step_ms, + aggregation, + }; + let limit = options.limit.unwrap_or(DEFAULT_QUERY_LIMIT); + let kind = aggregation::kind_type(aggregated_kind(sensor.sensor_type, aggregation)); + let samples = self + .query_aggregated_of_many(sensor.sensor_type, &[id], &read, limit) + .await? + .remove(&id) + .unwrap_or_else(|| empty_samples(kind)); + sensor.sensor_type = kind; + if aggregation.output_is_count() { + sensor.unit = None; } - let sensor_id = sensor_rs - .get_i64_by_name("sensor_id")? - .context("BigQuery row missing sensor_id")?; - let sensor_uuid = uuid::Uuid::parse_str( - &sensor_rs - .get_string_by_name("sensor_uuid")? - .context("BigQuery row missing sensor_uuid")?, - )?; - let sensor_name = sensor_rs - .get_string_by_name("sensor_name")? - .context("BigQuery row missing sensor_name")?; - let sensor_type = Self::parse_sensor_type( - &sensor_rs - .get_string_by_name("sensor_type")? - .context("BigQuery row missing sensor_type")?, - )?; - let unit_name = sensor_rs.get_string_by_name("unit_name")?; - let unit_description = sensor_rs.get_string_by_name("unit_description")?; - let unit = unit_name.map(|name| Unit::new(name, unit_description)); - - // Query labels - let labels_query = format!( - r#" - SELECT lnd.name as label_name, ldd.description as label_value - FROM `{}.{}.labels` l - JOIN `{}.{}.labels_name_dictionary` lnd ON l.name = lnd.id - JOIN `{}.{}.labels_description_dictionary` ldd ON l.description = ldd.id - WHERE l.sensor_id = {} - "#, - self.project_id, - self.dataset_id, - self.project_id, - self.dataset_id, - self.project_id, - self.dataset_id, - sensor_id - ); + // What is left to do is the simplification, on the buckets + let simplify_only = SensorDataQueryOptions { + start_time: options.start_time, + end_time: options.end_time, + limit: options.limit, + step_ms: None, + aggregation: None, + simplify: options.simplify, + }; + crate::storage::common::apply_query_options( + SensorData::new(sensor, samples), + &simplify_only, + ) + .map(Some) + } - let labels_rs = self - .client - .read() - .await - .job() - .query(&self.project_id, QueryRequest::new(labels_query)) + async fn query_sensor_data_latest( + &self, + sensor_uuid: &str, + start_time: Option, + end_time: Option, + ) -> Result> { + let Some((id, sensor)) = self.get_sensor_metadata(sensor_uuid).await? else { + return Ok(None); + }; + let samples = self + .query_samples_by_type(id, sensor.sensor_type, start_time, end_time, 1, true) .await?; - let mut labels_rs = ResultSet::new_from_query_response(labels_rs); - - let mut labels: SensAppLabels = smallvec![]; - while labels_rs.next_row() { - let label_name = labels_rs - .get_string_by_name("label_name")? - .context("BigQuery row missing label_name")?; - let label_value = labels_rs - .get_string_by_name("label_value")? - .context("BigQuery row missing label_value")?; - labels.push((label_name, label_value)); + if samples.is_empty() { + return Ok(None); } - - let sensor = Sensor::new( - sensor_uuid, - sensor_name.to_string(), - sensor_type, - unit, - Some(labels), - ); - - // For BigQuery, we'll return sensor metadata only for now - // Sample querying would require complex BigQuery-specific logic - let samples = crate::datamodel::TypedSamples::Integer(smallvec![]); - Ok(Some(SensorData::new(sensor, samples))) } async fn query_sensors_by_labels( &self, - _matchers: &[super::LabelMatcher], - _start_time: Option, - _end_time: Option, - _limit: Option, - _numeric_only: bool, + matchers: &[LabelMatcher], + start_time: Option, + end_time: Option, + limit: Option, + numeric_only: bool, ) -> Result> { - // TODO: Implement label-based query for BigQuery - anyhow::bail!("query_sensors_by_labels not yet implemented for BigQuery") + if matchers.is_empty() { + return Ok(Vec::new()); + } + let sensors = self + .find_sensors_by_matchers(matchers, numeric_only, None) + .await?; + let limit = limit.unwrap_or(DEFAULT_QUERY_LIMIT); + // In the order of the sensors, a few at a time: every read is a round trip + stream::iter(sensors) + .map(|(id, sensor)| async move { + let samples = self + .query_samples_by_type( + id, + sensor.sensor_type, + start_time, + end_time, + limit, + false, + ) + .await?; + Ok(SensorData::new(sensor, samples)) + }) + .buffered(READ_CONCURRENCY) + .collect::>>() + .await + .into_iter() + .collect() + } + + async fn query_selector( + &self, + matchers: &[LabelMatcher], + start_time: Option, + end_time: Option, + numeric_only: bool, + max_series: usize, + max_samples: usize, + ) -> Result { + crate::storage::selector::read_selector_in_bulk( + self, + matchers, + start_time, + end_time, + numeric_only, + max_series, + max_samples, + ) + .await + } + + async fn query_selector_aggregated( + &self, + matchers: &[LabelMatcher], + options: &SensorDataQueryOptions, + max_series: usize, + max_samples: usize, + ) -> Result { + crate::storage::selector::read_aggregated_selector_in_bulk( + self, + matchers, + options, + max_series, + max_samples, + ) + .await } - /// Health check for BigQuery storage - /// Executes a simple SELECT 1 query to verify BigQuery connectivity async fn health_check(&self) -> Result<()> { - let query = "SELECT 1".to_string(); - self.client - .read() - .await - .job() - .query(&self.project_id, QueryRequest::new(query)) - .await - .context("BigQuery health check failed")?; + tokio::time::timeout( + HEALTH_CHECK_TIMEOUT, + self.run_query("health check", "SELECT 1".to_string(), Vec::new()), + ) + .await + .map_err(|_| StorageError::Unavailable("health check timed out".to_string()))? + .context("BigQuery health check failed")?; Ok(()) } - /// Clean up all test data from the database (BigQuery implementation) #[cfg(any(test, feature = "test-utils"))] async fn cleanup_test_data(&self) -> Result<()> { - // BigQuery doesn't support traditional TRUNCATE/DELETE operations well - // For now, this is a no-op since tests typically use separate datasets - // In a real implementation, you might recreate the dataset or use partitioned tables + // DELETE, not TRUNCATE: recent rows of the Storage Write API can be deleted, and a delete + // that matches whole partitions only touches metadata. + let tables = crate::storage::common::VALUE_TABLES + .into_iter() + .chain(["labels", "sensors", "units"]); + futures::future::try_join_all(tables.map(|table| { + self.execute( + "clean the tables", + format!("DELETE FROM {} WHERE TRUE", self.table(table)), + Vec::new(), + ) + })) + .await?; + self.registered().clear(); Ok(()) } } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn registered_ids_expire_and_are_bounded() { + let mut ids = RegisteredIds::new(Duration::from_secs(60), 2); + ids.insert(1); + assert!(ids.contains(1) && !ids.contains(2)); + ids.remove(1); + assert!(!ids.contains(1)); + + // Too many: forgotten all at once, never a growing map + ids.insert(1); + ids.insert(2); + ids.insert(3); + assert!(ids.contains(3) && ids.seen.len() <= 2); + ids.clear(); + assert!(!ids.contains(3)); + + let mut short = RegisteredIds::new(Duration::ZERO, 10); + short.insert(1); + assert!(!short.contains(1), "expired at once"); + } + + #[test] + fn the_legacy_tables_are_not_in_the_schema() { + let schema = include_str!("migrations/init.sql"); + for table in LEGACY_TABLES { + assert!(!schema.contains(table), "{table}"); + } + } +} diff --git a/src/storage/bigquery/publishers.rs b/src/storage/bigquery/publishers.rs new file mode 100644 index 00000000..c0bab8e7 --- /dev/null +++ b/src/storage/bigquery/publishers.rs @@ -0,0 +1,436 @@ +//! Writing: the registration of new series and the samples of a batch, through the default stream of +//! the Storage Write API (rows are queryable at once, delivery is at least once). + +use super::BigQueryStorage; +use super::client::{grpc_code_is_transient, int_array_param, map_error}; +use super::rows::{self, *}; +use crate::datamodel::{Sensor, TypedSamples, batch::Batch}; +use crate::storage::StorageError; +use crate::storage::common::{datetime_to_micros, unit_name_to_id, uuid_to_sensor_id}; +use anyhow::Result; +use futures::future::try_join_all; +use gcp_bigquery_client::google::cloud::bigquery::storage::v1::append_rows_response::Response; +use gcp_bigquery_client::storage::{StreamName, TableBatch, TableDescriptor}; +use prost::Message; +use rust_decimal::{Decimal, RoundingStrategy}; +use std::collections::HashMap; +use std::sync::Arc; + +/// NUMERIC has 9 decimal digits. +const NUMERIC_SCALE: u32 = 9; + +/// The id of a series in the tables: the bits of the 64-bit key shared with ClickHouse. +pub fn uuid_id(uuid: &uuid::Uuid) -> i64 { + uuid_to_sensor_id(uuid) as i64 +} + +pub fn sensor_id(sensor: &Sensor) -> i64 { + uuid_id(&sensor.uuid) +} + +pub fn unit_id(name: &str) -> i64 { + unit_name_to_id(name) as i64 +} + +/// A decimal as BigQuery reads it in a NUMERIC column: rounded to 9 digits. +pub fn numeric_text(value: &Decimal) -> String { + value + .round_dp_with_strategy(NUMERIC_SCALE, RoundingStrategy::MidpointNearestEven) + .to_string() +} + +/// The rows of a batch, by table. +#[derive(Default)] +struct SampleRows { + integer: Vec, + numeric: Vec, + float: Vec, + string: Vec, + boolean: Vec, + location: Vec, + json: Vec, + blob: Vec, +} + +async fn sample_rows(batch: &Batch) -> SampleRows { + let mut rows = SampleRows::default(); + for single in batch.sensors.iter() { + let id = sensor_id(&single.sensor); + let samples = single.samples.read().await; + match &*samples { + TypedSamples::Integer(samples) => { + rows.integer.extend(samples.iter().map(|s| IntegerValueRow { + sensor_id: id, + timestamp: datetime_to_micros(&s.datetime), + value: s.value, + })) + } + TypedSamples::Numeric(samples) => { + rows.numeric.extend(samples.iter().map(|s| NumericValueRow { + sensor_id: id, + timestamp: datetime_to_micros(&s.datetime), + value: numeric_text(&s.value), + })) + } + TypedSamples::Float(samples) => { + rows.float.extend(samples.iter().map(|s| FloatValueRow { + sensor_id: id, + timestamp: datetime_to_micros(&s.datetime), + value: s.value, + })) + } + TypedSamples::String(samples) => { + rows.string.extend(samples.iter().map(|s| StringValueRow { + sensor_id: id, + timestamp: datetime_to_micros(&s.datetime), + value: s.value.clone(), + })) + } + TypedSamples::Boolean(samples) => { + rows.boolean.extend(samples.iter().map(|s| BooleanValueRow { + sensor_id: id, + timestamp: datetime_to_micros(&s.datetime), + value: s.value, + })) + } + TypedSamples::Location(samples) => { + rows.location + .extend(samples.iter().map(|s| LocationValueRow { + sensor_id: id, + timestamp: datetime_to_micros(&s.datetime), + latitude: s.value.y(), + longitude: s.value.x(), + })) + } + TypedSamples::Json(samples) => rows.json.extend(samples.iter().map(|s| JsonValueRow { + sensor_id: id, + timestamp: datetime_to_micros(&s.datetime), + value: s.value.to_string(), + })), + TypedSamples::Blob(samples) => rows.blob.extend(samples.iter().map(|s| BlobValueRow { + sensor_id: id, + timestamp: datetime_to_micros(&s.datetime), + value: s.value.clone(), + })), + } + } + rows +} + +impl BigQueryStorage { + /// Append rows to a table. A request over 10 MB is cut in several by the client; the answers are + /// read all the way: a rejected append says so in the `error` or the `row_errors` of its + /// answer, not only in the status of the call. + pub(super) async fn append( + &self, + table: &'static str, + descriptor: &Arc, + rows: Vec, + ) -> Result<()> { + if rows.is_empty() { + return Ok(()); + } + let stream = StreamName::new_default( + self.dataset.project_id.clone(), + self.dataset.dataset_id.clone(), + table.to_string(), + ); + let operation = format!("write to {table}"); + let trace_id = format!("sensapp-{table}-{}", uuid::Uuid::new_v4()); + let results = self + .client + .storage() + .append_table_batches_concurrent( + vec![TableBatch::new(stream, descriptor.clone(), rows)], + 1, + &trace_id, + ) + .await + .map_err(|error| map_error(&operation, error))?; + + for result in results { + for response in result.responses { + let response = response.map_err(|status| { + map_error( + &operation, + gcp_bigquery_client::error::BQError::from(status), + ) + })?; + if let Some(Response::Error(status)) = response.response { + return Err(append_failure(&operation, status.code, &status.message)); + } + if let Some(row_error) = response.row_errors.first() { + return Err(StorageError::OperationFailed { + operation: format!("BigQuery {operation}"), + details: format!( + "row {} rejected: {} ({} rows rejected in this append)", + row_error.index, + row_error.message, + response.row_errors.len() + ), + } + .into()); + } + } + } + Ok(()) + } + + /// Write the series that are not stored yet: their units, their labels, and last the sensors + /// themselves, so a series is never listed with a part of its labels missing. Two writers can + /// register the same series at the same time; the rows are identical and the reads collapse them. + pub(super) async fn register_sensors(&self, sensors: &[&Sensor]) -> Result<()> { + let mut by_id: HashMap = HashMap::with_capacity(sensors.len()); + for sensor in sensors { + by_id.entry(sensor_id(sensor)).or_insert(sensor); + } + by_id.retain(|id, _| !self.is_registered(*id)); + if by_id.is_empty() { + return Ok(()); + } + + let ids: Vec = by_id.keys().copied().collect(); + let stored = self.existing_ids("sensors", "sensor_id", &ids).await?; + for id in &stored { + self.remember_registered(*id); + by_id.remove(id); + } + if by_id.is_empty() { + return Ok(()); + } + + let mut units: HashMap = HashMap::new(); + for sensor in by_id.values() { + if let Some(unit) = &sensor.unit { + units.entry(unit_id(&unit.name)).or_insert(unit); + } + } + let unit_ids: Vec = units.keys().copied().collect(); + let stored_units = self.existing_ids("units", "id", &unit_ids).await?; + let unit_rows: Vec = units + .iter() + .filter(|(id, _)| !stored_units.contains(id)) + .map(|(id, unit)| UnitRow { + id: *id, + name: unit.name.clone(), + description: unit.description.clone(), + }) + .collect(); + let label_rows: Vec = by_id + .iter() + .flat_map(|(id, sensor)| { + sensor.labels.iter().map(|(name, description)| LabelRow { + sensor_id: *id, + name: name.clone(), + description: description.clone(), + }) + }) + .collect(); + tokio::try_join!( + self.append("units", &rows::UNITS, unit_rows), + self.append("labels", &rows::LABELS, label_rows), + )?; + + let sensor_rows: Vec = by_id + .iter() + .map(|(id, sensor)| SensorRow { + sensor_id: *id, + uuid: sensor.uuid.to_string(), + name: sensor.name.clone(), + r#type: sensor.sensor_type.to_string(), + unit: sensor.unit.as_ref().map(|unit| unit_id(&unit.name)), + }) + .collect(); + self.append("sensors", &rows::SENSORS, sensor_rows).await?; + for id in by_id.keys() { + self.remember_registered(*id); + } + Ok(()) + } + + /// Which of the `ids` are in the column `column` of `table`. + async fn existing_ids(&self, table: &str, column: &str, ids: &[i64]) -> Result> { + if ids.is_empty() { + return Ok(Vec::new()); + } + let sql = format!( + "SELECT DISTINCT {column} FROM {} WHERE {column} IN UNNEST(@ids)", + self.table(table) + ); + self.query_rows( + &format!("look up {table}"), + sql, + vec![int_array_param("ids", ids)], + |row| super::client::required(row.get_i64(0), column), + ) + .await + } + + pub(super) async fn publish_batch(&self, batch: &Batch) -> Result<()> { + let sensors: Vec<&Sensor> = batch.sensors.iter().map(|s| s.sensor.as_ref()).collect(); + self.register_sensors(&sensors).await?; + + let rows = sample_rows(batch).await; + try_join_all([ + Box::pin(self.append("integer_values", &rows::INTEGER_VALUES, rows.integer)) + as std::pin::Pin> + Send + '_>>, + Box::pin(self.append("numeric_values", &rows::NUMERIC_VALUES, rows.numeric)), + Box::pin(self.append("float_values", &rows::FLOAT_VALUES, rows.float)), + Box::pin(self.append("string_values", &rows::STRING_VALUES, rows.string)), + Box::pin(self.append("boolean_values", &rows::BOOLEAN_VALUES, rows.boolean)), + Box::pin(self.append("location_values", &rows::LOCATION_VALUES, rows.location)), + Box::pin(self.append("json_values", &rows::JSON_VALUES, rows.json)), + Box::pin(self.append("blob_values", &rows::BLOB_VALUES, rows.blob)), + ]) + .await?; + Ok(()) + } +} + +/// An append rejected by the service: `code` is a `google.rpc.Code`. +fn append_failure(operation: &str, code: i32, message: &str) -> anyhow::Error { + if grpc_code_is_transient(code) { + StorageError::Unavailable(format!("{operation}: {message}")).into() + } else { + StorageError::OperationFailed { + operation: format!("BigQuery {operation}"), + details: format!("code {code}: {message}"), + } + .into() + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::datamodel::{Sample, SensAppDateTime, SensorType, batch::SingleSensorBatch}; + use hifitime::Epoch; + use smallvec::smallvec; + use std::str::FromStr; + use tokio::sync::RwLock; + use uuid::Uuid; + + fn single(sensor_type: SensorType, samples: TypedSamples) -> SingleSensorBatch { + SingleSensorBatch { + sensor: Arc::new(Sensor::new( + Uuid::from_u128(0x0123_4567_89ab_cdef_fedc_ba98_7654_3210), + "test".to_string(), + sensor_type, + None, + None, + )), + samples: RwLock::new(samples), + } + } + + fn at(seconds: f64) -> SensAppDateTime { + Epoch::from_unix_seconds(seconds) + } + + #[test] + fn numeric_values_are_rounded_to_the_scale_of_the_column() { + let text = |value: &str| numeric_text(&Decimal::from_str(value).unwrap()); + assert_eq!(text("1.5"), "1.5"); + assert_eq!(text("-0.000000001"), "-0.000000001"); + assert_eq!(text("1.0000000014"), "1.000000001"); + assert_eq!(text("1.0000000016"), "1.000000002"); + assert_eq!( + text("79228162514264337593543950335"), + "79228162514264337593543950335" + ); + } + + #[test] + fn sensor_ids_are_the_clickhouse_ids_as_signed_integers() { + let sensor = Sensor::new( + Uuid::parse_str("9d87123d-9b47-466d-9eda-001c2ecf9c54").unwrap(), + "s".to_string(), + SensorType::Float, + None, + None, + ); + assert_eq!( + sensor_id(&sensor) as u64, + 0x9d87_123d_9b47_466d ^ 0x9eda_001c_2ecf_9c54 + ); + let negative = Sensor::new( + Uuid::parse_str("00000000-0000-4000-8000-000000000001").unwrap(), + "s".to_string(), + SensorType::Float, + None, + None, + ); + assert_eq!( + sensor_id(&negative), + i64::MIN + 0x4001, + "the top bit is the sign" + ); + } + + #[tokio::test] + async fn every_type_goes_to_its_table_with_microseconds_and_full_precision() { + let precise = 0.1 + 0.2; // not representable as an f32 + let batch = Batch { + sensors: smallvec![ + single( + SensorType::Float, + TypedSamples::Float(smallvec![Sample { + datetime: at(1_700_000_000.5), + value: precise + }]) + ), + single( + SensorType::Location, + TypedSamples::Location(smallvec![Sample { + datetime: at(1.0), + value: geo::Point::new(10.123_456_789_012, 59.987_654_321_098) + }]) + ), + single( + SensorType::Json, + TypedSamples::Json(smallvec![Sample { + datetime: at(2.0), + value: serde_json::json!({"a": [1, 2], "b": null}) + }]) + ), + single( + SensorType::Numeric, + TypedSamples::Numeric(smallvec![Sample { + datetime: at(3.0), + value: Decimal::from_str("12.3456789012").unwrap() + }]) + ), + single( + SensorType::Blob, + TypedSamples::Blob(smallvec![Sample { + datetime: at(4.0), + value: vec![0, 255, 7] + }]) + ), + ], + }; + let rows = sample_rows(&batch).await; + assert_eq!(rows.float[0].value, precise); + assert_eq!(rows.float[0].timestamp, 1_700_000_000_500_000); + assert_eq!(rows.location[0].latitude, 59.987_654_321_098); + assert_eq!(rows.location[0].longitude, 10.123_456_789_012); + assert_eq!(rows.json[0].value, r#"{"a":[1,2],"b":null}"#); + assert_eq!(rows.numeric[0].value, "12.345678901"); + assert_eq!(rows.blob[0].value, vec![0, 255, 7]); + assert!(rows.integer.is_empty() && rows.string.is_empty() && rows.boolean.is_empty()); + } + + #[test] + fn a_rejected_append_is_an_error_a_busy_service_is_unavailable() { + let busy = append_failure("write to sensors", 14, "unavailable"); + assert!(matches!( + busy.downcast_ref::(), + Some(StorageError::Unavailable(_)) + )); + let rejected = append_failure("write to sensors", 3, "schema mismatch"); + assert!(matches!( + rejected.downcast_ref::(), + Some(StorageError::OperationFailed { .. }) + )); + } +} diff --git a/src/storage/bigquery/reads.rs b/src/storage/bigquery/reads.rs new file mode 100644 index 00000000..19c36a43 --- /dev/null +++ b/src/storage/bigquery/reads.rs @@ -0,0 +1,479 @@ +//! Reading: sensors and their labels, and the samples of the eight types. + +use super::BigQueryStorage; +use super::client::{int_array_param, int_param, required, string_param}; +use super::publishers::uuid_id; +use crate::datamodel::sensapp_datetime::SensAppDateTimeExt; +use crate::datamodel::sensapp_vec::SensAppLabels; +use crate::datamodel::unit::Unit; +use crate::datamodel::{Sample, SensAppDateTime, Sensor, SensorType, TypedSamples}; +use crate::storage::StorageError; +use crate::storage::selector::empty_samples; +use anyhow::Result; +use gcp_bigquery_client::model::{query_parameter::QueryParameter, query_response::ResultSet}; +use rust_decimal::Decimal; +use std::collections::HashMap; +use std::str::FromStr; +use uuid::Uuid; + +/// Sensor ids sent to BigQuery in one query for their labels +const LABEL_LOOKUP_CHUNK: usize = 2000; + +/// `LIMIT` takes an INT64 +pub(super) const MAX_LIMIT: usize = i64::MAX as usize; + +/// BigQuery's TIMESTAMP goes from year 1 to year 9999 +const MIN_MICROS: i64 = -62_135_596_800_000_000; +const MAX_MICROS: i64 = 253_402_300_799_999_999; + +/// A bound of a time window, that `TIMESTAMP_MICROS` accepts. +pub fn clamp_micros(micros: i64) -> i64 { + micros.clamp(MIN_MICROS, MAX_MICROS) +} + +pub fn parse_uuid(sensor_uuid: &str) -> Result { + Uuid::from_str(sensor_uuid).map_err(|error| { + StorageError::invalid_data_format( + &format!("Invalid UUID '{sensor_uuid}': {error}"), + None, + None, + ) + .into() + }) +} + +/// The table of the samples of a type. +pub fn sample_table(sensor_type: SensorType) -> &'static str { + match sensor_type { + SensorType::Integer => "integer_values", + SensorType::Numeric => "numeric_values", + SensorType::Float => "float_values", + SensorType::String => "string_values", + SensorType::Boolean => "boolean_values", + SensorType::Location => "location_values", + SensorType::Json => "json_values", + SensorType::Blob => "blob_values", + } +} + +/// The value columns that `read_sample` expects after the sensor id and the timestamp. +fn sample_columns(sensor_type: SensorType) -> &'static str { + match sensor_type { + SensorType::Location => "latitude, longitude", + SensorType::Json => "TO_JSON_STRING(value) AS value", + SensorType::Blob => "TO_BASE64(value) AS value", + _ => "value", + } +} + +fn parse_error(what: &str, error: impl std::fmt::Display) -> anyhow::Error { + StorageError::invalid_data_format(&format!("Failed to parse {what}: {error}"), None, None) + .into() +} + +/// Add the sample of a row `sensor_id, timestamp_us, value...` to `samples`. +pub fn read_sample(samples: &mut TypedSamples, row: &ResultSet) -> Result<()> { + let datetime = + SensAppDateTime::from_unix_microseconds_i64(required(row.get_i64(1), "timestamp")?); + match samples { + TypedSamples::Integer(samples) => samples.push(Sample { + datetime, + value: required(row.get_i64(2), "value")?, + }), + TypedSamples::Numeric(samples) => { + let text = required(row.get_string(2), "value")?; + let value = Decimal::from_str(&text) + .or_else(|_| Decimal::from_scientific(&text)) + .map_err(|error| parse_error("a NUMERIC value", error))?; + samples.push(Sample { datetime, value }); + } + TypedSamples::Float(samples) => samples.push(Sample { + datetime, + value: required(row.get_f64(2), "value")?, + }), + TypedSamples::String(samples) => samples.push(Sample { + datetime, + value: required(row.get_string(2), "value")?, + }), + TypedSamples::Boolean(samples) => samples.push(Sample { + datetime, + value: required(row.get_bool(2), "value")?, + }), + TypedSamples::Location(samples) => samples.push(Sample { + datetime, + value: geo::Point::new( + required(row.get_f64(3), "longitude")?, + required(row.get_f64(2), "latitude")?, + ), + }), + TypedSamples::Json(samples) => { + let text = required(row.get_string(2), "value")?; + samples.push(Sample { + datetime, + value: serde_json::from_str(&text) + .map_err(|error| parse_error("a JSON value", error))?, + }); + } + TypedSamples::Blob(samples) => { + use base64::Engine; + let text = required(row.get_string(2), "value")?; + samples.push(Sample { + datetime, + value: base64::engine::general_purpose::STANDARD + .decode(text) + .map_err(|error| parse_error("a BYTES value", error))?, + }); + } + } + Ok(()) +} + +/// The `WHERE` conditions of a window on the timestamp, with the parameters `start` and `end`. +pub(super) fn window( + start_us: Option, + end_us: Option, + params: &mut Vec, +) -> String { + let mut conditions = String::new(); + if let Some(start_us) = start_us { + conditions.push_str(" AND timestamp >= TIMESTAMP_MICROS(@start)"); + params.push(int_param("start", clamp_micros(start_us))); + } + if let Some(end_us) = end_us { + conditions.push_str(" AND timestamp <= TIMESTAMP_MICROS(@end)"); + params.push(int_param("end", clamp_micros(end_us))); + } + conditions +} + +/// Where the tables are, and the statements that name them. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Dataset { + pub project_id: String, + pub dataset_id: String, +} + +impl Dataset { + /// The fully qualified, quoted name of a table of the dataset. + pub fn table(&self, name: &str) -> String { + format!("`{}.{}.{}`", self.project_id, self.dataset_id, name) + } + + /// The sensors table as a set: a sensor registered twice at once is one sensor. + pub fn sensors_set(&self) -> String { + format!( + "(SELECT sensor_id, ANY_VALUE(uuid) AS uuid, ANY_VALUE(name) AS name, \ + ANY_VALUE(type) AS type, ANY_VALUE(unit) AS unit \ + FROM {} GROUP BY sensor_id)", + self.table("sensors") + ) + } + + pub fn units_set(&self) -> String { + format!( + "(SELECT id, ANY_VALUE(name) AS name, ANY_VALUE(description) AS description \ + FROM {} GROUP BY id)", + self.table("units") + ) + } + + /// `SELECT` the sensors (`s`) and their units, then `conditions` and `tail` (order, limit). + pub fn sensors_sql(&self, conditions: &[String], tail: &str) -> String { + let mut sql = format!( + "SELECT s.sensor_id, s.uuid, s.name, s.type, \ + u.name AS unit_name, u.description AS unit_description \ + FROM {} s LEFT JOIN {} u ON s.unit = u.id", + self.sensors_set(), + self.units_set() + ); + if !conditions.is_empty() { + sql.push_str(" WHERE "); + sql.push_str(&conditions.join(" AND ")); + } + sql.push(' '); + sql.push_str(tail); + sql + } +} + +impl BigQueryStorage { + /// Read sensors with a query shaped by `sensors_sql`, and their labels with one more query. + pub(super) async fn read_sensors( + &self, + sql: String, + params: Vec, + ) -> Result> { + struct SensorRow { + sensor_id: i64, + uuid: Uuid, + name: String, + sensor_type: SensorType, + unit: Option, + } + let rows = self + .query_rows("read sensors", sql, params, |row| { + let uuid = required(row.get_string_by_name("uuid"), "uuid")?; + let uuid = + Uuid::from_str(&uuid).map_err(|error| parse_error("a sensor uuid", error))?; + let name = required(row.get_string_by_name("name"), "name")?; + let sensor_type = required(row.get_string_by_name("type"), "type")?; + let sensor_type = SensorType::from_str(&sensor_type).map_err(|error| { + StorageError::invalid_data_format( + &format!("Failed to parse sensor type '{sensor_type}': {error}"), + Some(uuid), + Some(&name), + ) + })?; + Ok(SensorRow { + sensor_id: required(row.get_i64_by_name("sensor_id"), "sensor_id")?, + uuid, + name, + sensor_type, + unit: row + .get_string_by_name("unit_name")? + .map(|name| -> Result { + Ok(Unit::new(name, row.get_string_by_name("unit_description")?)) + }) + .transpose()?, + }) + }) + .await?; + if rows.is_empty() { + return Ok(Vec::new()); + } + + let ids: Vec = rows.iter().map(|row| row.sensor_id).collect(); + let mut labels = self.labels_of_sensors(&ids).await?; + Ok(rows + .into_iter() + .map(|row| { + let labels = labels.remove(&row.sensor_id).unwrap_or_default(); + ( + row.sensor_id, + Sensor::new(row.uuid, row.name, row.sensor_type, row.unit, Some(labels)), + ) + }) + .collect()) + } + + /// The labels of many sensors, one query per chunk of ids. + pub(super) async fn labels_of_sensors( + &self, + ids: &[i64], + ) -> Result> { + let sql = format!( + "SELECT DISTINCT sensor_id, name, description FROM {} \ + WHERE sensor_id IN UNNEST(@ids) ORDER BY sensor_id, name, description", + self.table("labels") + ); + let mut labels: HashMap = HashMap::new(); + for chunk in ids.chunks(LABEL_LOOKUP_CHUNK) { + let rows = self + .query_rows( + "read labels", + sql.clone(), + vec![int_array_param("ids", chunk)], + |row| { + Ok(( + required(row.get_i64(0), "sensor_id")?, + required(row.get_string(1), "label name")?, + required(row.get_string(2), "label description")?, + )) + }, + ) + .await?; + for (sensor_id, name, description) in rows { + labels + .entry(sensor_id) + .or_default() + .push((name, description)); + } + } + Ok(labels) + } + + /// A series by its UUID, with its id in the tables. + pub(super) async fn get_sensor_metadata( + &self, + sensor_uuid: &str, + ) -> Result> { + let uuid = parse_uuid(sensor_uuid)?; + let id = uuid_id(&uuid); + let sql = self.dataset.sensors_sql( + &[ + "s.sensor_id = @id".to_string(), + "s.uuid = @uuid".to_string(), + ], + "LIMIT 1", + ); + let params = vec![int_param("id", id), string_param("uuid", &uuid.to_string())]; + Ok(self.read_sensors(sql, params).await?.into_iter().next()) + } + + /// The samples of one series, oldest first (newest first on request), at most `limit`. + pub(super) async fn query_samples_by_type( + &self, + id: i64, + sensor_type: SensorType, + start_time: Option, + end_time: Option, + limit: usize, + newest_first: bool, + ) -> Result { + let mut params = vec![int_param("id", id)]; + let conditions = window( + start_time + .as_ref() + .map(crate::storage::common::datetime_to_micros), + end_time + .as_ref() + .map(crate::storage::common::datetime_to_micros), + &mut params, + ); + let sql = format!( + "SELECT sensor_id, UNIX_MICROS(timestamp) AS timestamp_us, {} FROM {} \ + WHERE sensor_id = @id{conditions} ORDER BY timestamp {order} LIMIT {limit}", + sample_columns(sensor_type), + self.table(sample_table(sensor_type)), + limit = limit.min(MAX_LIMIT), + order = if newest_first { "DESC" } else { "ASC" }, + ); + let mut samples = empty_samples(sensor_type); + for row in self + .query_rows("read samples", sql, params, |row| { + let mut one = empty_samples(sensor_type); + read_sample(&mut one, row)?; + Ok(one) + }) + .await? + { + append_samples(&mut samples, row); + } + Ok(samples) + } + + /// The samples of many series of one numeric type, with one query, at most `limit` in total. + pub(super) async fn query_numeric_samples_of_many( + &self, + sensor_type: SensorType, + ids: &[i64], + start_us: Option, + end_us: Option, + limit: usize, + ) -> Result> { + let mut params = vec![int_array_param("ids", ids)]; + let conditions = window(start_us, end_us, &mut params); + let sql = format!( + "SELECT sensor_id, UNIX_MICROS(timestamp) AS timestamp_us, {} FROM {} \ + WHERE sensor_id IN UNNEST(@ids){conditions} \ + ORDER BY sensor_id, timestamp LIMIT {limit}", + sample_columns(sensor_type), + self.table(sample_table(sensor_type)), + limit = limit.min(MAX_LIMIT), + ); + let mut by_sensor: HashMap = HashMap::new(); + let rows = self + .query_rows("read samples", sql, params, |row| { + let sensor_id = required(row.get_i64(0), "sensor_id")?; + let mut one = empty_samples(sensor_type); + read_sample(&mut one, row)?; + Ok((sensor_id, one)) + }) + .await?; + for (sensor_id, one) in rows { + append_samples( + by_sensor + .entry(sensor_id) + .or_insert_with(|| empty_samples(sensor_type)), + one, + ); + } + Ok(by_sensor) + } +} + +/// Move the samples of `from` to the end of `to`, of the same type. +pub(super) fn append_samples(to: &mut TypedSamples, from: TypedSamples) { + match (to, from) { + (TypedSamples::Integer(to), TypedSamples::Integer(from)) => to.extend(from), + (TypedSamples::Numeric(to), TypedSamples::Numeric(from)) => to.extend(from), + (TypedSamples::Float(to), TypedSamples::Float(from)) => to.extend(from), + (TypedSamples::String(to), TypedSamples::String(from)) => to.extend(from), + (TypedSamples::Boolean(to), TypedSamples::Boolean(from)) => to.extend(from), + (TypedSamples::Location(to), TypedSamples::Location(from)) => to.extend(from), + (TypedSamples::Json(to), TypedSamples::Json(from)) => to.extend(from), + (TypedSamples::Blob(to), TypedSamples::Blob(from)) => to.extend(from), + _ => unreachable!("the samples of a table have the type of the table"), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn window_bounds_stay_inside_the_timestamp_range() { + assert_eq!(clamp_micros(0), 0); + assert_eq!(clamp_micros(i64::MIN), MIN_MICROS); + assert_eq!(clamp_micros(i64::MAX), MAX_MICROS); + // 0001-01-01 and 9999-12-31 23:59:59.999999 UTC + assert_eq!(MIN_MICROS / 1_000_000, -62_135_596_800); + assert_eq!(MAX_MICROS, 253_402_300_800 * 1_000_000 - 1); + } + + #[test] + fn sensors_are_read_as_a_set_with_their_units() { + let dataset = Dataset { + project_id: "my-project".to_string(), + dataset_id: "sensapp".to_string(), + }; + assert_eq!(dataset.table("labels"), "`my-project.sensapp.labels`"); + let sql = dataset.sensors_sql( + &["s.name = @m0".to_string()], + "ORDER BY s.sensor_id LIMIT 5", + ); + assert!( + sql.contains("FROM `my-project.sensapp.sensors` GROUP BY sensor_id"), + "{sql}" + ); + assert!( + sql.contains("FROM `my-project.sensapp.units` GROUP BY id"), + "{sql}" + ); + assert!( + sql.ends_with("WHERE s.name = @m0 ORDER BY s.sensor_id LIMIT 5"), + "{sql}" + ); + assert!(!dataset.sensors_sql(&[], "").contains("WHERE")); + } + + #[test] + fn the_tables_are_the_shared_list() { + use SensorType::*; + for sensor_type in [ + Integer, Numeric, Float, String, Boolean, Location, Json, Blob, + ] { + assert!(crate::storage::common::VALUE_TABLES.contains(&sample_table(sensor_type))); + } + } + + #[test] + fn windows_use_parameters() { + let mut params = Vec::new(); + let sql = window(Some(10), None, &mut params); + assert_eq!(sql, " AND timestamp >= TIMESTAMP_MICROS(@start)"); + assert_eq!(params.len(), 1); + let sql = window(Some(10), Some(20), &mut Vec::new()); + assert!(sql.contains("@start") && sql.contains("@end")); + assert_eq!(window(None, None, &mut Vec::new()), ""); + } + + #[test] + fn an_invalid_uuid_is_a_bad_request() { + let error = parse_uuid("not-a-uuid").unwrap_err(); + assert!(matches!( + error.downcast_ref::(), + Some(StorageError::InvalidDataFormat { .. }) + )); + } +} diff --git a/src/storage/bigquery/rows.rs b/src/storage/bigquery/rows.rs new file mode 100644 index 00000000..b2892159 --- /dev/null +++ b/src/storage/bigquery/rows.rs @@ -0,0 +1,315 @@ +//! The rows sent to the Storage Write API and the descriptors of their tables. +//! +//! Types follow the table of supported protocol buffer types of the API: a TIMESTAMP is an `int64` +//! of microseconds since the epoch, a NUMERIC a decimal `string`, a JSON a `string`. +//! `migrations/init.sql` is the other side of this file; a test compares their columns. +//! +//! Floats are `double`, which the first version of the backend could not write: with +//! `gcp-bigquery-client` 0.22 a `ColumnType::Float64` descriptor was declared to the service as a protobuf +//! `float` (32 bits) while the row carried a `double`, and BigQuery stored NULL +//! (), hence the `f32` and its comment. Since +//! 0.26 the descriptor has `ColumnType::Double`, which is declared as `double`, and the integration test +//! `every_type_comes_back_as_written` checks that `0.1 + 0.2` comes back exactly. + +use gcp_bigquery_client::storage::{ColumnMode, ColumnType, FieldDescriptor, TableDescriptor}; +use prost::Message; +use std::sync::{Arc, LazyLock}; + +#[derive(Clone, PartialEq, Message)] +pub struct UnitRow { + #[prost(int64, required, tag = "1")] + pub id: i64, + #[prost(string, required, tag = "2")] + pub name: String, + #[prost(string, optional, tag = "3")] + pub description: Option, +} + +#[derive(Clone, PartialEq, Message)] +pub struct SensorRow { + #[prost(int64, required, tag = "1")] + pub sensor_id: i64, + #[prost(string, required, tag = "2")] + pub uuid: String, + #[prost(string, required, tag = "3")] + pub name: String, + #[prost(string, required, tag = "4")] + pub r#type: String, + #[prost(int64, optional, tag = "5")] + pub unit: Option, +} + +#[derive(Clone, PartialEq, Message)] +pub struct LabelRow { + #[prost(int64, required, tag = "1")] + pub sensor_id: i64, + #[prost(string, required, tag = "2")] + pub name: String, + #[prost(string, required, tag = "3")] + pub description: String, +} + +#[derive(Clone, PartialEq, Message)] +pub struct IntegerValueRow { + #[prost(int64, required, tag = "1")] + pub sensor_id: i64, + #[prost(int64, required, tag = "2")] + pub timestamp: i64, + #[prost(int64, required, tag = "3")] + pub value: i64, +} + +#[derive(Clone, PartialEq, Message)] +pub struct NumericValueRow { + #[prost(int64, required, tag = "1")] + pub sensor_id: i64, + #[prost(int64, required, tag = "2")] + pub timestamp: i64, + #[prost(string, required, tag = "3")] + pub value: String, +} + +#[derive(Clone, PartialEq, Message)] +pub struct FloatValueRow { + #[prost(int64, required, tag = "1")] + pub sensor_id: i64, + #[prost(int64, required, tag = "2")] + pub timestamp: i64, + #[prost(double, required, tag = "3")] + pub value: f64, +} + +#[derive(Clone, PartialEq, Message)] +pub struct StringValueRow { + #[prost(int64, required, tag = "1")] + pub sensor_id: i64, + #[prost(int64, required, tag = "2")] + pub timestamp: i64, + #[prost(string, required, tag = "3")] + pub value: String, +} + +#[derive(Clone, PartialEq, Message)] +pub struct BooleanValueRow { + #[prost(int64, required, tag = "1")] + pub sensor_id: i64, + #[prost(int64, required, tag = "2")] + pub timestamp: i64, + #[prost(bool, required, tag = "3")] + pub value: bool, +} + +#[derive(Clone, PartialEq, Message)] +pub struct LocationValueRow { + #[prost(int64, required, tag = "1")] + pub sensor_id: i64, + #[prost(int64, required, tag = "2")] + pub timestamp: i64, + #[prost(double, required, tag = "3")] + pub latitude: f64, + #[prost(double, required, tag = "4")] + pub longitude: f64, +} + +#[derive(Clone, PartialEq, Message)] +pub struct JsonValueRow { + #[prost(int64, required, tag = "1")] + pub sensor_id: i64, + #[prost(int64, required, tag = "2")] + pub timestamp: i64, + #[prost(string, required, tag = "3")] + pub value: String, +} + +#[derive(Clone, PartialEq, Message)] +pub struct BlobValueRow { + #[prost(int64, required, tag = "1")] + pub sensor_id: i64, + #[prost(int64, required, tag = "2")] + pub timestamp: i64, + #[prost(bytes = "vec", required, tag = "3")] + pub value: Vec, +} + +/// Columns in the order of their protocol buffer tags, starting at 1. +fn descriptor(columns: &[(&str, ColumnType, ColumnMode)]) -> Arc { + Arc::new(TableDescriptor { + field_descriptors: columns + .iter() + .zip(1..) + .map(|((name, typ, mode), number)| FieldDescriptor { + name: name.to_string(), + number, + typ: *typ, + mode: *mode, + }) + .collect(), + }) +} + +fn sample_descriptor(value: &[(&str, ColumnType)]) -> Arc { + let mut columns = vec![ + ("sensor_id", ColumnType::Int64, ColumnMode::Required), + ("timestamp", ColumnType::Int64, ColumnMode::Required), + ]; + columns.extend( + value + .iter() + .map(|(name, typ)| (*name, *typ, ColumnMode::Required)), + ); + descriptor(&columns) +} + +pub static UNITS: LazyLock> = LazyLock::new(|| { + descriptor(&[ + ("id", ColumnType::Int64, ColumnMode::Required), + ("name", ColumnType::String, ColumnMode::Required), + ("description", ColumnType::String, ColumnMode::Nullable), + ]) +}); + +pub static SENSORS: LazyLock> = LazyLock::new(|| { + descriptor(&[ + ("sensor_id", ColumnType::Int64, ColumnMode::Required), + ("uuid", ColumnType::String, ColumnMode::Required), + ("name", ColumnType::String, ColumnMode::Required), + ("type", ColumnType::String, ColumnMode::Required), + ("unit", ColumnType::Int64, ColumnMode::Nullable), + ]) +}); + +pub static LABELS: LazyLock> = LazyLock::new(|| { + descriptor(&[ + ("sensor_id", ColumnType::Int64, ColumnMode::Required), + ("name", ColumnType::String, ColumnMode::Required), + ("description", ColumnType::String, ColumnMode::Required), + ]) +}); + +pub static INTEGER_VALUES: LazyLock> = + LazyLock::new(|| sample_descriptor(&[("value", ColumnType::Int64)])); +pub static NUMERIC_VALUES: LazyLock> = + LazyLock::new(|| sample_descriptor(&[("value", ColumnType::String)])); +pub static FLOAT_VALUES: LazyLock> = + LazyLock::new(|| sample_descriptor(&[("value", ColumnType::Double)])); +pub static STRING_VALUES: LazyLock> = + LazyLock::new(|| sample_descriptor(&[("value", ColumnType::String)])); +pub static BOOLEAN_VALUES: LazyLock> = + LazyLock::new(|| sample_descriptor(&[("value", ColumnType::Bool)])); +pub static LOCATION_VALUES: LazyLock> = LazyLock::new(|| { + sample_descriptor(&[ + ("latitude", ColumnType::Double), + ("longitude", ColumnType::Double), + ]) +}); +pub static JSON_VALUES: LazyLock> = + LazyLock::new(|| sample_descriptor(&[("value", ColumnType::String)])); +pub static BLOB_VALUES: LazyLock> = + LazyLock::new(|| sample_descriptor(&[("value", ColumnType::Bytes)])); + +#[cfg(test)] +mod tests { + use super::*; + use std::collections::BTreeMap; + + /// `table -> [(column, SQL type)]`, read from the migration. + fn schema() -> BTreeMap> { + let sql = include_str!("migrations/init.sql"); + let mut tables = BTreeMap::new(); + for statement in sql.split(';') { + let Some(start) = statement.find("CREATE TABLE IF NOT EXISTS `{dataset}.") else { + continue; + }; + let after = &statement[start + "CREATE TABLE IF NOT EXISTS `{dataset}.".len()..]; + let (name, rest) = after.split_once('`').unwrap(); + let columns = &rest[rest.find('(').unwrap() + 1..rest.find("\n)").unwrap()]; + let columns = columns + .lines() + .map(|line| line.trim().trim_end_matches(',')) + .filter(|line| !line.is_empty()) + .map(|line| { + let mut words = line.split_whitespace(); + ( + words.next().unwrap().to_string(), + words.next().unwrap().to_string(), + ) + }) + .collect(); + tables.insert(name.to_string(), columns); + } + tables + } + + fn proto_type(sql_type: &str) -> ColumnType { + match sql_type { + "INT64" | "TIMESTAMP" => ColumnType::Int64, + "FLOAT64" => ColumnType::Double, + "BOOL" => ColumnType::Bool, + "BYTES" => ColumnType::Bytes, + // STRING, JSON and NUMERIC (a decimal string) + _ => ColumnType::String, + } + } + + #[test] + fn descriptors_match_the_tables_of_the_migration() { + let descriptors: [(&str, &Arc); 11] = [ + ("units", &UNITS), + ("sensors", &SENSORS), + ("labels", &LABELS), + ("integer_values", &INTEGER_VALUES), + ("numeric_values", &NUMERIC_VALUES), + ("float_values", &FLOAT_VALUES), + ("string_values", &STRING_VALUES), + ("boolean_values", &BOOLEAN_VALUES), + ("location_values", &LOCATION_VALUES), + ("json_values", &JSON_VALUES), + ("blob_values", &BLOB_VALUES), + ]; + let schema = schema(); + assert_eq!(schema.len(), descriptors.len(), "{:?}", schema.keys()); + for (table, descriptor) in descriptors { + let columns = &schema[table]; + assert_eq!( + columns.len(), + descriptor.field_descriptors.len(), + "columns of {table}" + ); + for (field, (name, sql_type)) in descriptor.field_descriptors.iter().zip(columns) { + assert_eq!(&field.name, name, "column of {table}"); + assert_eq!( + format!("{:?}", field.typ), + format!("{:?}", proto_type(sql_type)), + "type of {table}.{name}" + ); + } + } + } + + #[test] + fn the_tables_of_the_value_types_are_the_shared_list() { + let schema = schema(); + for table in crate::storage::common::VALUE_TABLES { + assert!(schema.contains_key(table), "{table}"); + } + } + + #[test] + fn tags_follow_the_column_order() { + for descriptor in [&*UNITS, &*SENSORS, &*LOCATION_VALUES, &*BLOB_VALUES] { + for (index, field) in descriptor.field_descriptors.iter().enumerate() { + assert_eq!(field.number as usize, index + 1); + } + } + // The rows encode the fields at those tags + let row = LocationValueRow { + sensor_id: 1, + timestamp: 2, + latitude: 3.0, + longitude: 4.0, + }; + let decoded = LocationValueRow::decode(row.encode_to_vec().as_slice()).unwrap(); + assert_eq!(decoded, row); + assert_eq!(row.encode_to_vec()[0], 0x08, "field 1, varint"); + } +} diff --git a/src/storage/bigquery/selector.rs b/src/storage/bigquery/selector.rs new file mode 100644 index 00000000..f317c531 --- /dev/null +++ b/src/storage/bigquery/selector.rs @@ -0,0 +1,69 @@ +//! Reading the series of a selector with a few queries: one for the sensors and their labels, then +//! one per numeric type for all their samples. +//! +//! Aggregated selectors too: one aggregating statement per numeric type. + +use super::BigQueryStorage; +use crate::datamodel::{SensAppDateTime, Sensor, SensorType, TypedSamples}; +use crate::storage::LabelMatcher; +use crate::storage::selector::{AggregatedRead, BulkSelectorBackend}; +use anyhow::Result; +use async_trait::async_trait; +use std::collections::HashMap; + +#[async_trait] +impl BulkSelectorBackend for BigQueryStorage { + type SensorKey = i64; + + async fn find_selector_sensors( + &self, + matchers: &[LabelMatcher], + numeric_only: bool, + limit: Option, + ) -> Result> { + self.find_sensors_by_matchers(matchers, numeric_only, limit) + .await + } + + async fn read_numeric_samples( + &self, + sensor_type: SensorType, + sensors: &[i64], + start_us: Option, + end_us: Option, + limit: usize, + ) -> Result> { + self.query_numeric_samples_of_many(sensor_type, sensors, start_us, end_us, limit) + .await + } + + async fn read_aggregated_samples( + &self, + sensor_type: SensorType, + sensors: &[i64], + read: &AggregatedRead, + limit: usize, + ) -> Result> { + self.query_aggregated_of_many(sensor_type, sensors, read, limit) + .await + } + + async fn read_other_samples( + &self, + sensor_key: i64, + sensor: &Sensor, + start_time: Option, + end_time: Option, + limit: usize, + ) -> Result { + self.query_samples_by_type( + sensor_key, + sensor.sensor_type, + start_time, + end_time, + limit, + false, + ) + .await + } +} diff --git a/src/storage/clickhouse/clickhouse_utilities.rs b/src/storage/clickhouse/clickhouse_utilities.rs index 26a8a8ba..fb8185c5 100644 --- a/src/storage/clickhouse/clickhouse_utilities.rs +++ b/src/storage/clickhouse/clickhouse_utilities.rs @@ -10,25 +10,7 @@ use uuid::Uuid; pub const CLICKHOUSE_NUMERIC_SCALE: u32 = 8; -/// Convert a UUID to the `UInt64` key used by every ClickHouse table. -/// -/// The ids are stored, so this function is part of the on-disk format and must never change: -/// the XOR of the two big-endian 64-bit halves of the UUID. Sensor UUIDs are hashes, so the -/// result is uniformly distributed. Do not use `std`'s `DefaultHasher` here, its algorithm is -/// explicitly allowed to change between Rust releases, which would orphan every stored sample. -pub fn uuid_to_sensor_id(uuid: &Uuid) -> u64 { - let value = uuid.as_u128(); - ((value >> 64) as u64) ^ (value as u64) -} - -/// Id of a unit in the `units` table: the first 8 bytes of the BLAKE3 hash of its name, -/// read as a little-endian integer. Part of the on-disk format, like [`uuid_to_sensor_id`]. -pub fn unit_name_to_id(name: &str) -> u64 { - let hash = blake3::hash(name.as_bytes()); - let mut bytes = [0u8; 8]; - bytes.copy_from_slice(&hash.as_bytes()[..8]); - u64::from_le_bytes(bytes) -} +pub use crate::storage::common::{unit_name_to_id, uuid_to_sensor_id}; /// Convert SensAppDateTime to microseconds timestamp - using common implementation pub use crate::storage::common::datetime_to_micros; @@ -372,37 +354,4 @@ mod tests { let error = map_clickhouse_error(bad_response(62), None, Some("temperature")); assert!(format!("{error}").contains("temperature"), "{error}"); } - - // These values are stored in ClickHouse. If one of these tests fails, the on-disk format - // changed: existing data would no longer be found. - #[test] - fn sensor_ids_are_pinned() { - let id = |uuid: &str| uuid_to_sensor_id(&Uuid::parse_str(uuid).unwrap()); - assert_eq!( - id("00000000-0000-4000-8000-000000000001"), - 0x8000_0000_0000_4001 - ); - assert_eq!(id("00000000-0000-0000-0000-000000000000"), 0); - assert_eq!(id("ffffffff-ffff-ffff-ffff-ffffffffffff"), 0); - assert_eq!( - id("0123456789abcdeffedcba9876543210"), - 0xffff_ffff_ffff_ffff - ); - assert_eq!( - id("9d87123d-9b47-466d-9eda-001c2ecf9c54"), - 0x9d87_123d_9b47_466d ^ 0x9eda_001c_2ecf_9c54 - ); - } - - #[test] - fn unit_ids_are_pinned() { - assert_eq!(unit_name_to_id("°C"), UNIT_CELSIUS_ID); - assert_eq!(unit_name_to_id(""), UNIT_EMPTY_ID); - assert_ne!(unit_name_to_id("m"), unit_name_to_id("s")); - } - - const UNIT_CELSIUS_ID: u64 = 14_058_937_354_730_164_320; - // First 8 bytes of the published BLAKE3 test vector for empty input - // (af1349b9f5f9a1a6...), read as little-endian. - const UNIT_EMPTY_ID: u64 = 0xa6a1_f9f5_b949_13af; } diff --git a/src/storage/common.rs b/src/storage/common.rs index 18612e6b..e91532f8 100644 --- a/src/storage/common.rs +++ b/src/storage/common.rs @@ -9,6 +9,7 @@ use rust_decimal::{Decimal, prelude::ToPrimitive}; use simplify_polyline::{Point, simplify}; use smallvec::smallvec; use std::collections::BTreeSet; +use uuid::Uuid; /// Whether an error comes from a foreign-key violation reported by the database. /// @@ -56,6 +57,27 @@ pub fn duplicate_key_columns(table: &str, time_column: &str) -> String { } } +/// Convert a UUID to the 64-bit key of the series in the ClickHouse and BigQuery tables (BigQuery +/// stores the same bits as a signed `INT64`). +/// +/// The ids are stored, so this function is part of the on-disk format and must never change: +/// the XOR of the two big-endian 64-bit halves of the UUID. Sensor UUIDs are hashes, so the +/// result is uniformly distributed. Do not use `std`'s `DefaultHasher` here, its algorithm is +/// explicitly allowed to change between Rust releases, which would orphan every stored sample. +pub fn uuid_to_sensor_id(uuid: &Uuid) -> u64 { + let value = uuid.as_u128(); + ((value >> 64) as u64) ^ (value as u64) +} + +/// Id of a unit in the `units` table: the first 8 bytes of the BLAKE3 hash of its name, +/// read as a little-endian integer. Part of the on-disk format, like [`uuid_to_sensor_id`]. +pub fn unit_name_to_id(name: &str) -> u64 { + let hash = blake3::hash(name.as_bytes()); + let mut bytes = [0u8; 8]; + bytes.copy_from_slice(&hash.as_bytes()[..8]); + u64::from_le_bytes(bytes) +} + /// Convert SensAppDateTime to Unix microseconds for database storage pub fn datetime_to_micros(datetime: &SensAppDateTime) -> i64 { // Use to_unix with Microsecond unit to get a f64 in microseconds, @@ -578,6 +600,39 @@ fn aggregate_numeric_samples( #[cfg(test)] mod tests { + // These values are stored in ClickHouse and BigQuery. If one of these tests fails, the on-disk format + // changed: existing data would no longer be found. + #[test] + fn sensor_ids_are_pinned() { + let id = |uuid: &str| uuid_to_sensor_id(&Uuid::parse_str(uuid).unwrap()); + assert_eq!( + id("00000000-0000-4000-8000-000000000001"), + 0x8000_0000_0000_4001 + ); + assert_eq!(id("00000000-0000-0000-0000-000000000000"), 0); + assert_eq!(id("ffffffff-ffff-ffff-ffff-ffffffffffff"), 0); + assert_eq!( + id("0123456789abcdeffedcba9876543210"), + 0xffff_ffff_ffff_ffff + ); + assert_eq!( + id("9d87123d-9b47-466d-9eda-001c2ecf9c54"), + 0x9d87_123d_9b47_466d ^ 0x9eda_001c_2ecf_9c54 + ); + } + + #[test] + fn unit_ids_are_pinned() { + assert_eq!(unit_name_to_id("°C"), UNIT_CELSIUS_ID); + assert_eq!(unit_name_to_id(""), UNIT_EMPTY_ID); + assert_ne!(unit_name_to_id("m"), unit_name_to_id("s")); + } + + const UNIT_CELSIUS_ID: u64 = 14_058_937_354_730_164_320; + // First 8 bytes of the published BLAKE3 test vector for empty input + // (af1349b9f5f9a1a6...), read as little-endian. + const UNIT_EMPTY_ID: u64 = 0xa6a1_f9f5_b949_13af; + use super::*; use crate::datamodel::sensapp_datetime::SensAppDateTimeExt; diff --git a/tests/integration/advanced_backend_queries.rs b/tests/integration/advanced_backend_queries.rs index 59e1aba9..caad25ac 100644 --- a/tests/integration/advanced_backend_queries.rs +++ b/tests/integration/advanced_backend_queries.rs @@ -285,3 +285,17 @@ async fn test_clickhouse_native_bucketed_average_query() -> Result<()> { async fn test_clickhouse_native_latest_and_availability_queries() -> Result<()> { assert_latest_and_availability_for_backend(DatabaseType::ClickHouse).await } + +#[cfg(feature = "bigquery")] +#[tokio::test] +#[serial] +async fn test_bigquery_bucketed_average_query() -> Result<()> { + assert_bucketed_average_for_backend(DatabaseType::BigQuery).await +} + +#[cfg(feature = "bigquery")] +#[tokio::test] +#[serial] +async fn test_bigquery_latest_and_availability_queries() -> Result<()> { + assert_latest_and_availability_for_backend(DatabaseType::BigQuery).await +} diff --git a/tests/integration/bigquery_integration.rs b/tests/integration/bigquery_integration.rs new file mode 100644 index 00000000..b079cf5f --- /dev/null +++ b/tests/integration/bigquery_integration.rs @@ -0,0 +1,809 @@ +//! What is specific to BigQuery. The backend-generic suites run on it too +//! (`TEST_DATABASE_URL=bigquery://... cargo test --no-default-features --features bigquery +//! --test integration`), see `docs/BIGQUERY.md` for the setup and the cost. +//! +//! These tests need a real project: without a `bigquery:` `TEST_DATABASE_URL` they say so and pass. + +#[cfg(feature = "bigquery")] +mod bigquery_tests { + use crate::common::{DatabaseType, TestDb}; + use anyhow::Result; + use rust_decimal::Decimal; + use sensapp::config::load_configuration_for_tests; + use sensapp::datamodel::batch_builder::BatchBuilder; + use sensapp::datamodel::sensapp_vec::SensAppLabels; + use sensapp::datamodel::unit::Unit; + use sensapp::datamodel::{Sample, Sensor, SensorType, TypedSamples}; + use sensapp::storage::query::LabelMatcher; + use sensapp::storage::storage_factory::create_storage_from_connection_string; + use sensapp::storage::{StorageError, StorageInstance}; + use serial_test::serial; + use smallvec::smallvec; + use std::str::FromStr; + use std::sync::Arc; + use uuid::Uuid; + + static INIT: std::sync::Once = std::sync::Once::new(); + + /// The test database, or `None` (and a message) when no BigQuery dataset is configured. + async fn open() -> Result> { + INIT.call_once(|| { + load_configuration_for_tests().expect("Failed to load configuration for tests"); + }); + if !std::env::var("TEST_DATABASE_URL").is_ok_and(|url| url.starts_with("bigquery:")) { + eprintln!("skipping: TEST_DATABASE_URL is not a bigquery:// connection string"); + return Ok(None); + } + Ok(Some(TestDb::new_with_type(DatabaseType::BigQuery).await?)) + } + + fn at(seconds: f64) -> hifitime::Epoch { + hifitime::Epoch::from_unix_seconds(seconds) + } + + fn sensor(name: &str, sensor_type: SensorType, labels: &[(&str, &str)]) -> Arc { + let labels: SensAppLabels = labels + .iter() + .map(|(name, value)| (name.to_string(), value.to_string())) + .collect(); + Arc::new(Sensor::new( + Uuid::new_v4(), + name.to_string(), + sensor_type, + None, + Some(labels), + )) + } + + async fn publish( + storage: &Arc, + sensors: Vec<(Arc, TypedSamples)>, + ) -> Result<()> { + let mut builder = BatchBuilder::new()?; + for (sensor, samples) in sensors { + builder.add(sensor, samples).await?; + } + builder.send_what_is_left(storage.clone()).await?; + Ok(()) + } + + #[tokio::test] + #[serial] + async fn migration_can_run_again_and_the_health_check_passes() -> Result<()> { + let Some(db) = open().await? else { + return Ok(()); + }; + let storage = db.storage(); + storage.create_or_migrate().await?; + storage.create_or_migrate().await?; + storage.health_check().await?; + Ok(()) + } + + /// Floats and locations went through `f32`, JSON values became `""`, NUMERIC lost its digits + /// beyond the ninth: every type comes back as it was written, microseconds included. + #[tokio::test] + #[serial] + async fn every_type_comes_back_as_written() -> Result<()> { + let Some(db) = open().await? else { + return Ok(()); + }; + let storage = db.storage(); + let when = at(1_704_067_200.123_456); + + let float = sensor("bq_float", SensorType::Float, &[("site", "oslo")]); + let integer = sensor("bq_integer", SensorType::Integer, &[]); + let numeric = sensor("bq_numeric", SensorType::Numeric, &[]); + let string = sensor("bq_string", SensorType::String, &[]); + let boolean = sensor("bq_boolean", SensorType::Boolean, &[]); + let location = sensor("bq_location", SensorType::Location, &[]); + let json = sensor("bq_json", SensorType::Json, &[]); + let blob = sensor("bq_blob", SensorType::Blob, &[]); + let json_value = serde_json::json!({"a": [1, 2.5, "x"], "b": {"c": null}, "d": true}); + let blob_value: Vec = (0..=255).collect(); + + publish( + &storage, + vec![ + ( + float.clone(), + TypedSamples::Float(smallvec![ + Sample { + datetime: when, + value: 0.1 + 0.2 + }, + Sample { + datetime: at(1_704_067_201.0), + value: f64::MAX + } + ]), + ), + ( + integer.clone(), + TypedSamples::Integer(smallvec![Sample { + datetime: when, + value: i64::MIN + }]), + ), + ( + numeric.clone(), + TypedSamples::Numeric(smallvec![Sample { + datetime: when, + value: Decimal::from_str("-12345678901234567890.123456789")? + }]), + ), + ( + string.clone(), + TypedSamples::String(smallvec![Sample { + datetime: when, + value: "héllo \"wörld\" 🌍\n".to_string() + }]), + ), + ( + boolean.clone(), + TypedSamples::Boolean(smallvec![Sample { + datetime: when, + value: false + }]), + ), + ( + location.clone(), + TypedSamples::Location(smallvec![Sample { + datetime: when, + value: geo::Point::new(10.123_456_789_012, 59.987_654_321_098) + }]), + ), + ( + json.clone(), + TypedSamples::Json(smallvec![Sample { + datetime: when, + value: json_value.clone() + }]), + ), + ( + blob.clone(), + TypedSamples::Blob(smallvec![Sample { + datetime: when, + value: blob_value.clone() + }]), + ), + ], + ) + .await?; + + let read = |sensor: &Arc| { + let storage = storage.clone(); + let uuid = sensor.uuid.to_string(); + async move { + Ok::<_, anyhow::Error>( + storage + .query_sensor_data(&uuid, None, None, None) + .await? + .expect("the sensor is stored"), + ) + } + }; + + let data = read(&float).await?; + assert_eq!( + data.sensor.labels.as_slice(), + &[("site".to_string(), "oslo".to_string())] + ); + let TypedSamples::Float(samples) = &data.samples else { + panic!("floats") + }; + assert_eq!(samples.len(), 2); + assert_eq!(samples[0].value, 0.1 + 0.2); + assert_eq!(samples[0].datetime, when); + assert_eq!(samples[1].value, f64::MAX); + + let TypedSamples::Integer(samples) = &read(&integer).await?.samples else { + panic!("integers") + }; + assert_eq!(samples[0].value, i64::MIN); + + let TypedSamples::Numeric(samples) = &read(&numeric).await?.samples else { + panic!("numerics") + }; + assert_eq!( + samples[0].value, + Decimal::from_str("-12345678901234567890.123456789")? + ); + + let TypedSamples::String(samples) = &read(&string).await?.samples else { + panic!("strings") + }; + assert_eq!(samples[0].value, "héllo \"wörld\" 🌍\n"); + + let TypedSamples::Boolean(samples) = &read(&boolean).await?.samples else { + panic!("booleans") + }; + assert!(!samples[0].value); + + let TypedSamples::Location(samples) = &read(&location).await?.samples else { + panic!("locations") + }; + assert_eq!(samples[0].value.x(), 10.123_456_789_012); + assert_eq!(samples[0].value.y(), 59.987_654_321_098); + + let TypedSamples::Json(samples) = &read(&json).await?.samples else { + panic!("json") + }; + assert_eq!(samples[0].value, json_value); + + let TypedSamples::Blob(samples) = &read(&blob).await?.samples else { + panic!("blobs") + }; + assert_eq!(samples[0].value, blob_value); + Ok(()) + } + + #[tokio::test] + #[serial] + async fn a_unit_and_a_time_window_are_kept() -> Result<()> { + let Some(db) = open().await? else { + return Ok(()); + }; + let storage = db.storage(); + let temperature = Arc::new(Sensor::new( + Uuid::new_v4(), + "bq_temperature".to_string(), + SensorType::Float, + Some(Unit::new( + "°C".to_string(), + Some("degrees Celsius".to_string()), + )), + None, + )); + publish( + &storage, + vec![( + temperature.clone(), + TypedSamples::Float( + (0..10) + .map(|i| Sample { + datetime: at(1_704_067_200.0 + f64::from(i) * 60.0), + value: f64::from(i), + }) + .collect(), + ), + )], + ) + .await?; + + let uuid = temperature.uuid.to_string(); + let window = storage + .query_sensor_data( + &uuid, + Some(at(1_704_067_200.0 + 120.0)), + Some(at(1_704_067_200.0 + 300.0)), + None, + ) + .await? + .unwrap(); + let unit = window.sensor.unit.clone().unwrap(); + assert_eq!(unit.name, "°C"); + assert_eq!(unit.description.as_deref(), Some("degrees Celsius")); + let TypedSamples::Float(samples) = &window.samples else { + panic!("floats") + }; + let values: Vec = samples.iter().map(|s| s.value).collect(); + assert_eq!( + values, + vec![2.0, 3.0, 4.0, 5.0], + "both bounds are inclusive" + ); + + let limited = storage + .query_sensor_data(&uuid, None, None, Some(3)) + .await? + .unwrap(); + assert_eq!(limited.samples.len(), 3); + + let latest = storage + .query_sensor_data_latest(&uuid, None, None) + .await? + .unwrap(); + let TypedSamples::Float(samples) = &latest.samples else { + panic!("floats") + }; + assert_eq!(samples.len(), 1); + assert_eq!(samples[0].value, 9.0); + Ok(()) + } + + /// Two instances (their own caches) and many concurrent writers register the same series: one + /// series is listed, with its labels once, and every sample is there. + #[tokio::test] + #[serial] + async fn concurrent_instances_registering_one_series_leave_one_series() -> Result<()> { + let Some(db) = open().await? else { + return Ok(()); + }; + let url = std::env::var("TEST_DATABASE_URL")?; + let other: Arc = create_storage_from_connection_string(&url).await?; + let storage = db.storage(); + let shared = sensor( + "bq_shared", + SensorType::Integer, + &[("a", "1"), ("b", "2"), ("c", "3")], + ); + + let writers = (0..8).map(|writer| { + let storage = if writer % 2 == 0 { + storage.clone() + } else { + other.clone() + }; + let shared = shared.clone(); + async move { + publish( + &storage, + vec![( + shared, + TypedSamples::Integer(smallvec![Sample { + datetime: at(1_704_067_200.0 + f64::from(writer)), + value: i64::from(writer), + }]), + )], + ) + .await + } + }); + for result in futures::future::join_all(writers).await { + result?; + } + + let listed = storage.list_series(Some("bq_shared"), None, None).await?; + assert_eq!(listed.series.len(), 1, "{:?}", listed.series); + assert_eq!(listed.series[0].labels.len(), 3, "labels once each"); + let data = storage + .query_sensor_data(&shared.uuid.to_string(), None, None, None) + .await? + .unwrap(); + assert_eq!(data.samples.len(), 8); + assert_eq!(data.sensor.labels.len(), 3); + Ok(()) + } + + #[tokio::test] + #[serial] + async fn the_listing_pages_through_the_series_and_filters_by_metric() -> Result<()> { + let Some(db) = open().await? else { + return Ok(()); + }; + let storage = db.storage(); + let sensors: Vec<_> = (0..5) + .map(|i| { + ( + sensor( + "bq_paged", + SensorType::Integer, + &[("index", &i.to_string())], + ), + TypedSamples::one_integer(i, at(1_704_067_200.0)), + ) + }) + .chain([( + sensor("bq_other", SensorType::Integer, &[]), + TypedSamples::one_integer(0, at(1_704_067_200.0)), + )]) + .collect(); + let expected: std::collections::BTreeSet<_> = sensors + .iter() + .filter(|(sensor, _)| sensor.name == "bq_paged") + .map(|(sensor, _)| sensor.uuid) + .collect(); + publish(&storage, sensors).await?; + + let mut seen = Vec::new(); + let mut bookmark: Option = None; + let mut pages = Vec::new(); + loop { + let page = storage + .list_series(Some("bq_paged"), Some(2), bookmark.as_deref()) + .await?; + pages.push(page.series.len()); + seen.extend(page.series.iter().map(|sensor| sensor.uuid)); + bookmark = page.bookmark; + if bookmark.is_none() { + break; + } + } + assert_eq!(pages, vec![2, 2, 1]); + assert_eq!(seen.len(), 5); + assert_eq!( + seen.iter() + .copied() + .collect::>(), + expected + ); + + let metrics = storage.list_metrics().await?; + let paged = metrics.iter().find(|m| m.name == "bq_paged").unwrap(); + assert_eq!(paged.series_count, 5); + assert!( + storage + .list_series(None, Some(1), Some("not a number")) + .await + .is_err() + ); + Ok(()) + } + + #[tokio::test] + #[serial] + async fn label_selectors_use_prometheus_semantics() -> Result<()> { + let Some(db) = open().await? else { + return Ok(()); + }; + let storage = db.storage(); + let oslo = sensor( + "bq_sel", + SensorType::Float, + &[("site", "oslo"), ("env", "prod")], + ); + let bergen = sensor("bq_sel", SensorType::Float, &[("site", "bergen")]); + let text = sensor( + "bq_sel", + SensorType::String, + &[("site", "oslo"), ("kind", "text")], + ); + publish( + &storage, + vec![ + ( + oslo.clone(), + TypedSamples::Float(smallvec![Sample { + datetime: at(1_704_067_200.0), + value: 1.0 + }]), + ), + ( + bergen.clone(), + TypedSamples::Float(smallvec![Sample { + datetime: at(1_704_067_200.0), + value: 2.0 + }]), + ), + ( + text.clone(), + TypedSamples::String(smallvec![Sample { + datetime: at(1_704_067_200.0), + value: "x".to_string() + }]), + ), + ], + ) + .await?; + + let find = |matchers: Vec, numeric_only: bool| { + let storage = storage.clone(); + async move { + let mut uuids: Vec = storage + .query_sensors_by_labels(&matchers, None, None, None, numeric_only) + .await? + .into_iter() + .map(|data| data.sensor.uuid) + .collect(); + uuids.sort(); + Ok::<_, anyhow::Error>(uuids) + } + }; + let sorted = |mut uuids: Vec| { + uuids.sort(); + uuids + }; + let name = || LabelMatcher::eq("__name__", "bq_sel"); + + assert_eq!( + find(vec![name(), LabelMatcher::eq("site", "oslo")], false).await?, + sorted(vec![oslo.uuid, text.uuid]) + ); + assert_eq!( + find(vec![name(), LabelMatcher::eq("site", "oslo")], true).await?, + vec![oslo.uuid], + "numeric only leaves the string series out" + ); + assert_eq!( + find(vec![name(), LabelMatcher::neq("env", "prod")], true).await?, + vec![bergen.uuid], + "a series without the label is not equal to it" + ); + assert_eq!( + find(vec![name(), LabelMatcher::regex("site", "o.*")], true).await?, + vec![oslo.uuid] + ); + assert_eq!( + find(vec![name(), LabelMatcher::regex("site", "o")], true).await?, + Vec::::new(), + "regexes are anchored" + ); + assert_eq!( + find( + vec![ + LabelMatcher::regex("__name__", "bq_s.l"), + LabelMatcher::not_regex("site", "oslo") + ], + true + ) + .await?, + vec![bergen.uuid] + ); + assert!(find(vec![], false).await?.is_empty()); + // A quote in a value is data, not SQL + assert!( + find(vec![LabelMatcher::eq("site", "o' OR '1'='1")], false) + .await? + .is_empty() + ); + assert!(matches!( + storage + .query_sensor_data("not a uuid", None, None, None) + .await + .unwrap_err() + .downcast_ref::(), + Some(StorageError::InvalidDataFormat { .. }) + )); + Ok(()) + } + + /// Aggregation is done by BigQuery (a `GROUP BY` of buckets), not on the samples read back: every + /// numeric type and aggregation gives what the local reference computes from the raw samples. + /// The values are multiples of a quarter, so that no order of summation changes a result. + #[tokio::test] + #[serial] + async fn aggregation_in_bigquery_equals_the_local_reference() -> Result<()> { + use sensapp::storage::common::apply_query_options; + use sensapp::storage::{Aggregation, SensorDataQueryOptions}; + + let Some(db) = open().await? else { + return Ok(()); + }; + let storage = db.storage(); + let start = 1_704_067_200.0 + 1_234.0; // not aligned on anything + // Three days, a sample every 37 minutes + let times: Vec = (0..117).map(|i| start + f64::from(i) * 2_220.0).collect(); + let integer = sensor("bq_agg_integer", SensorType::Integer, &[]); + let float = sensor("bq_agg_float", SensorType::Float, &[]); + let numeric = sensor("bq_agg_numeric", SensorType::Numeric, &[]); + publish( + &storage, + vec![ + ( + integer.clone(), + TypedSamples::Integer( + times + .iter() + .enumerate() + .map(|(i, t)| Sample { + datetime: at(*t), + value: (i as i64 * 7) % 23 - 11, + }) + .collect(), + ), + ), + ( + float.clone(), + TypedSamples::Float( + times + .iter() + .enumerate() + .map(|(i, t)| Sample { + datetime: at(*t), + value: ((i * 5) % 17) as f64 * 0.25 - 2.0, + }) + .collect(), + ), + ), + ( + numeric.clone(), + TypedSamples::Numeric( + times + .iter() + .enumerate() + .map(|(i, t)| Sample { + datetime: at(*t), + value: Decimal::new(((i * 3) % 19) as i64 * 25, 2), + }) + .collect(), + ), + ), + ], + ) + .await?; + + // A window that starts and ends inside buckets of 6 hours + let window_start = Some(at(start + 3_000.0)); + let window_end = Some(at(start + 2.5 * 86_400.0)); + for sensor in [&integer, &float, &numeric] { + let uuid = sensor.uuid.to_string(); + for aggregation in [ + Aggregation::Avg, + Aggregation::Min, + Aggregation::Max, + Aggregation::Sum, + Aggregation::Count, + Aggregation::First, + Aggregation::Last, + ] { + let options = SensorDataQueryOptions { + start_time: window_start, + end_time: window_end, + limit: None, + step_ms: Some(6 * 3_600_000), + aggregation: Some(aggregation), + simplify: None, + }; + let raw = storage + .query_sensor_data(&uuid, window_start, window_end, None) + .await? + .unwrap(); + let expected = apply_query_options(raw, &options)?; + let got = storage + .query_sensor_data_advanced(&uuid, &options) + .await? + .unwrap(); + + let what = format!("{} {aggregation:?}", sensor.name); + assert_eq!( + got.sensor.sensor_type, expected.sensor.sensor_type, + "{what}" + ); + assert_eq!(got.sensor.unit, expected.sensor.unit, "{what}"); + // Averages: BigQuery's AVG is not the exact sum over the count in the last bits + // (-5.5e-17 for integers that average to 0), and it rounds NUMERIC to 9 digits + match (&got.samples, &expected.samples) { + (TypedSamples::Float(got), TypedSamples::Float(expected)) + if aggregation == Aggregation::Avg => + { + assert_eq!(got.len(), expected.len(), "{what}"); + for (got, expected) in got.iter().zip(expected) { + assert_eq!(got.datetime, expected.datetime, "{what}"); + assert!( + (got.value - expected.value).abs() < 1e-9, + "{what}: {} against {}", + got.value, + expected.value + ); + } + } + (TypedSamples::Numeric(got), TypedSamples::Numeric(expected)) + if aggregation == Aggregation::Avg => + { + assert_eq!(got.len(), expected.len(), "{what}"); + for (got, expected) in got.iter().zip(expected) { + assert_eq!(got.datetime, expected.datetime, "{what}"); + assert!( + (got.value - expected.value).abs() < Decimal::new(1, 8), + "{what}" + ); + } + } + (got, expected) => assert_eq!(got, expected, "{what}"), + } + assert!(got.samples.len() > 8, "{what}: several buckets"); + } + + // The limit counts buckets + let limited = storage + .query_sensor_data_advanced( + &uuid, + &SensorDataQueryOptions { + start_time: window_start, + end_time: window_end, + limit: Some(3), + step_ms: Some(6 * 3_600_000), + aggregation: Some(Aggregation::Max), + simplify: None, + }, + ) + .await? + .unwrap(); + assert_eq!(limited.samples.len(), 3); + } + + // Only numbers are aggregated + let text = sensor("bq_agg_text", SensorType::String, &[]); + publish( + &storage, + vec![( + text.clone(), + TypedSamples::String(smallvec![Sample { + datetime: at(start), + value: "x".to_string() + }]), + )], + ) + .await?; + assert!( + storage + .query_sensor_data_advanced( + &text.uuid.to_string(), + &SensorDataQueryOptions { + start_time: None, + end_time: None, + limit: None, + step_ms: Some(1000), + aggregation: Some(Aggregation::Count), + simplify: None, + }, + ) + .await + .is_err() + ); + Ok(()) + } + + /// The first answer of a query is at most 10 MB: a result over that needs the next pages. + #[tokio::test] + #[serial] + async fn a_result_over_one_page_is_read_whole() -> Result<()> { + let Some(db) = open().await? else { + return Ok(()); + }; + let storage = db.storage(); + const COUNT: usize = 150_000; + let big = sensor("bq_big", SensorType::Float, &[]); + publish( + &storage, + vec![( + big.clone(), + TypedSamples::Float( + (0..COUNT) + .map(|i| Sample { + datetime: at(1_704_067_200.0 + i as f64), + value: i as f64 + 0.5, + }) + .collect(), + ), + )], + ) + .await?; + + let data = storage + .query_sensor_data(&big.uuid.to_string(), None, None, None) + .await? + .unwrap(); + let TypedSamples::Float(samples) = &data.samples else { + panic!("floats") + }; + assert_eq!(samples.len(), COUNT); + assert!( + samples + .windows(2) + .all(|pair| pair[0].datetime < pair[1].datetime) + ); + assert_eq!(samples[COUNT - 1].value, COUNT as f64 - 0.5); + Ok(()) + } + + #[tokio::test] + #[serial] + async fn the_cost_cap_of_the_connection_string_stops_expensive_queries() -> Result<()> { + let Some(db) = open().await? else { + return Ok(()); + }; + db.storage().create_or_migrate().await?; + let url = std::env::var("TEST_DATABASE_URL")?; + let separator = if url.contains('?') { '&' } else { '?' }; + let capped = + create_storage_from_connection_string(&format!("{url}{separator}max_bytes_billed=1")) + .await?; + // The statement that looks at the metadata of the dataset bills at least 10 MB. (A read of + // rows that were just written is not billed: they are in the streaming buffer, so a cap + // cannot be tested on them.) + let error = capped + .create_or_migrate() + .await + .expect_err("a statement over the cap fails"); + assert!( + matches!( + error.downcast_ref::(), + Some(StorageError::OperationFailed { .. }) + ), + "{error:#}" + ); + Ok(()) + } +} diff --git a/tests/integration/common/mod.rs b/tests/integration/common/mod.rs index 37366844..196f33c1 100644 --- a/tests/integration/common/mod.rs +++ b/tests/integration/common/mod.rs @@ -18,6 +18,7 @@ pub enum DatabaseType { TimescaleDB, DuckDB, RRDcached, + BigQuery, } impl DatabaseType { @@ -40,6 +41,9 @@ impl DatabaseType { .unwrap_or_else(|_| "duckdb://test.duckdb".to_string()), DatabaseType::RRDcached => std::env::var("TEST_DATABASE_URL") .unwrap_or_else(|_| "rrdcached://127.0.0.1:42217?preset=hoarder".to_string()), + // Costs money and needs a Google Cloud project: no default, see docs/BIGQUERY.md + DatabaseType::BigQuery => std::env::var("TEST_DATABASE_URL") + .unwrap_or_else(|_| "bigquery://?TEST_DATABASE_URL_is_not_set".to_string()), } } @@ -55,6 +59,8 @@ impl DatabaseType { DatabaseType::DuckDB } else if connection_string.starts_with("rrdcached://") { DatabaseType::RRDcached + } else if connection_string.starts_with("bigquery:") { + DatabaseType::BigQuery } else { // Default to PostgreSQL for postgres://, postgresql://, or any other prefix DatabaseType::PostgreSQL @@ -77,6 +83,8 @@ impl DatabaseType { DatabaseType::SQLite } else if cfg!(feature = "rrdcached") { DatabaseType::RRDcached + } else if cfg!(feature = "bigquery") { + DatabaseType::BigQuery } else { DatabaseType::PostgreSQL } @@ -114,6 +122,7 @@ impl TestDb { DatabaseType::TimescaleDB => "sensapp-test".to_string(), DatabaseType::DuckDB => "test.duckdb".to_string(), DatabaseType::RRDcached => "rrdcached".to_string(), + DatabaseType::BigQuery => "bigquery".to_string(), }; let connection_string = db_type.default_connection_string(); diff --git a/tests/integration/data_lifecycle.rs b/tests/integration/data_lifecycle.rs index d26a784b..297601d0 100644 --- a/tests/integration/data_lifecycle.rs +++ b/tests/integration/data_lifecycle.rs @@ -397,3 +397,4 @@ backend_tests!("sqlite", DatabaseType::SQLite, sqlite); backend_tests!("duckdb", DatabaseType::DuckDB, duckdb); backend_tests!("timescaledb", DatabaseType::TimescaleDB, timescaledb); backend_tests!("clickhouse", DatabaseType::ClickHouse, clickhouse); +backend_tests!("bigquery", DatabaseType::BigQuery, bigquery); diff --git a/tests/integration/deduplication.rs b/tests/integration/deduplication.rs index 0dac9dc3..2e7bc6aa 100644 --- a/tests/integration/deduplication.rs +++ b/tests/integration/deduplication.rs @@ -560,7 +560,7 @@ async fn only_the_backends_that_can_be_exact_accept_the_switch() -> Result<()> { let accepted = deduplicate_on_ingest(&storage, true).await?; let expected = !matches!( test_db.db_type, - DatabaseType::ClickHouse | DatabaseType::RRDcached + DatabaseType::ClickHouse | DatabaseType::RRDcached | DatabaseType::BigQuery ); assert_eq!(accepted, expected, "{:?}", test_db.db_type); Ok(()) diff --git a/tests/integration/main.rs b/tests/integration/main.rs index d577efcb..0479ff75 100644 --- a/tests/integration/main.rs +++ b/tests/integration/main.rs @@ -8,6 +8,7 @@ mod advanced_backend_queries; mod aggregated_windows; mod arrow_integration; mod batched_inserts; +mod bigquery_integration; mod clickhouse_http_lifecycle; mod clickhouse_integration; mod cross_series_reads;