diff --git a/.agents/skills/README.md b/.agents/skills/README.md index cafaa40..01d2d74 100644 --- a/.agents/skills/README.md +++ b/.agents/skills/README.md @@ -13,7 +13,7 @@ Wrappers around `python -m raincloud.pipeline.`. Side-effecting ones set | `/raincloud-build` | `raincloud.pipeline.build` | Full pipeline (fetch → … → write_canonical → validate → run_exporters) for one or more slugs. | | `/raincloud-fetch` | `raincloud.pipeline.fetch` | Download raw bytes only. | | `/raincloud-extract` | `raincloud.pipeline.extract` | Unpack archives into the recipe's scratch directory. | -| `/raincloud-export` | `raincloud.pipeline.export` | Re-derive Parquet/Vortex from the canonical Arrow already on disk, without refetching; the refresh path for one format. | +| `/raincloud-export` | `raincloud.pipeline.export` | Re-derive any format from the canonical Arrow already on disk, without refetching; the refresh path for one format, or for new encoder settings. | | `/raincloud-convert` | `raincloud.pipeline.convert` | Re-encode Vortex with the Python writer (v1 catalogs: from Parquet). For v2, prefer `/raincloud-export --format vortex`. | | `/raincloud-hydrate` | `raincloud.pipeline.hydrate` | Build a `-hydrated` dataset (URL columns fetched from the open web) with the safe defaults, or write a scratch sample with non-default options (`--limit`/`--block`/`--urlhaus`/`--max-bytes`/`--timeout`/bypass), never published or served. Side-effecting (outbound HTTP); safety-filter-gated; `disable-model-invocation: true`. | | `/raincloud-docs` | `raincloud.pipeline.docs` | Regenerate derived docs. *(model-invocable — regen is mostly idempotent.)* | @@ -35,6 +35,7 @@ These guide multi-step procedures from [`SKILLS.md`](../context/SKILLS.md). Defa | `/raincloud-add-handler` | Writing a new transform handler under `raincloud/pipeline/handlers/`. | | `/raincloud-add-kaggle-tos` | Adding a Kaggle dataset gated behind a one-time ToS click-through. | | `/raincloud-promote-variant` | JSON → VARIANT via the transform recipe and a rebuild. | +| `/raincloud-write-settings` | Write files with chosen encoder settings: a Parquet page index (all or the first N columns), compression level, statistics, dictionaries, checksums; ORC/Avro codecs; Vortex compact. | | `/raincloud-debug-build` | Diagnostic checklist for a failing build — isolate which stage broke. | | `/raincloud-large-build` | Run a memory- or runtime-heavy build safely (caps, nohup, logging). *(side-effecting — `disable-model-invocation: true`.)* | | `/raincloud-remove-dataset` | Remove a dataset from the manifest and clean up its outputs. *(destructive — `disable-model-invocation: true`.)* | diff --git a/.agents/skills/raincloud-build/SKILL.md b/.agents/skills/raincloud-build/SKILL.md index 9d5faeb..c0ab130 100644 --- a/.agents/skills/raincloud-build/SKILL.md +++ b/.agents/skills/raincloud-build/SKILL.md @@ -1,7 +1,7 @@ --- name: raincloud-build description: Run the full Raincloud pipeline (fetch → extract → parse → transform → canonical Arrow → validate → exports) for one or more dataset slugs. Use when the user asks to build a dataset, rebuild a slug, or process a batch. -argument-hint: ... | --all [--strict] [--clean-workdir] [--retry-errors] +argument-hint: ... | --all [--format FORMAT] [--only] [--strict] [--clean-workdir] [--retry-errors] disable-model-invocation: true allowed-tools: Bash(python -m raincloud.pipeline.build *) --- @@ -20,6 +20,10 @@ Modifiers: - `--strict` — make validation drift an error. By default, row-count mismatches are warnings; use that default for first builds with estimated counts. - `--clean-workdir` — clear the selected `.recipes///` scratch directory after each successful build. Essential for large batch runs (Public BI decompressed CSVs can hit ~100 GB). - `--retry-errors` — attempt a format even when its writer, with this toolchain, already failed to write it at this recipe (see below). Without it that format is skipped. +- `--format FORMAT` (repeatable or comma-separated) — write these formats instead of the install's `formats` setting (only `vortex` by default; `parquet`, `orc`, `avro`, `nimble`, or `arrow` to keep the canonical). Only the format is taken: a writer suffix (`parquet@rs`) is dropped, and the writer comes from `export.priority`. +- `--only` — for a generated table, build its whole group but keep only the tables named. + +Encoder settings — a Parquet page index, statistics for the first N columns, a compression level, dictionaries, page checksums, ORC/Avro codecs, Vortex compact encodings — are environment settings every writer of the format reads (`RAINCLOUD_PARQUET_*`, `RAINCLOUD_ORC_*`, `RAINCLOUD_AVRO_*`, `RAINCLOUD_VORTEX_*`); unset is each library's default. Pass them in the build's environment; see `/raincloud-write-settings` for what each does, which writer refuses which, and how to check the result. Before running: - **Confirm with the user** before triggering anything non-trivial. JSONBench 100M ≈ 6 h, Wikipedia Structured Contents → ~70 GB parquet, OSM Germany ~45 min per kind. Small (<100 MB) parquets are fine without asking. (See [AGENTS.md "Confirm before rebuilding"](../../context/AGENTS.md#confirm-before-rebuilding).) diff --git a/.agents/skills/raincloud-export/SKILL.md b/.agents/skills/raincloud-export/SKILL.md index 9695250..cbe973b 100644 --- a/.agents/skills/raincloud-export/SKILL.md +++ b/.agents/skills/raincloud-export/SKILL.md @@ -1,7 +1,7 @@ --- name: raincloud-export -description: Re-derive a dataset's Parquet and/or Vortex files from the canonical Arrow file already on disk, without refetching or re-transforming. Use when a change touches only the export stage (row-group sizing, a codec, a Vortex upgrade) or to refresh one format. -argument-hint: ... | --all [--format parquet|vortex|parquet@rs|...] [--dry-run] [--retry-errors] +description: Re-derive a dataset's Parquet, Vortex, ORC, Avro or Nimble files from the canonical Arrow file already on disk, without refetching or re-transforming. Use when a change touches only the export stage (row-group sizing, a codec, an encoder setting such as a Parquet page index, a writer upgrade) or to refresh one format. +argument-hint: ... | --all [--format parquet|vortex|orc|avro|nimble|parquet@rs|...] [--dry-run] [--retry-errors] disable-model-invocation: true allowed-tools: Bash(python -m raincloud.pipeline.export *) --- @@ -17,9 +17,10 @@ Selection (one required): - `--all` — every dataset. Hours of work on a full store; confirm first. Modifiers: -- `--format FORMAT` (repeatable) — export only this format, replacing the spec's `export.formats` for this run: `parquet` or `vortex`. The writer is chosen by `export.priority` as in a build. It overrides a dataset whose policy leaves the format out, so check `python -m raincloud.pipeline.list_datasets --no-vortex --json` before forcing Vortex. +- `--format FORMAT` (repeatable) — export only this format, replacing the install's formats for this run: `parquet`, `vortex`, `orc`, `avro` or `nimble`. The writer is chosen by `export.priority` as in a build. It overrides a dataset whose policy leaves the format out, so check `python -m raincloud.pipeline.list_datasets --no-vortex --json` before forcing Vortex. - `--format parquet@rs` (or another `@`) — use that writer for this run. The file is still `parquet/.parquet`, but its bytes and sha256 change, so it no longer matches the catalog until a maintainer regenerates it. Confirm before doing this to datasets others read. - `--dry-run` — list what would be exported and exit. +- Encoder settings come from the environment (`RAINCLOUD_PARQUET_PAGE_INDEX=1`, `RAINCLOUD_PARQUET_STATISTICS_COLUMNS=100`, `RAINCLOUD_PARQUET_COMPRESSION_LEVEL=9`, ...): this is the cheapest way to rewrite a file with them, since the canonical is the input. A writer that cannot honour a set one refuses (`[unavailable]`, ` cannot honour =...`); name another with `--format @`. See `/raincloud-write-settings`. - `--retry-errors` — attempt a format even when its writer, with this toolchain, already failed to write it at this recipe (see below). Behavior: diff --git a/.agents/skills/raincloud-load/SKILL.md b/.agents/skills/raincloud-load/SKILL.md index 54491c0..96fd2d6 100644 --- a/.agents/skills/raincloud-load/SKILL.md +++ b/.agents/skills/raincloud-load/SKILL.md @@ -1,7 +1,7 @@ --- name: raincloud-load description: Read prepared Raincloud artifacts, inspect catalog metadata, or load batches from local storage or a configured mirror. -argument-hint: [--format auto|arrow|parquet|vortex] +argument-hint: [--format auto|arrow|parquet|vortex|orc|avro|nimble] disable-model-invocation: true allowed-tools: Bash(python -m raincloud describe *), Bash(python -m raincloud list *), Bash(python -m raincloud load *), Bash(python -m raincloud config show), Bash(python -m raincloud capabilities), Bash(python examples/use_loader.py *) --- @@ -37,7 +37,13 @@ resolve the artifact as needed. `.to_arrow()` materializes the whole table. selected format for any engine (DuckDB, Polars, pyarrow). `.to_vortex()` requires `[vortex]` and a Vortex artifact. `.path()`, `.batches()`, `.to_arrow()` and `.dataset()` retain the selected artifact; `.to_vortex()` resolves the dataset's -Vortex file. Each format is one file; `describe` shows which writer made it. +Vortex file. Each format is one file; `describe` shows which writer made it. ORC reads +through pyarrow; Avro and Nimble are served by `.path()` only. + +A load serves the file it finds. Encoder settings (a Parquet page index, compression +level, ...) apply only when this install writes a file, so a file already on disk, in +the store or on a mirror is served as it was written, and `--build` builds only a file +that is missing. To get one written with settings, see `/raincloud-write-settings`. Use `raincloud capabilities` for installed reader modules and `raincloud config show` for effective paths. Optional TOML settings and environment overrides share diff --git a/.agents/skills/raincloud-write-settings/SKILL.md b/.agents/skills/raincloud-write-settings/SKILL.md new file mode 100644 index 0000000..93335bf --- /dev/null +++ b/.agents/skills/raincloud-write-settings/SKILL.md @@ -0,0 +1,95 @@ +--- +name: raincloud-write-settings +description: Write a dataset's files with chosen encoder settings — a Parquet page index (for every column or the first N), Parquet compression level, statistics, page sizes, dictionaries or page checksums; the ORC or Avro codec and level; Vortex compact encodings or block sizes. Use when the user wants page statistics / a page index / column indexes in Parquet, a different codec or compression level, or to compare encoder settings across writers. +argument-hint: [setting=value ...] +--- + +Encoder settings are **install settings in the environment**, read by every writer of the +format alike (`raincloud/pipeline/spec.py`: `_PARQUET_SETTINGS`, `FORMAT_SETTINGS`). They are +not recipe fields and not load options. The full list, with defaults, is the +`RAINCLOUD_PARQUET_*`, `RAINCLOUD_ORC_*`, `RAINCLOUD_AVRO_*` and `RAINCLOUD_VORTEX_*` rows of +the table in [AGENTS.md "Data locations"](../../context/AGENTS.md#data-locations); which +writer honours which is tabulated in [`sidecars/README.md`](../../../sidecars/README.md). + +## What to know before changing a setting + +- **Unset means each writer library's own default**, and is what every published file was + written with. pyarrow (the default Parquet writer, `parquet@py`) writes **no page + index** unless asked; arrow-rs and parquet-java write one. +- **A setting applies to files this install writes.** It does not change a file already on + disk, in this machine's store, or on a mirror, and `raincloud load` serves whichever of + those it finds first. To get a file with the setting, write it again (below). +- **A file written with a setting differs from the catalog's** (another sha256). The build + record (`/builds.json`) records it, and the loader serves this install's file. + Do not do this to a shared store others read without asking. +- **A writer that cannot honour a set value refuses**, rather than writing something else: + the format is recorded unavailable with the reason (`[unavailable] /parquet: + parquet@py cannot honour RAINCLOUD_PARQUET_PAGE_INDEX_COLUMNS=100: ...`). When the default + writer refuses, pick another: `export --format parquet@rs` names it for one export; a + build takes only the format, so set the machine's writer preference instead + (`RAINCLOUD_EXPORT_PRIORITY=rs,py`; a recipe's or the catalog's `export.priority` outranks + it). arrow-rs (`rs`) is the writer for a page index on the first N columns only. +- **Contradictions are refused before any writer runs**: a page index while the recipe turns + statistics off, `RAINCLOUD_PARQUET_PAGE_INDEX=0` with `_PAGE_INDEX_COLUMNS`, a level for a + codec without levels, a level out of range. A malformed value names the variable. +- **A set value is part of the writer's toolchain**, so a failure recorded without it is + attempted again, and one recorded with it is skipped until something changes. + +## Parquet page index + +```bash +# every column; the default writer (pyarrow) does this +export RAINCLOUD_PARQUET_PAGE_INDEX=1 +# or: page statistics for the first 100 leaf columns only, chunk statistics for all +# (arrow-rs only: pyarrow writes a page index for all columns or none) +export RAINCLOUD_PARQUET_PAGE_INDEX_COLUMNS=100 +# or: all statistics, chunk and page, for the first 100 leaf columns only +# (pyarrow, arrow-rs and parquet-java; Hardwood refuses) +export RAINCLOUD_PARQUET_STATISTICS_COLUMNS=100 +``` + +"Leaf columns" are the Parquet column chunks, in schema order: a struct or list contributes +one per leaf field. + +## Writing the file again + +```bash +# The canonical Arrow is on disk (keep_canonical, or a maintainer's store): re-derive only +python -m raincloud.pipeline.export --format parquet +# Otherwise rebuild (refetches unless the raw download was kept) +python -m raincloud.pipeline.build --format parquet +``` + +`raincloud load --format parquet --build` builds only when no file is found, so it +does not rewrite an existing one. Confirm with the user before rebuilding anything large +([AGENTS.md "Confirm before rebuilding"](../../context/AGENTS.md#confirm-before-rebuilding)). + +## Checking the result + +```python +import pyarrow.parquet as pq, raincloud +meta = pq.ParquetFile(raincloud.load("", format="parquet").path()).metadata +chunks = [meta.row_group(g).column(c) for g in range(meta.num_row_groups) for c in range(meta.num_columns)] +print(sum(c.has_column_index for c in chunks), "of", len(chunks), "column chunks have a page index") +``` + +`raincloud describe ` shows the writer that made the file; an `[unavailable]` line or a +`FormatUnavailable` from `load` quotes a writer's refusal. + +## Other settings + +| want | set | +|---|---| +| Parquet zstd/gzip/brotli level | `RAINCLOUD_PARQUET_COMPRESSION_LEVEL` (zstd 1-22, gzip 0-9, brotli 0-11; Hardwood refuses) | +| Parquet page size / rows per page | `RAINCLOUD_PARQUET_PAGE_BYTES`, `RAINCLOUD_PARQUET_PAGE_ROWS` | +| Parquet without dictionaries | `RAINCLOUD_PARQUET_DICTIONARY=0` | +| Parquet page checksums | `RAINCLOUD_PARQUET_PAGE_CHECKSUMS=1` (arrow-rs refuses) or `=0` (Hardwood refuses) | +| ORC codec / stripes | `RAINCLOUD_ORC_COMPRESSION`, `RAINCLOUD_ORC_STRIPE_BYTES`, `RAINCLOUD_ORC_COMPRESSION_BLOCK_BYTES` | +| Avro codec / level / blocks | `RAINCLOUD_AVRO_COMPRESSION`, `RAINCLOUD_AVRO_COMPRESSION_LEVEL`, `RAINCLOUD_AVRO_BLOCK_BYTES` (the last two Avro Java only) | +| Vortex compact encodings | `RAINCLOUD_VORTEX_COMPACT=1` (vortex@jni refuses) | + +The Parquet codec itself is the recipe's `write.compression`, which wins over the +environment; changing it is a recipe edit, which changes the recipe hash. + +Context: [AGENTS.md "How a build works"](../../context/AGENTS.md#how-a-build-works), +[SKILLS.md](../../context/SKILLS.md#writing-files-with-chosen-encoder-settings). diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 4b41a03..0d05fba 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -72,8 +72,8 @@ jobs: 17 cache: gradle - - name: Build + test JVM sidecars (conformance-common + parquet@java, parquet@hardwood, vortex@jni lanes) - run: ./sidecars/java/gradlew -p sidecars/java :conformance-common:test :parquet-java:test :parquet-java:installDist :parquet-hardwood:test :parquet-hardwood:installDist :vortex-jni-reader:test :vortex-jni-reader:installDist + - name: Build + test JVM sidecars (conformance-common + parquet@java, parquet@hardwood, vortex@jni, avro@java lanes) + run: ./sidecars/java/gradlew -p sidecars/java :conformance-common:test :parquet-java:test :parquet-java:installDist :parquet-hardwood:test :parquet-hardwood:installDist :vortex-jni-reader:test :vortex-jni-reader:installDist :avro-java:test :avro-java:installDist - uses: astral-sh/setup-uv@v5 with: @@ -89,6 +89,8 @@ jobs: RAINCLOUD_READER_PARQUET_HARDWOOD: ${{ github.workspace }}/sidecars/java/parquet-hardwood/build/install/raincloud-export-parquet-hardwood/bin/raincloud-read-parquet-hardwood RAINCLOUD_SIDECAR_VORTEX_JNI: ${{ github.workspace }}/sidecars/java/vortex-jni-reader/build/install/raincloud-read-vortex-jni/bin/raincloud-export-vortex-jni RAINCLOUD_READER_VORTEX_JNI: ${{ github.workspace }}/sidecars/java/vortex-jni-reader/build/install/raincloud-read-vortex-jni/bin/raincloud-read-vortex-jni + RAINCLOUD_SIDECAR_AVRO_JAVA: ${{ github.workspace }}/sidecars/java/avro-java/build/install/raincloud-export-avro-java/bin/raincloud-export-avro-java + RAINCLOUD_READER_AVRO_JAVA: ${{ github.workspace }}/sidecars/java/avro-java/build/install/raincloud-export-avro-java/bin/raincloud-read-avro-java run: uv run --no-sync pytest -q tests/test_reader_fidelity.py tests/test_java_reader_schema_compare.py tests/test_jvm_sidecar_lanes.py wheel: diff --git a/AGENTS.md b/AGENTS.md index 76300e7..e31447a 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -88,9 +88,18 @@ every output format is derived from it by an exporter. | transform | `transform.py` | `transform.*` | in-memory `(slug, Table)` | | write_canonical | `canonical.py` | transform output | `outputs/v{n}//arrow/.arrow.zstd` | | validate | `validate.py` | `expect.*` | hashes canonical schema, checks rows; `[WARN]` unless `--strict` | -| run_exporters | `export/` | `export.formats`, `export.priority` | `parquet/`, `vortex/` under `outputs/v{n}//`; the build record | +| run_exporters | `export/` | the install's `formats`, `export.priority` | `/` under `outputs/v{n}//`; the build record | | hydrate *(named builds only)* | `hydrate.py` | `derive.hydrate` | a `-hydrated` dataset — outbound HTTP, safety-filter gated | +**Formats are opt-in, per install.** A dataset offers every exported format (a recipe's +`export.formats` can only narrow that); a build writes the install's `formats` setting — +only Vortex by default — or what `--format` names, and then removes the raw download and +the canonical unless `keep_raw` / `keep_canonical` are set (a canonical from which no +format was written stays: it is the dataset's file). Maintaining the catalog wants +everything, so a maintainer's config (or environment) sets `formats = "all"`, +`keep_raw = true` and `keep_canonical = true`; without them a checkout build deletes the +raw bytes a re-run would reuse. + `run_exporters` is also invokable on its own, which is the whole job whenever a change touches only the export stage (row-group sizing, a codec, a new cell) — the canonical is the input and is left alone: @@ -127,7 +136,7 @@ reject it in a v2 manifest, while a released v2 catalog that still carries `convert.vortex: false` (and no `export.formats`) keeps reading as Parquet-only. A format is one file whichever writer makes it. The writer is the first *installed* one in `export.priority`, looked up in the spec, then the catalog's `export_priority`, then -`RAINCLOUD_EXPORT_PRIORITY`, then the built-in `py, rs, java`. The spec and catalog +`RAINCLOUD_EXPORT_PRIORITY`, then the built-in `py, rs, java, cpp`. The spec and catalog levels take a list, which applies to every format and so must name a writer for each one the dataset exports, or a map from format to list (`{"parquet": ["rs", "py"]}`); a format the map leaves out falls through to the next level. The machine level is a @@ -136,6 +145,20 @@ list. The build record, and after regeneration the catalog, records the writer a `vortex@rs`, `vortex@jni`) run only where their binary is installed. Compliance measures every writer in scratch, never over the dataset's file. +Every Parquet writer is given one set of options (`spec.parquet_options`): the recipe's +`write.compression`, `write.statistics` and row cap, and the install's +`RAINCLOUD_PARQUET_*` settings in the table below (compression level, statistics and the +page index for all or the first N columns, page size and rows, dictionaries, page +checksums). An unset setting leaves each library's own default, which differ (pyarrow +writes no page index; arrow-rs and parquet-java do). A set one reaches every writer and +becomes part of its toolchain, and a writer whose library cannot do what it asks fails +that export as a measurement rather than writing something else; `sidecars/README.md` +tabulates which writer honours what, and the `/raincloud-write-settings` skill is the +procedure (a setting changes only files this install writes, so an existing file must be +re-exported or rebuilt to gain it). ORC, Avro and Vortex have the same kind of settings +(`spec.FORMAT_SETTINGS`, `RAINCLOUD_ORC_*`, `RAINCLOUD_AVRO_*`, `RAINCLOUD_VORTEX_*`), read +and refused the same way; their codec, unset, is the zstd raincloud has always written. + `export.formats` lists the formats a dataset wants. When the planned writer cannot produce one for the dataset -- it raises, dies, reports a failed round-trip, or exceeds `RAINCLOUD_EXPORT_TIMEOUT` -- the previous file comes back and the build records the @@ -237,6 +260,9 @@ its own `outputs/` rather than a machine's shared store. | `RAINCLOUD_MIRROR` | a private artifact store readers fall back to (`s3://` needs `[s3]`, `https://` needs `[http]`) | unset | | `RAINCLOUD_OFFLINE` | `1`: read only local files; never contact the mirror | unset | | `RAINCLOUD_RETRY_ERRORS` | `1`: a build attempts a format whose writer, with this toolchain, already failed at the recipe (as `--retry-errors`) | unset | +| `RAINCLOUD_FORMATS` | formats a build writes, e.g. `vortex,parquet`, or `all` (as `--format` for one build) | `vortex` | +| `RAINCLOUD_KEEP_RAW` | `1`: a successful build keeps the raw download (generated datasets always keep their generator output) | unset (removed) | +| `RAINCLOUD_KEEP_CANONICAL` | `1`: a successful build keeps the canonical Arrow (kept anyway when no other format was written) | unset (removed) | | `RAINCLOUD_CONFIG` / `RAINCLOUD_NO_CONFIG` | select or disable the config file | unset | | `RAINCLOUD_SETTINGS` | settings JSON the CLI reads with `--settings-env`; how native readers pass options | unset | | `RAINCLOUD_DUCKDB_MEMORY_LIMIT` | DuckDB memory ceiling, applied by `raincloud.duckdb_connect` | DuckDB default (~80% RAM) | @@ -250,8 +276,27 @@ its own `outputs/` rather than a machine's shared store. | `RAINCLOUD_ROW_GROUP_MAX_ROWS` | row cap per group, used only when a spec omits `write.row_group_size_rows`; a spec's cap wins in every writer, sidecars included | 10,000,000 | | `RAINCLOUD_ROW_GROUP_TARGET_BYTES` | memory guard: decoded Arrow bytes buffered for one row group | 512 MiB | | `RAINCLOUD_ROW_GROUP_PROBE_ROWS` | rows the Python Parquet writer samples to size its groups (must be > 0) | 262,144 | +| `RAINCLOUD_PARQUET_COMPRESSION_LEVEL` | the recipe's codec at this level in every Parquet writer (zstd 1-22, gzip 0-9, brotli 0-11) | unset (each writer's own) | +| `RAINCLOUD_PARQUET_STATISTICS_COLUMNS` | statistics (chunk and page) only for the first N leaf columns (`0`: every column) | unset (each writer's own: every column) | +| `RAINCLOUD_PARQUET_PAGE_INDEX` | `1`: every Parquet writer writes a page index (ColumnIndex + OffsetIndex); `0`: none | unset (each writer's own: pyarrow and Hardwood none, arrow-rs and parquet-java one) | +| `RAINCLOUD_PARQUET_PAGE_INDEX_COLUMNS` | page statistics only for the first N leaf columns, chunk statistics for every column (`0`: every column) | unset (each writer's own) | +| `RAINCLOUD_PARQUET_PAGE_BYTES` | data page size target in every Parquet writer, each measuring a page its own way (`0`: no limit) | unset (each writer's own) | +| `RAINCLOUD_PARQUET_PAGE_ROWS` | data page row limit in every Parquet writer (`0`: no limit) | unset (each writer's own) | +| `RAINCLOUD_PARQUET_DICTIONARY` | `0`: no dictionary encoding (PLAIN); `1`: dictionaries | unset (each writer's own: on) | +| `RAINCLOUD_PARQUET_DICTIONARY_PAGE_BYTES` | dictionary page size limit, past which a column falls back to PLAIN (`0`: no limit) | unset (each writer's own) | +| `RAINCLOUD_PARQUET_PAGE_CHECKSUMS` | `1`: a CRC in every page header; `0`: none | unset (each writer's own: parquet-java and Hardwood write them, pyarrow and arrow-rs do not) | +| `RAINCLOUD_ORC_COMPRESSION` | ORC codec in every ORC writer: `zstd`, `snappy`, `zlib`, `lz4`, `none` | unset (zstd, as always) | +| `RAINCLOUD_ORC_COMPRESSION_STRATEGY` | `speed` or `compression` (ORC C++ only; orc-rust refuses) | unset (each writer's own) | +| `RAINCLOUD_ORC_STRIPE_BYTES` | ORC stripe size target, measured encoded and compressed | unset (each writer's own) | +| `RAINCLOUD_ORC_COMPRESSION_BLOCK_BYTES` | ORC compression block size (ORC C++ takes only multiples of 64 KiB) | unset (each writer's own) | +| `RAINCLOUD_AVRO_COMPRESSION` | Avro codec in every Avro writer: `zstd`, `deflate`, `snappy`, `bzip2`, `xz`, `none` | unset (zstd, as always) | +| `RAINCLOUD_AVRO_COMPRESSION_LEVEL` | level for zstd (1-22), deflate or xz (0-9); Avro Java only, arrow-avro refuses | unset (each writer's own) | +| `RAINCLOUD_AVRO_BLOCK_BYTES` | Avro block (sync interval) size; Avro Java only, arrow-avro refuses | unset (each writer's own) | +| `RAINCLOUD_VORTEX_COMPACT` | `1`: BtrBlocks' compact encodings (vortex@py and vortex@rs; vortex@jni refuses) | unset (each writer's own: default) | +| `RAINCLOUD_VORTEX_ROW_BLOCK_ROWS` | Vortex row block size (vortex@rs only) | unset (each writer's own) | +| `RAINCLOUD_VORTEX_DATA_BLOCK_BYTES` | Vortex data block size target (vortex@rs only) | unset (each writer's own) | | `RAINCLOUD_BATCH_ROWS` / `RAINCLOUD_BATCH_BYTES` | batch bounds in the streaming ingestion paths (memory only, NOT the row-group size) | 4096 rows / 16 MiB | -| `RAINCLOUD_EXPORT_PRIORITY` | machine writer preference, e.g. `rs,py` | unset (`py, rs, java`) | +| `RAINCLOUD_EXPORT_PRIORITY` | machine writer preference, e.g. `rs,py` | unset (`py, rs, java, cpp`) | | `RAINCLOUD_EXPORT_TIMEOUT` | ceiling on one export: an in-process writer (run in a child process) or a sidecar writer; hitting it records the format unavailable | 6 h (`0` disables) | | `RAINCLOUD_EXPORT_MEMORY` | ceiling on one in-process export's resident memory (bytes); the parent stops a writer over it and records the format unavailable | half of physical memory (`0` disables) | | `RAINCLOUD_SIDECAR_TIMEOUT` | ceiling on one sidecar reader call (sidecar writers use `RAINCLOUD_EXPORT_TIMEOUT`) | 30 min (`0` disables) | diff --git a/CHANGELOG.md b/CHANGELOG.md index 678f5e1..a52447a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,107 @@ All notable changes to this project will be documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project follows [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [0.3.1] - 2026-10-08 + +The compliance ledger, `docs/v2/compliance.json`, is still 0.3.0's measurement: the ORC, +Avro and Nimble writers and readers added here, and the encoder settings, are measured +in a later release. + +### Added + +- **Parquet write options, the same in every writer.** Install settings that reach + `parquet@py`, `parquet@rs`, `parquet@java` and `parquet@hardwood` alike: + `RAINCLOUD_PARQUET_COMPRESSION_LEVEL`, `_STATISTICS_COLUMNS` (statistics only for the + first N leaf columns), `_PAGE_INDEX` (a ColumnIndex and OffsetIndex for every column + chunk, or none), `_PAGE_INDEX_COLUMNS` (page statistics only for the first N leaf + columns), `_PAGE_BYTES`, `_PAGE_ROWS`, `_DICTIONARY`, `_DICTIONARY_PAGE_BYTES` and + `_PAGE_CHECKSUMS`. The recipe's `write.compression` and `write.statistics` now reach the + three sidecar writers too, which used to pick their own. `parquet@java` moves to + parquet-arrow-java 0.3.0, for a compression level, per-column statistics and LZ4_RAW, + and `parquet@hardwood` ships brotli4j, so every Parquet writer but `parquet@java` + writes Brotli. Unset, a setting leaves each + library's default as before, so no file changes: pyarrow and Hardwood write no page + index, arrow-rs and parquet-java write one; parquet-java and Hardwood write page + checksums, the others do not. A set one is part of the writer's toolchain, so changing + it retries a recorded failure. A writer whose library cannot do what is asked records + Parquet unavailable rather than writing something else (`sidecars/README.md` tabulates + which writer honours what). + +- **ORC, Avro and Vortex write settings.** The same kind of install settings for the + other formats, given alike to every writer of the format and refused by a writer whose + library cannot honour them: `RAINCLOUD_ORC_COMPRESSION`, `_COMPRESSION_STRATEGY`, + `_STRIPE_BYTES`, `_COMPRESSION_BLOCK_BYTES`; `RAINCLOUD_AVRO_COMPRESSION` (every Avro + codec: `avro@rs` now builds arrow-avro's deflate, snappy, bzip2 and xz too, and + `avro@java` ships Avro's optional snappy-java 1.1.10.8 and xz 1.10), `_COMPRESSION_LEVEL`, + `_BLOCK_BYTES`; `RAINCLOUD_VORTEX_COMPACT`, `_ROW_BLOCK_ROWS`, `_DATA_BLOCK_BYTES`. Unset, + each writer writes what it did before. +- **ORC.** A dataset can be built as ORC (`formats = ["orc"]`, or + `raincloud build --format orc`) and loaded with `format="orc"`; `auto` never picks + it. Two writers: `orc@py`, pyarrow's Apache ORC C++ library, and `orc@rs`, + orc-rust 0.9.0 in the Rust sidecar (`orc-write` / `orc-read`). Both write zstd. + Neither has a type converted for it: a column the library does not write + (unsigned integers and string views for pyarrow; anything beyond signed + integers, floats, strings, binary, booleans, dates and timestamps for orc-rust) + records ORC unavailable for that dataset, with the library's error. + +- **Avro.** A dataset can be built as an Avro object container file + (`formats = ["avro"]`, `--format avro`). Two writers, both sidecars: `avro@rs`, + arrow-avro 59.2 (`avro-write` / `avro-read`), and `avro@java`, Arrow Java 19.0.0's Avro + adapter (the new `avro-java` Gradle project). Both write zstandard and the same fixed + sync marker, so a rebuild gives the same bytes. pyarrow has no Avro support, so + `load(..., format="avro")` serves the file's `path()` and its readers raise + `MissingDependency`. +- **Nimble.** A dataset can be built as a Nimble file (`formats = ["nimble"]`, + `--format nimble`) by `nimble@cpp`: upstream Nimble's C++ writer and reader, built from + source by `sidecars/nimble/build.sh` into `raincloud-export-nimble-cpp` / + `raincloud-read-nimble-cpp`, which link the Rust crate's new `nimble-ffi` member for the + sidecar contract and hand batches across in memory. Served by path. The built-in writer + order is `py, rs, java, cpp, canonical`. +- **One comparison rule set for every lane.** Where equality turns on representation + rather than data, the Python, Rust and JVM comparators follow the cases in + `sidecars/compare_cases` (pairs of Arrow files and their verdicts, read by every + lane's tests): a union of exactly `null` and `T` is a nullable `T`; zoned timestamps + compare by instant, whatever zone labels them; an integer and a scale-0 decimal holding + the same values are equal. +- **Generated groups build together.** A generated table is one of a group its generator + writes at once (every TPC-H table of a scale factor, say): building one builds the whole + group in the same formats, and its generator output is removed once the group has + built, unless `keep_raw` keeps it. `raincloud build --only` builds just the tables named. +- A format raincloud only serves by path loads like any other: `path()` works and + `to_arrow()` / `batches()` / `dataset()` raise `MissingDependency`. + +### Changed + +- **Formats are opt-in, per install.** A build writes only Vortex unless the new + `formats` setting (`RAINCLOUD_FORMATS`; `"all"` for every format) or + `raincloud build --format` asks for more, and a load that names another format + builds just that one. `format="auto"` picks among the formats the install builds, + Vortex then Parquet, and falls back to the canonical Arrow when neither can be had. + Every dataset now offers every exported format; a recipe's `export.formats` only + narrows that. To keep 0.3.0's behaviour, set `formats = ["vortex", "parquet"]`. +- **A build cleans up after itself.** Once a dataset's files are written, a + successful build removes its raw download and its canonical Arrow, unless the new + `keep_raw` / `keep_canonical` settings keep them. A canonical from which no other + format was written is the dataset's only file and stays. Generated datasets keep + their generator output, which a group of datasets shares. A shared store or a + maintainer's checkout sets `formats = "all"`, `keep_raw = true` and + `keep_canonical = true`. +- `python -m raincloud.pipeline.export` without `--format`, and `run_exporters` / + `plan` without `formats`, write the install's formats, as a build does, instead of + every format the recipe lists. +- **Both ORC lanes widen what ORC cannot hold**, always rather than by the data's range: + uint8 → int16, uint16 → int32, uint32 → int64, uint64 → decimal(20, 0), and view types + to their plain types. orc-rust writes no decimals, so a uint64 column is still + unavailable in `orc@rs`. +- **`export.formats` is no longer part of a recipe's fingerprint**: an install chooses + what it builds, so which formats a dataset offers decides no file's bytes. The four + recipes that restated the old default drop the line; their fingerprints change once. +- **v1 catalogs behave as in 0.3.0**: a build writes every format the recipe lists and + removes nothing, and `auto` keeps the vortex, parquet, arrow order. +- Each artifact format is declared once, in `raincloud._registry.FORMATS`, and + everything that listed Parquet and Vortex by hand derives from it. A format with + no in-process reader raises `MissingDependency` instead of being opened as Vortex. + ## [0.3.0] - 2026-09-28 A breaking release: the pipeline's entry points, the default data locations and @@ -882,6 +983,7 @@ This release bundles: this repository" button in the repo sidebar with BibTeX / APA / Chicago exports. +[0.3.1]: https://github.com/spiraldb/raincloud/releases/tag/v0.3.1 [0.3.0]: https://github.com/spiraldb/raincloud/releases/tag/v0.3.0 [0.2.1]: https://github.com/spiraldb/raincloud/releases/tag/v0.2.1 [0.2.0]: https://github.com/spiraldb/raincloud/releases/tag/v0.2.0 diff --git a/CITATION.cff b/CITATION.cff index 2dad3fe..7fafb01 100644 --- a/CITATION.cff +++ b/CITATION.cff @@ -7,8 +7,8 @@ abstract: >- transform, validate, and convert behaviour, supporting verifiable provenance and consistent re-derivation of dataset artefacts across users and machines. type: software -version: 0.3.0 -date-released: 2026-09-28 +version: 0.3.1 +date-released: 2026-10-08 url: "https://github.com/spiraldb/raincloud" repository-code: "https://github.com/spiraldb/raincloud" license: Apache-2.0 diff --git a/README.md b/README.md index 99aca1b..53163d7 100644 --- a/README.md +++ b/README.md @@ -128,13 +128,24 @@ raincloud.load("uci-iris", build=True) raincloud load uci-iris --build ``` -**Which format you get.** A dataset can have Arrow IPC, Parquet and Vortex files. -`format="auto"`, the default, picks Vortex, then Parquet, then Arrow, among the -formats you have a reader for. The base install reads Arrow and Parquet; `[vortex]` -adds Vortex, and `[build]` includes it. Installing either one therefore changes what -`auto` returns, down to the Arrow types: Vortex returns some string columns as -`string_view`. Name the format when that matters: -`raincloud.load("uci-iris", format="parquet")`. +**Which format you get.** A dataset can have Arrow IPC, Parquet, Vortex, ORC, Avro and +Nimble files (Avro and Nimble are served by path: raincloud has no Python reader for them). +An install builds only Vortex unless it opts into more: the `formats` setting +(`formats = ["vortex", "parquet"]` in the config file, `RAINCLOUD_FORMATS`, or +`"all"`) names what a build writes, and a load that names another format builds just +that one. `format="auto"`, the default, picks Vortex, then Parquet, among the formats +the install builds and you have a reader for, and falls back to the canonical Arrow +when neither can be had (for a dataset whose Vortex writer is measured unable to write +it, say). The base install reads Arrow and Parquet; `[vortex]` adds Vortex, and +`[build]` includes it. Installing either one therefore changes what `auto` returns, +down to the Arrow types: Vortex returns some string columns as `string_view`. Name the +format when that matters: `raincloud.load("uci-iris", format="parquet")`. + +A build removes what it was made from once the dataset's files are written: the raw +download and the canonical Arrow. Set `keep_raw` and `keep_canonical` (or +`RAINCLOUD_KEEP_RAW=1`, `RAINCLOUD_KEEP_CANONICAL=1`) to keep them, as a shared store +or a maintainer does; a later request for another format then needs no fetch. A +canonical that is the dataset's only file is always kept. Other extras: `[s3]` and `[http]` add mirror transports, `[pandas]` backs `.to_pandas()`. @@ -266,6 +277,32 @@ writes the setting, and `raincloud config show` prints what is in effect. Some datasets are large: single datasets can run for hours, and a full catalog build is measured in days. Build what you need. +**Encoder settings.** Each format's writers take the same settings from the environment: +for Parquet a page index (ColumnIndex and OffsetIndex) on every column or only the first N, +statistics for only the first N columns, the compression level, page size and rows, +dictionaries and page checksums; for ORC and Avro the codec, level and block sizes; for +Vortex compact encodings. Unset, each writer library's default applies, which is how the +catalog's files are written: pyarrow, the default Parquet writer, writes no page index. To +build a dataset's Parquet with one: + +```bash +RAINCLOUD_PARQUET_PAGE_INDEX=1 raincloud build uci-iris --format parquet +``` + +```python +import os, raincloud +os.environ["RAINCLOUD_PARQUET_PAGE_INDEX"] = "1" # before the build runs +ds = raincloud.load("uci-iris", format="parquet", build=True) +``` + +A setting applies to files this install writes; a file already on disk or on a mirror is +served as it is, and `build=True` builds only what is missing, so rebuild (or re-export +from a kept canonical, `python -m raincloud.pipeline.export uci-iris --format parquet`) to +rewrite one. A writer that cannot do what a setting asks refuses it, and the format is +recorded unavailable with the reason. Every setting, its default, and which writer honours +it: the `RAINCLOUD_PARQUET_*`, `_ORC_*`, `_AVRO_*` and `_VORTEX_*` rows of +[AGENTS.md](AGENTS.md#data-locations) and [`sidecars/README.md`](sidecars/README.md). + A build does not fail because one format's writer cannot handle a dataset. Every export is bounded by `RAINCLOUD_EXPORT_TIMEOUT` (default 6 h; `0` disables it); a writer that raises, crashes, reports a failed round-trip or runs out of time leaves diff --git a/SKILLS.md b/SKILLS.md index dc8de4b..67354aa 100644 --- a/SKILLS.md +++ b/SKILLS.md @@ -15,6 +15,7 @@ Prereqs: Python 3.11+ and [uv](https://docs.astral.sh/uv/). A bare `uv sync --in - [Validating `sources.json`](#validating-sourcesjson) — schema + cross-checks - [Adding a new dataset](#adding-a-new-dataset) - [Emitting a Vortex file alongside the Parquet](#emitting-a-vortex-file-alongside-the-parquet) +- [Writing files with chosen encoder settings](#writing-files-with-chosen-encoder-settings) — Parquet page index, codecs, levels - [Adding a Kaggle dataset gated behind ToS acceptance](#adding-a-kaggle-dataset-gated-behind-tos-acceptance) - [Adding a new transform handler](#adding-a-new-transform-handler) - [Writing a streaming handler](#writing-a-streaming-handler) @@ -198,7 +199,7 @@ The companion [`sources.schema.md`](sources.schema.md) is the human-friendly ref Vortex (https://github.com/spiraldb/vortex) is one of the default exports: under `schema_version` 2 a dataset exports `vortex/.vortex` from its canonical Arrow file, beside `parquet/.parquet`. -1. `export.formats` is the only declaration of which formats a dataset wants. Without it, the defaults export Parquet and Vortex. A deliberate policy may leave Vortex out (`"export": {"formats": ["parquet"]}`, with `notes` if the reason is worth keeping). Do **not** leave it out because the Vortex writer fails on the data: keep it listed and let the build measure that. (`convert.vortex` is v1-only; v2 validation rejects it.) +1. `export.formats` is the only declaration of which formats a dataset offers. Without it, a dataset offers every exported format, and a build writes the ones the install's `formats` setting names (only Vortex by default; `all`, or `RAINCLOUD_FORMATS=vortex,parquet`, for more). A deliberate policy may leave Vortex out (`"export": {"formats": ["parquet"]}`, with `notes` if the reason is worth keeping). Do **not** leave it out because the Vortex writer fails on the data: keep it listed and let the build measure that. (`convert.vortex` is v1-only; v2 validation rejects it.) 2. Build all requested exports, or refresh only the Vortex file from the canonical already on disk: @@ -217,6 +218,44 @@ Other format-level caveats: - VARIANT columns surface as their shredded struct in Vortex (the VARIANT logical annotation isn't preserved), which can make a `.vortex` file much larger than the Parquet on heavily nested data such as Open Library. - Expect per-file overhead to dominate on very small datasets — the Vortex/Parquet size ratio can exceed 1.0 below a few MB. +## Writing files with chosen encoder settings + +Encoder settings are install settings in the environment, given alike to every writer of a +format: for Parquet the page index, statistics, compression level, page size and rows, +dictionaries and page checksums (`RAINCLOUD_PARQUET_*`); the ORC and Avro codec, level and +block sizes (`RAINCLOUD_ORC_*`, `RAINCLOUD_AVRO_*`); Vortex compact encodings and block sizes +(`RAINCLOUD_VORTEX_*`). Every variable and its default is in +[AGENTS.md "Data locations"](AGENTS.md#data-locations); which writer honours which is in +[`sidecars/README.md`](sidecars/README.md). The `/raincloud-write-settings` skill walks +through it. + +1. Unset, each writer library's own default applies, and that is what every published + file was written with. pyarrow, the default Parquet writer, writes no page index unless + asked. + +2. Set what you want and write the file again. A setting changes only files this install + writes; `raincloud load` serves a file already on disk, in the store or on a mirror as + it is. + + ```bash + export RAINCLOUD_PARQUET_PAGE_INDEX=1 # a page index on every column + # export RAINCLOUD_PARQUET_STATISTICS_COLUMNS=100 # or: statistics only for the first 100 leaf columns + python -m raincloud.pipeline.export --format parquet # the canonical is on disk + python -m raincloud.pipeline.build --format parquet # otherwise + ``` + +3. A writer whose library cannot do what a setting asks refuses it: the format is recorded + unavailable with ` cannot honour =: `, and the build goes + on with the formats that worked. pyarrow cannot put a page index on only the first N + columns (`RAINCLOUD_PARQUET_PAGE_INDEX_COLUMNS`); arrow-rs can, so name it: + `export --format parquet@rs`, or for a build `RAINCLOUD_EXPORT_PRIORITY=rs,py`. + Settings that contradict each other or the recipe are refused before any writer runs. + +4. The file now differs from the catalog's. The build record keeps its sha256 and the + writer's toolchain, which includes every setting that was set, so a recorded failure is + retried when a setting changes. Check a Parquet file's page index with + `pq.ParquetFile(path).metadata.row_group(0).column(0).has_column_index`. + ## Adding a Kaggle dataset gated behind ToS acceptance Some Kaggle datasets (many academic re-uploads, certain restricted-licence mirrors) require a one-time click-through acceptance of the dataset's distribution terms on the Kaggle web UI before the API will serve downloads. Attempting to fetch one of these returns HTTP 403. (Note: 403 can also indicate the slug itself is wrong — double-check the Kaggle URL before reaching for this pattern.) diff --git a/clients/rust/Cargo.lock b/clients/rust/Cargo.lock index ba840b9..17415c2 100644 --- a/clients/rust/Cargo.lock +++ b/clients/rust/Cargo.lock @@ -2015,7 +2015,7 @@ checksum = "dc33ff2d4973d518d823d61aa239014831e521c75da58e3df4840d3f47749d09" [[package]] name = "raincloud-reader" -version = "0.3.0" +version = "0.3.1" dependencies = [ "arrow-array", "arrow-ipc", diff --git a/clients/rust/Cargo.toml b/clients/rust/Cargo.toml index aa4ebbd..6f93ae9 100644 --- a/clients/rust/Cargo.toml +++ b/clients/rust/Cargo.toml @@ -2,7 +2,7 @@ # SPDX-License-Identifier: Apache-2.0 [package] name = "raincloud-reader" -version = "0.3.0" +version = "0.3.1" edition = "2021" rust-version = "1.95" publish = false diff --git a/raincloud/__init__.py b/raincloud/__init__.py index 1309f3d..fd32bfc 100644 --- a/raincloud/__init__.py +++ b/raincloud/__init__.py @@ -12,8 +12,9 @@ from ._catalog import load_catalog from ._catalog import unverified as _catalog_unverified from ._duckdb import duckdb_connect -from ._formats import select_format -from ._readers import reader_capabilities, require_reader +from ._formats import auto_formats, select_format +from ._readers import open_batches, open_dataset, reader_capabilities, require_reader +from ._registry import FORMATS from .config import Config, get_config, resolve_config from .exceptions import ( # noqa: F401 ArtifactNotFound, @@ -39,7 +40,7 @@ # and `clients/java/build.gradle.kts` reads it directly. `clients/rust/Cargo.toml` # and `CITATION.cff` are hand-bumped copies: bump them with this literal, and # tests/test_loader_package.py::test_version_mirrors_agree fails if they disagree. -__version__ = "0.3.0" +__version__ = "0.3.1" _DEFAULT_FORMAT = "auto" # Spellings people type for a format, suggested (never silently substituted). @@ -222,32 +223,9 @@ def _batches(self, *, batch_size: int = 65536, columns: list[str] | None = None) import pyarrow as pa with ExitStack() as stack: def open_reader(path): - if fmt == "parquet": - import pyarrow.parquet as pq - source = stack.enter_context(pa.OSFile(str(path), "r")) - reader = stack.enter_context(pq.ParquetFile(source)) - schema = reader.schema_arrow - if columns is not None: - _check_columns(schema, columns, fmt, self.slug) - schema = reader.read_row_groups([], columns=columns).schema - native = reader.iter_batches(batch_size=batch_size, columns=columns) - elif fmt == "arrow": - source = stack.enter_context(pa.memory_map(str(path), "r")) - reader = pa.ipc.open_file(source) - schema = reader.schema - if columns is not None: - _check_columns(schema, columns, fmt, self.slug) - schema = pa.schema([schema.field(c) for c in columns], metadata=schema.metadata) - native = (reader.get_batch(i) if columns is None else reader.get_batch(i).select(columns) - for i in range(reader.num_record_batches)) - else: - import vortex - file = vortex.open(str(path)) - if columns is not None: - _check_columns(file.dtype.to_arrow_schema(), columns, fmt, self.slug) - # Projection is pushed into the scan: unrequested columns are never read. - reader = stack.enter_context(file.to_arrow(projection=columns, batch_size=batch_size)) - schema, native = reader.schema, reader + schema, native = open_batches( + fmt, path, stack, columns=columns, batch_size=batch_size, + check_columns=lambda found: _check_columns(found, columns, fmt, self.slug)) def chunks(): with _decoding(path, fmt): @@ -341,23 +319,7 @@ def dataset(self): """ fmt = self.format require_reader(fmt) - import pyarrow as pa - import pyarrow.dataset as pads - - def open_dataset(path): - if fmt == "vortex": - import vortex - return vortex.open(str(path)).to_dataset() - if fmt == "parquet": - file_format, source = pads.ParquetFileFormat(), pa.OSFile(str(path), "r") - else: - file_format, source = pads.IpcFileFormat(), pa.memory_map(str(path), "r") - # A fragment over the open file, not the path: the dataset keeps - # reading this generation for as long as it lives. - fragment = file_format.make_fragment(source) - return pads.FileSystemDataset([fragment], fragment.physical_schema, file_format) - - return self._acquire(self.format, open_dataset) + return self._acquire(fmt, lambda path: open_dataset(fmt, path)) def to_pandas(self): try: @@ -378,6 +340,8 @@ def _choose_format(entry, requested: str, readable: bool, readers: set[str] | No path to native readers that bring their own) every recorded format counts. `readers`, when given, is the set "auto" chooses among instead. + "auto" tries the install's `auto_formats`: the formats it builds (only + Vortex by default), in vortex, parquet order, then the canonical Arrow. "auto" also skips a format a build measured unavailable at this recipe (`_resolve.measured_unavailable`); asking for one outright raises FormatUnavailable quoting that measurement, unless `build` allows a new @@ -414,10 +378,13 @@ def _choose_format(entry, requested: str, readable: bool, readers: set[str] | No raise MissingDependency(f"{entry.slug} is prepared only as {', '.join(sorted(formats))}: {exc}") from None formats = usable try: - fmt = select_format(formats, fmt) + fmt = select_format(formats, fmt, auto_formats(config, entry.version)) except FormatUnavailable as exc: raise FormatUnavailable(f"{entry.slug}: {exc}") from None - if readable: + # A format raincloud reads in-process must be readable here; one it only + # serves by path (`_registry.FORMATS` declares no reader) loads for its + # `path()`, and its readers raise MissingDependency saying so. + if readable and FORMATS[fmt]["reader"] is not None: require_reader(fmt, import_native=False) return fmt diff --git a/raincloud/_bundle.py b/raincloud/_bundle.py index fb92d1f..46e0a8a 100644 --- a/raincloud/_bundle.py +++ b/raincloud/_bundle.py @@ -68,6 +68,15 @@ def recipe_hash(spec: dict, version: int, *, specs) -> str: # still carries it, so artifacts built against it stop matching their pins. # (`hydrate` is the pre-0.3.0 spelling of a hydration block on the parent.) recipe = {key: spec[key] for key in _RECIPE_KEYS if key in spec} + # Which formats a dataset offers decides no file's bytes (since 0.3.1 an install + # chooses what it builds), so `export.formats` stays out, and an `export` left + # empty without it fingerprints like none at all. + if isinstance(recipe.get("export"), dict) and "formats" in recipe["export"]: + export = {k: v for k, v in recipe["export"].items() if k != "formats"} + if export: + recipe["export"] = export + else: + del recipe["export"] parent = (spec.get("derive") or {}).get("from") if parent and specs and parent in specs and parent != spec.get("slug"): recipe["from_recipe"] = recipe_hash(specs[parent], version, specs=specs) diff --git a/raincloud/_cache.py b/raincloud/_cache.py index ba4f2bb..ea39cc6 100644 --- a/raincloud/_cache.py +++ b/raincloud/_cache.py @@ -18,14 +18,14 @@ import uuid from pathlib import Path +from ._registry import FORMATS from .exceptions import ChecksumMismatch -# Format -> on-disk file extension. Identity for parquet/vortex, but kept as a -# map precisely so a format whose extension differs from its name slots in -# without touching call sites. `arrow` -> `arrow.zstd` is the live example: an -# Arrow IPC file whose buffers are zstd-compressed inside the IPC format (there -# is no outer zstd frame). The suffix is part of the native-client path contract. -EXT = {"parquet": "parquet", "vortex": "vortex", "arrow": "arrow.zstd"} +# Format -> on-disk file extension, from `_registry.FORMATS`. `arrow` -> +# `arrow.zstd` is an Arrow IPC file whose buffers are zstd-compressed inside the +# IPC format (there is no outer zstd frame). The suffix is part of the +# native-client path contract. +EXT = {fmt: info["ext"] for fmt, info in FORMATS.items()} # A rollback copy another process left this long ago is an orphan of a crash # (SIGKILL, OOM) mid-publish; nothing else would ever remove it. diff --git a/raincloud/_formats.py b/raincloud/_formats.py index 96a0ba6..12637a8 100644 --- a/raincloud/_formats.py +++ b/raincloud/_formats.py @@ -3,11 +3,7 @@ """Lightweight manifest export policy, shared by the loader and build pipeline.""" from __future__ import annotations -from ._registry import exporter_cells - -# Formats a dataset exports unless its spec says otherwise. Canonical Arrow is -# not among them: it is what every exporter reads. -DEFAULT_FORMATS = ("parquet", "vortex") +from ._registry import FORMATS, exporter_cells # Which writer produces a format, when neither the spec, the catalog nor the # machine names one. Python first: it is always installed, so by default a @@ -23,7 +19,10 @@ # RAINCLOUD_EXPORT_PRIORITY may name a writer this release does not ship. A # manifest is stricter: validate_manifest rejects a name with no export cell, # since there a typo would silently fall through to the next writer. -DEFAULT_EXPORT_PRIORITY = ("py", "rs", "java", "canonical") +# +# `cpp` is Nimble's only writer (upstream's C++, `nimble@cpp`): every format has +# a writer this order names, so a dataset needs no priority to export it. +DEFAULT_EXPORT_PRIORITY = ("py", "rs", "java", "cpp", "canonical") def priority_shape_error(value, where: str) -> str | None: @@ -112,21 +111,63 @@ def resolve_export_cell(fmt: str, priority, *, is_available) -> str | None: def export_formats(spec: dict, version: int = 2) -> list[str]: - """The formats `spec` exports, each written once, to `/`. - - In schema_version 2 `export.formats` is the only declaration: omitted, a - dataset exports DEFAULT_FORMATS. `convert.vortex` is the schema_version 1 - opt-in. The schema and validate_manifest reject it in a v2 manifest, but - v2 catalogs released before that rule still carry it, and they must keep - reading: there a `convert.vortex: false` without `export.formats` still - means no Vortex. + """The formats `spec` offers, each written once, to `/`. + + In schema_version 2 a dataset offers every exported format unless + `export.formats` narrows the list. Which of them a build actually writes is + the install's choice (`build_formats`). `convert.vortex` is the + schema_version 1 opt-in. The schema and validate_manifest reject it in a v2 + manifest, but v2 catalogs released before that rule still carry it, and + they must keep reading: there a `convert.vortex: false` without + `export.formats` still means no Vortex. """ if version < 2: return ["parquet", "vortex"] if (spec.get("convert") or {}).get("vortex") else ["parquet"] requested = (spec.get("export") or {}).get("formats") if requested is None and (spec.get("convert") or {}).get("vortex") is False: return ["parquet"] - return list(requested) if requested is not None else list(DEFAULT_FORMATS) + return list(requested) if requested is not None else list(EXPORTED_FORMATS) + + +def wanted_formats(config) -> tuple[str, ...]: + """The exported formats this install builds when none is asked for: its + `formats` setting, with `all` expanded.""" + names = config.formats + return EXPORTED_FORMATS if "all" in names else tuple(f for f in EXPORTED_FORMATS if f in names) + + +def build_formats(spec: dict, version: int, config, requested=None) -> list[str]: + """The exported formats a build of `spec` writes: `requested` when given + (`raincloud build --format`, or the format a load asked for), else the + install's `wanted_formats`, in either case only those the dataset offers. + + A requested format the dataset does not offer raises ValueError. `arrow` + may be requested: it is written by every build, so it adds no export. + """ + offered = export_formats(spec, version) + if requested is None and version < 2: + # A v1 catalog predates install formats: it builds what its recipe lists, as in 0.3.0. + return offered + if requested is None: + wanted = wanted_formats(config) + return [fmt for fmt in offered if fmt in wanted] + requested = [base_format(fmt) for fmt in requested] + missing = [fmt for fmt in requested if fmt != "arrow" and fmt not in offered] + if missing: + raise ValueError(f"{spec['slug']} does not offer {', '.join(missing)} " + f"(it offers {', '.join(offered) or 'only its canonical Arrow'})") + return [fmt for fmt in offered if fmt in requested] + + +def auto_formats(config, version: int = 2) -> tuple[str, ...]: + """What "auto" tries, in order, for this install: the AUTO_FORMATS it + builds, then the canonical Arrow, which every dataset has -- the file a + caller gets when none of those can be made (a writer measured unable). + A v1 catalog predates install formats and keeps 0.3.0's order.""" + if version < 2: + return AUTO_FORMATS + wanted = wanted_formats(config) + return tuple(fmt for fmt in AUTO_FORMATS if fmt in wanted or fmt == "arrow") def export_cells(spec: dict, manifest: dict | None = None) -> list[str]: @@ -206,6 +247,9 @@ def _writers() -> dict[str, tuple[str, ...]]: writers: dict[str, tuple[str, ...]] = {} for cell in exporter_cells(): base, _, writer = cell.partition("@") + if base not in FORMATS or base == "arrow": + raise RuntimeError(f"exporter cell {cell!r} writes {base!r}, which _registry.FORMATS " + f"does not declare as an exported format") writers[base] = (*writers.get(base, ()), writer) return {**writers, "arrow": ("canonical",)} @@ -215,6 +259,8 @@ def _writers() -> dict[str, tuple[str, ...]]: # artifact format: those plus the canonical Arrow they are all written from. EXPORTED_FORMATS = tuple(fmt for fmt in WRITERS if fmt != "arrow") ALL_FORMATS = (*EXPORTED_FORMATS, "arrow") +# What "auto" tries, in order: the formats a caller need not name. +AUTO_FORMATS = tuple(fmt for fmt, info in FORMATS.items() if info["auto"] and fmt in WRITERS) def base_format(fmt: str) -> str: @@ -229,17 +275,18 @@ def split_cell(cell: str) -> tuple[str, str]: return base, writer -def select_format(formats, requested: str = "auto") -> str: - """The format to open: `requested`, or the first present of vortex, parquet, - arrow for "auto". Each format is one file; which writer made it is recorded - in the catalog, not chosen here.""" +def select_format(formats, requested: str = "auto", order: tuple[str, ...] = AUTO_FORMATS) -> str: + """The format to open: `requested`, or for "auto" the first present of + `order` (by default AUTO_FORMATS: vortex, parquet, arrow; the loader passes + the install's `auto_formats`). Each format is one file; which writer made it + is recorded in the catalog, not chosen here.""" from .exceptions import FormatUnavailable if "@" in requested: raise FormatUnavailable( f"format {requested!r}: a dataset has one file per format; ask for " f"{base_format(requested)!r} (`raincloud describe` shows which writer made it)" ) - bases = ("vortex", "parquet", "arrow") if requested == "auto" else (requested,) + bases = order if requested == "auto" else (requested,) for base in bases: if base in formats: return base diff --git a/raincloud/_readers.py b/raincloud/_readers.py index b06ddb9..4784b47 100644 --- a/raincloud/_readers.py +++ b/raincloud/_readers.py @@ -1,27 +1,130 @@ # SPDX-FileCopyrightText: 2026 Raincloud Maintainers # SPDX-License-Identifier: Apache-2.0 -"""Reader availability without importing optional native extensions.""" +"""In-process format readers, and their availability without importing optional native extensions. + +Which formats have a reader is declared in `_registry.FORMATS`; how each one +opens a file is here. A format with no in-process reader is still served by +path (`Dataset.path()`), for a reader the caller brings. +""" from importlib.util import find_spec +from ._registry import FORMATS from .exceptions import MissingDependency def reader_capabilities() -> dict: """Which format readers this install has, per format; imports and fetches nothing.""" - return { - "arrow": {"available": True, "implementation": "pyarrow"}, - "parquet": {"available": True, "implementation": "pyarrow"}, - "vortex": {"available": find_spec("vortex") is not None, "implementation": "vortex-python", "extra": "vortex"}, - } + capabilities = {} + for fmt, info in FORMATS.items(): + module = info["reader"] + entry = {"available": module is not None and find_spec(module) is not None, + "implementation": info.get("implementation")} + if info.get("extra"): + entry["extra"] = info["extra"] + capabilities[fmt] = entry + return capabilities + + +def _missing(fmt: str) -> str: + info = FORMATS[fmt] + if info["reader"] is None: + return (f"raincloud has no in-process {fmt} reader; `Dataset.path()` gives the file " + f"for a reader of your own") + if fmt == "vortex": + return "Vortex reads require raincloud[vortex] and a supported native wheel for this platform" + extra = f"raincloud[{info['extra']}]" if info.get("extra") else info["reader"] + return f"{fmt} reads require {extra}" def require_reader(fmt: str, *, import_native: bool = True): + if not reader_capabilities()[fmt]["available"]: + raise MissingDependency(_missing(fmt)) + if not import_native: + return + try: + __import__(FORMATS[fmt]["reader"]) + except (ImportError, OSError) as exc: + raise MissingDependency(_missing(fmt)) from exc + + +def open_batches(fmt: str, path, stack, *, columns, batch_size: int, check_columns): + """(schema, iterator of record batches) over `path`, closed by `stack`. + + `check_columns(schema)` refuses a requested column the file lacks before + any batch is read; the iterator yields only `columns` when they are given. + """ + opener = _BATCHES.get(fmt) + if opener is None: + raise MissingDependency(_missing(fmt)) + return opener(path, stack, columns=columns, batch_size=batch_size, check_columns=check_columns) + + +def open_dataset(fmt: str, path): + """A lazy pyarrow Dataset over the open file at `path`.""" + import pyarrow as pa + import pyarrow.dataset as pads if fmt == "vortex": - if not reader_capabilities()["vortex"]["available"]: - raise MissingDependency("Vortex reads require raincloud[vortex] and a supported native wheel for this platform") - if not import_native: - return - try: - import vortex # noqa: F401 - except (ImportError, OSError) as exc: - raise MissingDependency("Vortex reads require raincloud[vortex] and a supported native wheel for this platform") from exc + import vortex + return vortex.open(str(path)).to_dataset() + if fmt == "parquet": + file_format, source = pads.ParquetFileFormat(), pa.OSFile(str(path), "r") + elif fmt == "arrow": + file_format, source = pads.IpcFileFormat(), pa.memory_map(str(path), "r") + elif fmt == "orc": + file_format, source = pads.OrcFileFormat(), pa.OSFile(str(path), "r") + else: + raise MissingDependency(_missing(fmt)) + # A fragment over the open file, not the path: the dataset keeps reading + # this generation for as long as it lives. + fragment = file_format.make_fragment(source) + return pads.FileSystemDataset([fragment], fragment.physical_schema, file_format) + + +def _parquet_batches(path, stack, *, columns, batch_size, check_columns): + import pyarrow as pa + import pyarrow.parquet as pq + source = stack.enter_context(pa.OSFile(str(path), "r")) + reader = stack.enter_context(pq.ParquetFile(source)) + schema = reader.schema_arrow + if columns is not None: + check_columns(schema) + schema = reader.read_row_groups([], columns=columns).schema + return schema, reader.iter_batches(batch_size=batch_size, columns=columns) + + +def _arrow_batches(path, stack, *, columns, batch_size, check_columns): + import pyarrow as pa + source = stack.enter_context(pa.memory_map(str(path), "r")) + reader = pa.ipc.open_file(source) + schema = reader.schema + if columns is not None: + check_columns(schema) + schema = pa.schema([schema.field(c) for c in columns], metadata=schema.metadata) + return schema, (reader.get_batch(i) if columns is None else reader.get_batch(i).select(columns) + for i in range(reader.num_record_batches)) + + +def _vortex_batches(path, stack, *, columns, batch_size, check_columns): + import vortex + file = vortex.open(str(path)) + if columns is not None: + check_columns(file.dtype.to_arrow_schema()) + # Projection is pushed into the scan: unrequested columns are never read. + reader = stack.enter_context(file.to_arrow(projection=columns, batch_size=batch_size)) + return reader.schema, reader + + +def _orc_batches(path, stack, *, columns, batch_size, check_columns): + import pyarrow as pa + import pyarrow.orc as orc + file = orc.ORCFile(stack.enter_context(pa.OSFile(str(path), "r"))) + schema = file.schema + if columns is not None: + check_columns(schema) + schema = pa.schema([schema.field(c) for c in columns], metadata=schema.metadata) + # One stripe at a time; `Dataset.batches` slices each to `batch_size`. + return schema, (file.read_stripe(i, columns=columns) for i in range(file.nstripes)) + + +_BATCHES = {"parquet": _parquet_batches, "arrow": _arrow_batches, "vortex": _vortex_batches, + "orc": _orc_batches} diff --git a/raincloud/_registry.py b/raincloud/_registry.py index 745deb6..00e2843 100644 --- a/raincloud/_registry.py +++ b/raincloud/_registry.py @@ -54,12 +54,38 @@ "tpcgen-rs-tpch": "tpch:RustTPCH", } +# Artifact formats. Each is one file per dataset, `/.`, +# whichever writer made it; `arrow` is the canonical every exporter reads. +# ext the file extension; part of the native-client path contract. +# auto whether `load(slug)` with no format may pick it. Those that +# may are tried in declaration order; any other format opens +# only when asked for by name. +# reader the module whose presence means this install reads the +# format in-process, or None when raincloud only serves the +# file's path to a reader the caller brings. +# implementation that reader, as `raincloud describe --readers` names it. +# extra the raincloud extra that installs the reader, if any. +# Every exporter cell's format must be declared here; `_formats` checks at import. +FORMATS: dict[str, dict] = { + "vortex": {"ext": "vortex", "auto": True, "reader": "vortex", + "implementation": "vortex-python", "extra": "vortex"}, + "parquet": {"ext": "parquet", "auto": True, "reader": "pyarrow", "implementation": "pyarrow"}, + "arrow": {"ext": "arrow.zstd", "auto": True, "reader": "pyarrow", "implementation": "pyarrow"}, + # pyarrow's ORC support is a compiled extension some builds leave out. + "orc": {"ext": "orc", "auto": False, "reader": "pyarrow._orc", "implementation": "pyarrow"}, + # pyarrow reads no Avro: the loader serves the file's path. + "avro": {"ext": "avro", "auto": False, "reader": None, "implementation": None}, + # Nimble is C++ only: served by path. + "nimble": {"ext": "nimble", "auto": False, "reader": None, "implementation": None}, +} + # In-process exporter cells, as ":" under `raincloud.pipeline.export`. # The built-in priority's first choice; the class carries its own `cell_id`, and # the key here must match it. PY_EXPORTERS: dict[str, str] = { "parquet@py": "exporters:ParquetExporter", "vortex@py": "exporters:VortexExporter", + "orc@py": "exporters:OrcExporter", } # Opt-in exporter cells that shell out to a PATH-discovered reference writer, @@ -73,6 +99,10 @@ "parquet@hardwood": ("parquet", "raincloud-export-parquet-hardwood"), "vortex@rs": ("vortex", "raincloud-export-vortex-rs"), "vortex@jni": ("vortex", "raincloud-export-vortex-jni"), + "orc@rs": ("orc", "raincloud-export-orc-rs"), + "avro@rs": ("avro", "raincloud-export-avro-rs"), + "avro@java": ("avro", "raincloud-export-avro-java"), + "nimble@cpp": ("nimble", "raincloud-export-nimble-cpp"), } # Custom fetchers, as ":" under `raincloud.pipeline`. A recipe diff --git a/raincloud/_resolve.py b/raincloud/_resolve.py index d0df866..e6d77cc 100644 --- a/raincloud/_resolve.py +++ b/raincloud/_resolve.py @@ -372,7 +372,14 @@ def resolve( pinned = stack.enter_context(context.pinned(config)) if context else config try: # The build log goes to stderr: stdout is the caller's (a path, or JSON). - subprocess.run([sys.executable, "-m", "raincloud.pipeline.build", slug], check=True, + # Only the format asked for: the install's other formats are + # its own choice to build, not this load's. A v1 catalog predates + # install formats and builds what its recipe lists, as in 0.3.0. + command = [sys.executable, "-m", "raincloud.pipeline.build", slug] + if entry.version >= 2: + command += ["--format", fmt] + subprocess.run(command, + check=True, env=pinned.subprocess_env(), stdout=_stderr_fd()) except (subprocess.CalledProcessError, OSError) as e: # Honour the typed-error contract — callers catch RaincloudError, diff --git a/raincloud/cli.py b/raincloud/cli.py index f92e6c6..4880724 100644 --- a/raincloud/cli.py +++ b/raincloud/cli.py @@ -17,6 +17,8 @@ from urllib.parse import urlsplit from . import __version__, _open, describe +from ._cache import EXT +from ._formats import AUTO_FORMATS from ._locking import atomic_write, creation_mode from .config import config_path, resolve_config, use_config from .exceptions import BuildToolingMissing, HydratedDatasetWarning, RaincloudError @@ -34,6 +36,9 @@ # cache_dir = "~/.cache/raincloud" # mirror downloads (default: data_dir) # scratch_dir = "/tmp/raincloud" # build scratch space # mirror = "s3://bucket/prefix" # a store to fetch prepared files from +# formats = ["vortex", "parquet"] # formats a build writes (default: vortex; "all") +# keep_raw = true # keep raw downloads after a build +# keep_canonical = true # keep the canonical Arrow after a build """ @@ -120,7 +125,7 @@ def _overview(settings) -> str: raincloud list [WORD...] find datasets, e.g. `raincloud list tpch lineitem` raincloud describe SLUG what a dataset is: rows, columns, formats, license - raincloud load SLUG print the path to its file (--format vortex|parquet|arrow) + raincloud load SLUG print the path to its file (--format {'|'.join(EXT)}) raincloud build SLUG prepare a dataset on this machine (needs raincloud[build]) raincloud browse browse interactively (needs raincloud[tui]) @@ -219,7 +224,8 @@ def command(name, help, aliases=(), **kw): def resolution(sub): sub.add_argument("slug", nargs="?", help="dataset name; `raincloud list` finds them") sub.add_argument("-f", "--format", default="auto", - help="vortex, parquet or arrow; auto (the default) picks the first prepared, in that order") + help=f"{', '.join(EXT)}; auto (the default) picks the first prepared of " + f"{', '.join(AUTO_FORMATS)}, in that order") sub.add_argument("--readers", metavar="FORMATS", help="comma-separated formats the caller decodes itself (native readers); " "auto chooses among these") @@ -377,7 +383,6 @@ def _catalog_text(action: str, result, settings) -> str: def _readers(value: str | None) -> set[str] | None: if value is None: return None - from ._cache import EXT from ._suggest import hint names = {part.strip().lower() for part in value.split(",") if part.strip()} for name in names - set(EXT): diff --git a/raincloud/config.py b/raincloud/config.py index e481a2a..c8a7800 100644 --- a/raincloud/config.py +++ b/raincloud/config.py @@ -21,9 +21,13 @@ _PATHS = {"data_dir", "scratch_dir", "raw_dir", "cache_dir", "manifest", "snapshot", "catalog_dir"} # retry_errors: a build re-attempts a format whose writer, with this toolchain, # already failed to write it at the recipe (`export.run_exporters`). -_BOOLS = {"offline", "retry_errors"} -# Comma-separated writer names, most preferred first (e.g. "rs,java,py"). -_LISTS = {"export_priority"} +# keep_raw / keep_canonical: a successful build keeps the dataset's raw +# download / its canonical Arrow instead of removing it (`build._clean`). +_BOOLS = {"offline", "retry_errors", "keep_raw", "keep_canonical"} +# Comma-separated names: export_priority names writers, most preferred first +# (e.g. "rs,java,py"); formats names the formats a build writes (e.g. +# "vortex,parquet", or "all"). +_LISTS = {"export_priority", "formats"} _KEYS = _PATHS | _BOOLS | _LISTS | {"mirror", "catalog", "catalog_url"} # Named catalog selectors; anything else is a revision id (or its prefix) or a directory. _SELECTORS = {"auto", "active", "checkout", "bundled", "local"} @@ -31,13 +35,31 @@ REVISION_PREFIX = re.compile(r"[0-9a-f]{4,63}") -def _writers(value: str | list | tuple) -> tuple[str, ...]: - """Parse "rs,java,py" (or a list of names) into a writer tuple, ignoring blanks.""" +def _names(value: str | list | tuple, key: str = "export_priority") -> tuple[str, ...]: + """Parse "rs,java,py" (or a list of names) into a tuple, ignoring blanks.""" if isinstance(value, str): value = value.split(",") elif not isinstance(value, (list, tuple)) or not all(isinstance(v, str) for v in value): - raise ValueError(f"writer priority must be a list of writer names, not {value!r}") - return tuple(part.strip() for part in value if part.strip()) + raise ValueError(f"{key} must be a list of {_NOUNS[key]}, not {value!r}") + names = tuple(part.strip() for part in value if part.strip()) + if key == "formats": + _check_formats(names) + return names + + +_NOUNS = {"export_priority": "writer names", "formats": "format names"} + + +def _check_formats(names: tuple[str, ...]) -> None: + """Refuse a `formats` entry that names no exported format, with a did-you-mean.""" + from ._formats import EXPORTED_FORMATS + from ._suggest import hint + for name in names: + if name == "arrow": + raise ValueError("formats: the canonical Arrow is not an exported format; " + "set keep_canonical = true to keep it") + if name != "all" and name not in EXPORTED_FORMATS: + raise ValueError("formats: " + hint(name, [*EXPORTED_FORMATS, "all"], noun="format")) def redact_url(value: str) -> str: @@ -57,6 +79,8 @@ def redact_url(value: str) -> str: "manifest": "RAINCLOUD_MANIFEST", "snapshot": "RAINCLOUD_SNAPSHOT", "mirror": "RAINCLOUD_MIRROR", "offline": "RAINCLOUD_OFFLINE", "export_priority": "RAINCLOUD_EXPORT_PRIORITY", "retry_errors": "RAINCLOUD_RETRY_ERRORS", + "formats": "RAINCLOUD_FORMATS", "keep_raw": "RAINCLOUD_KEEP_RAW", + "keep_canonical": "RAINCLOUD_KEEP_CANONICAL", } @@ -105,6 +129,14 @@ class Config: retry_errors: bool = False # Machine-level writer preference; a catalog or a single spec can override it. export_priority: tuple[str, ...] | None = None + # The formats a build writes when none is asked for (`all`: every format a + # dataset offers). Only Vortex by default: every other format is opt-in. + formats: tuple[str, ...] = ("vortex",) + # Whether a successful build keeps what it built from. Off by default: the + # dataset's file is what most installs want, and either can be made again + # (the raw by fetching, the canonical by building). + keep_raw: bool = False + keep_canonical: bool = False manifest: Path | None = None snapshot: Path | None = None file: Path | None = None @@ -232,12 +264,10 @@ def resolve_config(*, config: str | Path | None = None, no_config: bool = False, elif key in _LISTS: # A TOML array is the natural spelling here; a comma string is # accepted so the file and the env var take the same value. - if isinstance(value, str): - value = layer[key] = _writers(value) - elif isinstance(value, list) and all(isinstance(v, str) for v in value): - value = layer[key] = tuple(value) - else: - raise ValueError(f"{settings_file}: {key} must be a list of writer names") + try: + value = layer[key] = _names(value, key) + except ValueError as exc: + raise ValueError(f"{settings_file}: {exc}") from None elif not isinstance(value, str) or (key in _PATHS and not value): raise ValueError(f"{settings_file}: {key} must be a nonempty path/string") if key in _PATHS: @@ -265,7 +295,7 @@ def resolve_config(*, config: str | Path | None = None, no_config: bool = False, if key in _BOOLS: values[key] = _flag(value) elif key in _LISTS: - values[key] = _writers(value) + values[key] = _names(value, key) else: values[key] = _path(value, cwd) if key in _PATHS else value origins[key] = name @@ -274,7 +304,7 @@ def resolve_config(*, config: str | Path | None = None, no_config: bool = False, if key in _BOOLS and not isinstance(value, bool): raise ValueError(f"{key} must be a boolean") if key in _LISTS: - value = _writers(value) + value = _names(value, key) values[key] = _path(value, cwd) if key in _PATHS else value origins[key] = "explicit" values.setdefault("catalog_dir", native_data / "catalogs") diff --git a/raincloud/pipeline/build.py b/raincloud/pipeline/build.py index 5126178..839de94 100644 --- a/raincloud/pipeline/build.py +++ b/raincloud/pipeline/build.py @@ -22,6 +22,21 @@ build exits 0; the loader then reports the format as unavailable, quoting the measurement, and the catalog shows it once a maintainer regenerates it. +Which formats a build writes is the install's choice: its `formats` setting +(only Vortex by default; `all` for every format the dataset offers), or the +formats named with `--format`, which is how a load that needs one other file +builds just that one. The canonical Arrow is written by every build; once the +formats are written, a successful build removes it unless `keep_canonical` is +set, `--format arrow` asked for it, or no format was written (it is then the +dataset's only file: the one `load` serves when, say, the Vortex writer cannot +make Vortex). It removes the raw download too, unless `keep_raw` is set. + +A generated dataset is one table of a group its generator writes at once (every +TPC-H table of one scale factor, say), so building one builds the whole group: +every table, in the same formats. Once a group has built, its generator output +is removed too, unless `keep_raw` is set. `--only` builds just the tables named, +and the generator output still goes: the rest would be generated again. + A failure already measured is not repeated: when the measurement that applies (this install's build record at the recipe, else the catalog's) records the writer that would run now, with the same toolchain, reading the same canonical, @@ -43,15 +58,19 @@ python -m raincloud.pipeline.build clickbench-hits python -m raincloud.pipeline.build uci-iris uci-wine-quality python -m raincloud.pipeline.build --all --strict # CI mode + python -m raincloud.pipeline.build tpch-sf1-lineitem --format parquet --format orc """ from __future__ import annotations import argparse +import contextlib import shutil import sys import traceback from raincloud._cache import sha256_file +from raincloud._formats import build_formats +from raincloud.config import get_config from raincloud.exceptions import BuildToolingMissing from .canonical import write_canonical @@ -67,6 +86,8 @@ display_path, load_manifest, prepared_arrow, + raw_downloads_root, + raw_slug_dir, recipe_scratch, workdir_root, ) @@ -76,7 +97,8 @@ def _run_one(spec: dict, outputs, *, strict: bool, clean_workdir: bool = False, unavailable: list | None = None, skipped: list | None = None, - unverified: list | None = None, retry_errors: bool = False) -> bool: + unverified: list | None = None, retry_errors: bool = False, + formats: list[str] | None = None) -> bool: print("\n" + "=" * 72) print(f" {spec['slug']}") print("=" * 72) @@ -85,9 +107,22 @@ def _run_one(spec: dict, outputs, *, strict: bool, clean_workdir: bool = False, context = current() context.build_check(spec) outputs.preflight(spec["slug"]) + config = get_config() + try: + exports = build_formats(spec, context.manifest["schema_version"], config, formats) + except ValueError as exc: + print(f" FAILED: {exc}") + return False + if formats is None and "all" in config.formats: + # `all` is every format this machine can write: one with no + # installed writer is left out, said, rather than failing the build. + # A format named outright still fails below. + exports = [fmt for fmt in exports if _installed(spec, fmt)] + print(f" formats: {', '.join(exports) or 'none'} (canonical Arrow " + f"{'kept' if _keeps_canonical(config, formats) else 'removed once they are written'})") # Fail in a second, not after the transform, when a format this - # dataset exports has no installed writer here. - plan(spec) + # build writes has no installed writer here. + plan(spec, exports) if spec.get("derive"): # A derived dataset's input is another dataset, not an upstream. # Its options arrive through a context variable: `hydrate.main` @@ -128,6 +163,7 @@ def _run_one(spec: dict, outputs, *, strict: bool, clean_workdir: bool = False, for canonical in canonicals: record_build({slug_from_canonical(canonical): { "arrow": (sha256_file(canonical), canonical.stat().st_size)}}) + written = {} for canonical in canonicals: slug = slug_from_canonical(canonical) def note(failure, recorded, slug=slug): @@ -142,8 +178,11 @@ def unchecked(result, why, slug=slug): unverified.append((slug, result.format_id, why)) # A planned writer's failure is recorded and the build goes on; one # already measured with this writer and toolchain is skipped. - run_exporters(spec, canonical, retry_errors=retry_errors, on_skip=skip, - **recorders(slug, note, on_unverified=unchecked)) + written[canonical] = run_exporters(spec, canonical, exports, retry_errors=retry_errors, + on_skip=skip, **recorders(slug, note, on_unverified=unchecked)) + if context.manifest["schema_version"] >= 2: + # A v1 catalog builds as in 0.3.0, keeping what it was made from. + _clean(spec, written, config, formats) if clean_workdir: wd = workdir_root() / spec["slug"] if wd.exists(): @@ -177,6 +216,126 @@ def unchecked(result, why, slug=slug): return False +def _installed(spec: dict, fmt: str) -> bool: + try: + plan(spec, [fmt]) + except BuildToolingMissing as exc: + print(f" [skip] {fmt}: {exc}") + return False + return True + + +def _keeps_canonical(config, formats) -> bool: + return config.keep_canonical or "arrow" in (formats or ()) + + +def _clean(spec: dict, written: dict, config, formats) -> None: + """Remove what a successful build was made from, unless the install keeps it. + + `written` maps each canonical to the exports written from it. A canonical + from which nothing was written is the dataset's only file and stays. The + raw download goes unless `keep_raw`; a generated dataset's generator + output is shared by its group and stays. + """ + if not _keeps_canonical(config, formats): + for canonical, results in written.items(): + if not results: + print(f" [keep] {display_path(canonical)}: no other format was written, so it is " + f"the dataset's file") + elif canonical.is_file(): + canonical.unlink() + with contextlib.suppress(OSError): + canonical.parent.rmdir() # `arrow/`, when nothing else is in it + print(f" [clean] removed {display_path(canonical)} (set keep_canonical to keep it)") + fetch_type = (spec.get("fetch") or {}).get("type") + if config.keep_raw or spec.get("derive") or fetch_type == "generated": + return + raw = raw_slug_dir(spec["slug"], spec) + if raw.exists(): + try: + shutil.rmtree(raw) + except OSError as e: + print(f" [clean] could not remove {display_path(raw)}: {e}", file=sys.stderr) + return + print(f" [clean] removed {display_path(raw)} (set keep_raw to keep it)") + # A recipe generation lives under `/.recipes/`; drop what it leaves empty. + root = raw_downloads_root() + for parent in raw.parents: + if parent == root or root not in parent.parents: + break + try: + parent.rmdir() + except OSError: + break + + +def _generation(spec: dict) -> str | None: + """The key of the generator group `spec` is a table of, or None.""" + from raincloud._generated import generation_key + fetch = spec.get("fetch") or {} + return generation_key(fetch) if fetch.get("type") == "generated" and not spec.get("derive") else None + + +def _with_groups(selected: list[dict], manifest: dict) -> list[dict]: + """`selected` with every other table of each generator group it names, in + manifest order after the first table selected from that group.""" + chosen = {spec["slug"] for spec in selected} + members: dict[str, list[dict]] = {} + for spec in manifest["datasets"]: + key = _generation(spec) + if key is not None: + members.setdefault(key, []).append(spec) + out, seen = [], set() + for spec in selected: + if spec["slug"] in seen: + continue + out.append(spec) + seen.add(spec["slug"]) + key = _generation(spec) + added = [m for m in members.get(key, []) if m["slug"] not in chosen and m["slug"] not in seen] + if added: + print(f"[group] {spec['slug']}: building its generator group too: " + + ", ".join(m["slug"] for m in added)) + out += added + seen.update(m["slug"] for m in added) + return out + + +def _clean_generated(selected: list[dict], built: dict[str, bool]) -> None: + """Remove the generator output of each group whose tables in this run all + built, unless `keep_raw` keeps it.""" + from .generate import group_root + from .lifecycle import operation_lock + + if get_config().keep_raw: + return + groups: dict[str, list[dict]] = {} + for spec in selected: + key = _generation(spec) + if key is not None: + groups.setdefault(key, []).append(spec) + for specs in groups.values(): + if not all(built.get(spec["slug"]) for spec in specs): + continue + root = group_root(specs[0]["fetch"]) + if not root.exists(): + continue + with operation_lock(resources=True): + try: + shutil.rmtree(root) + except OSError as e: + print(f" [clean] could not remove {display_path(root)}: {e}", file=sys.stderr) + else: + print(f" [clean] removed {display_path(root)}, the generator output of " + f"{', '.join(spec['slug'] for spec in specs)} (set keep_raw to keep it)") + + +def _formats_arg(values: list[str] | None) -> list[str] | None: + if not values: + return None + return [name.strip() for value in values for name in value.split(",") if name.strip()] + + def _main(argv: list[str] | None = None) -> int: ap = argparse.ArgumentParser(prog="raincloud build", allow_abbrev=False) ap.add_argument("slugs", nargs="*", help="specific slugs to build") @@ -189,6 +348,11 @@ def _main(argv: list[str] | None = None) -> int: help="after each successful build, remove the selected recipe's scratch directory " "so large decompressed intermediates (e.g. Public BI bz2→csv) " "don't accumulate during batch runs") + ap.add_argument("--format", action="append", dest="formats", metavar="FORMAT", + help="write this format instead of the install's `formats` setting (repeatable, or " + "comma-separated); `arrow` keeps the canonical Arrow") + ap.add_argument("--only", action="store_true", + help="build only the generated tables named, not the rest of their generator's group") ap.add_argument("--retry-errors", action="store_true", help="attempt a format even when its writer, with this toolchain, already failed to " "write it at this recipe (skipped by default; see `[skip]` lines)") @@ -200,18 +364,33 @@ def _main(argv: list[str] | None = None) -> int: ap.error(str(exc)) # Every name is checked before any work: a typo is an error with a # did-you-mean, never a silently smaller build. - selected = select_or_exit(ap, load_manifest(), args.slugs, all_=args.all, verb="build") + manifest = load_manifest() + selected = select_or_exit(ap, manifest, args.slugs, all_=args.all, verb="build") + if manifest["schema_version"] >= 2 and not args.only: + selected = _with_groups(selected, manifest) + formats = _formats_arg(args.formats) + if formats is not None: + from raincloud._formats import ALL_FORMATS + from raincloud._suggest import hint + for name in formats: + if name not in ALL_FORMATS: + ap.error("--format: " + hint(name, list(ALL_FORMATS), noun="format")) ok = failed = 0 unavailable: list = [] skipped: list = [] unverified: list = [] + built: dict[str, bool] = {} for spec in selected: - if run_one(spec, strict=args.strict, clean_workdir=args.clean_workdir, unavailable=unavailable, - skipped=skipped, unverified=unverified, retry_errors=args.retry_errors): + built[spec["slug"]] = run_one(spec, strict=args.strict, clean_workdir=args.clean_workdir, + unavailable=unavailable, skipped=skipped, unverified=unverified, + retry_errors=args.retry_errors, formats=formats) + if built[spec["slug"]]: ok += 1 else: failed += 1 + if manifest["schema_version"] >= 2: + _clean_generated(selected, built) print(f"\nsummary: ok={ok} failed/skipped={failed}" + (f" unavailable={len(unavailable)}" if unavailable else "") + (f" known failures not retried={len(skipped)}" if skipped else "") @@ -231,13 +410,15 @@ def _main(argv: list[str] | None = None) -> int: def run_one(spec: dict, *, strict: bool, clean_workdir: bool = False, unavailable: list | None = None, - skipped: list | None = None, unverified: list | None = None, retry_errors: bool = False) -> bool: + skipped: list | None = None, unverified: list | None = None, retry_errors: bool = False, + formats: list[str] | None = None) -> bool: """Build one dataset; True when it built. A format its planned writer could not produce is appended to `unavailable` as (slug, export.Unavailable); one skipped as an already-measured failure, to `skipped` as (slug, export.Skipped). `retry_errors` attempts those instead. A file promoted without its writer verifying it reads back (a sidecar's `roundtrip: null`) - is appended to `unverified` as (slug, writer cell, its reason).""" + is appended to `unverified` as (slug, writer cell, its reason). `formats` + replaces the install's `formats` setting for this build (`--format`).""" from .lifecycle import operation_lock with operation_lock(resources=True): @@ -246,7 +427,7 @@ def run_one(spec: dict, *, strict: bool, clean_workdir: bool = False, unavailabl with build_outputs(spec) as outputs: return _run_one(spec, outputs, strict=strict, clean_workdir=clean_workdir, unavailable=unavailable, skipped=skipped, unverified=unverified, - retry_errors=retry_errors) + retry_errors=retry_errors, formats=formats) def main(argv: list[str] | None = None): diff --git a/raincloud/pipeline/docs.py b/raincloud/pipeline/docs.py index 577b6e9..02950ac 100644 --- a/raincloud/pipeline/docs.py +++ b/raincloud/pipeline/docs.py @@ -55,11 +55,13 @@ import pyarrow as pa import pyarrow.parquet as pq +from raincloud._formats import ALL_FORMATS + from .discovery import SHOWCASE_TIERS, _is_variant_field, bucket_for_size from .spec import ( REPO_ROOT, load_manifest, - prepared_arrow, + prepared_artifact, prepared_parquet, prepared_vortex, spec_field, @@ -717,15 +719,12 @@ def generate_snapshot(*, overwrite_missing: bool = False, rehash: bool = False, for spec in manifest["datasets"]: slug = spec["slug"] recipe = recipe_hash(spec, manifest["schema_version"], specs=specs) - parquet = prepared_parquet(slug) - vortex = prepared_vortex(slug) expected_rows = spec_field(spec, "expect.rows") prior_for_slug = existing_slugs.get(slug) prior = prior_for_slug or {} - parquet_bytes = parquet.stat().st_size if parquet.exists() else None - vortex_bytes = vortex.stat().st_size if vortex.exists() else None - arrow = prepared_arrow(slug) - arrow_bytes = arrow.stat().st_size if arrow.exists() else None + paths = {fmt: prepared_artifact(slug, fmt) for fmt in ALL_FORMATS} + sizes = {fmt: path.stat().st_size if path.exists() else None for fmt, path in paths.items()} + parquet, arrow = paths["parquet"], paths["arrow"] # Preserve each absent format independently. A local Parquet file does # not prove that an Arrow/Vortex artifact recorded elsewhere disappeared. fresh: dict = { @@ -740,9 +739,8 @@ def generate_snapshot(*, overwrite_missing: bool = False, rehash: bool = False, "expected_rows": expected_rows, } n_unavailable = 0 - for fmt, path, size in (("parquet", parquet, parquet_bytes), - ("vortex", vortex, vortex_bytes), - ("arrow", arrow, arrow_bytes)): + for fmt, path in paths.items(): + size = sizes[fmt] built = builds.get(artifact_key(slug, fmt, manifest["schema_version"])) or {} if isinstance(built.get("unavailable"), dict) and built.get("recipe") == recipe: # This install's build measured that its writer cannot make the @@ -826,8 +824,7 @@ def generate_snapshot(*, overwrite_missing: bool = False, rehash: bool = False, for k in ("size_bucket", "shape_traits"): if k in snap_fragment: fresh[k] = snap_fragment[k] - fresh_has_data = n_unavailable or any(size is not None - for size in (parquet_bytes, vortex_bytes, arrow_bytes)) + fresh_has_data = n_unavailable or any(size is not None for size in sizes.values()) if fresh_has_data or overwrite_missing or slug not in existing_slugs: out["slugs"][slug] = _without_stale_measurements(fresh, recipe) else: diff --git a/raincloud/pipeline/export/__init__.py b/raincloud/pipeline/export/__init__.py index 4a74015..e78e2f9 100644 --- a/raincloud/pipeline/export/__init__.py +++ b/raincloud/pipeline/export/__init__.py @@ -81,7 +81,8 @@ def all_exporters() -> list[Exporter]: def plan(spec: dict, formats: list[str] | None = None) -> list[str]: - """The writer cell for each format to write. + """The writer cell for each format to write: `formats`, else the install's + (`build_formats`). A bare format takes the first INSTALLED writer in its priority. The priority is the most specific one that names the format -- spec, then @@ -98,7 +99,7 @@ def plan(spec: dict, formats: list[str] | None = None) -> list[str]: def _plan(spec: dict, formats: list[str] | None) -> list[tuple[str, bool]]: """`plan`, with whether each cell was named outright rather than planned.""" - from raincloud._formats import export_formats, export_priority, resolve_export_cell + from raincloud._formats import build_formats, export_priority, resolve_export_cell from raincloud.catalogs import current from raincloud.config import get_config @@ -107,7 +108,7 @@ def _plan(spec: dict, formats: list[str] | None) -> list[tuple[str, bool]]: version = int(manifest["schema_version"]) if manifest else 2 config = get_config() cells = [] - for fmt in (formats if formats is not None else export_formats(spec, version)): + for fmt in (formats if formats is not None else build_formats(spec, version, config)): if "@" in fmt: cells.append((fmt, True)) continue @@ -193,12 +194,13 @@ def _error_text(text: str) -> str: # The Python distributions an in-process writer runs. -_DISTRIBUTIONS = {"parquet@py": ("pyarrow",), "vortex@py": ("vortex-data", "pyarrow")} +_DISTRIBUTIONS = {"parquet@py": ("pyarrow",), "vortex@py": ("vortex-data", "pyarrow"), "orc@py": ("pyarrow",)} def writer_toolchain(exporter: Exporter) -> dict[str, str]: """The versions that decide what `exporter` can write: its libraries for an - in-process writer, the binary for a sidecar (`SidecarExporter.toolchain`).""" + in-process writer, the binary for a sidecar (`SidecarExporter.toolchain`), + and the format's write settings that are set (`spec.chosen_settings`).""" import platform from importlib import metadata @@ -210,6 +212,8 @@ def writer_toolchain(exporter: Exporter) -> dict[str, str]: versions[distribution] = metadata.version(distribution) except metadata.PackageNotFoundError: versions[distribution] = "not installed" + from ..spec import chosen_settings + versions.update(chosen_settings(exporter.format_id)) return versions @@ -222,8 +226,8 @@ def run_exporters(spec: dict, canonical: Path, formats: list[str] | None = None, """Write each of `spec`'s formats from its canonical Arrow artifact. Every format is one file, `/.`, whichever writer makes it - (see `plan`). `formats` replaces the spec's list for this run; an entry may - be a bare format or a writer cell (`parquet@rs`). + (see `plan`). `formats` replaces the install's formats (`build_formats`) + for this run; an entry may be a bare format or a writer cell (`parquet@rs`). Each file is published under a `Publication`, and each writer runs under the export time limit (`bounded.run_bounded`). Every writer reads back what diff --git a/raincloud/pipeline/export/__main__.py b/raincloud/pipeline/export/__main__.py index 7036499..6fea990 100644 --- a/raincloud/pipeline/export/__main__.py +++ b/raincloud/pipeline/export/__main__.py @@ -21,7 +21,10 @@ at all (`records.check_canonical`). A cell named with `--format` whose writer is not installed fails the slug. -A planned writer (a spec's formats, or a bare `--format vortex`) that cannot +Without `--format` it writes the formats a build would: the install's +`formats` setting (only Vortex by default), among those the dataset offers. + +A planned writer (one of those formats, or a bare `--format vortex`) that cannot produce its file -- it raises, dies, reports a failed round-trip or exceeds `RAINCLOUD_EXPORT_TIMEOUT` -- is recorded as that format's "unavailable" measurement (`records.record_unavailable`), as in a build. Without `--format` @@ -48,6 +51,9 @@ import sys from pathlib import Path +from raincloud._formats import build_formats +from raincloud.config import get_config + from ..lifecycle import maintenance from ..selection import select_or_exit from ..spec import check_env_knobs, display_path, load_manifest, prepared_arrow @@ -64,7 +70,7 @@ def main(argv: list[str] | None = None) -> int: help="export every dataset except hydrated ones, which are exported by name") ap.add_argument("--format", action="append", metavar="FORMAT", dest="formats", help="export only this format (parquet), or this format by a named " - "writer (parquet@rs); repeatable. Overrides the spec's export.formats. " + "writer (parquet@rs); repeatable. Overrides the install's `formats` setting. " "Either way the file is /..") ap.add_argument("--dry-run", action="store_true", help="list what would be exported and exit") @@ -78,11 +84,11 @@ def main(argv: list[str] | None = None) -> int: ap.error(str(exc)) if args.formats: - from raincloud._formats import WRITERS + from raincloud._formats import EXPORTED_FORMATS for cell in args.formats: # fail before touching anything if "@" not in cell: - if cell not in WRITERS or cell == "arrow": - ap.error(f"no exported format {cell!r}; formats: parquet, vortex") + if cell not in EXPORTED_FORMATS: + ap.error(f"no exported format {cell!r}; formats: {', '.join(EXPORTED_FORMATS)}") continue try: get_exporter(cell) @@ -103,14 +109,17 @@ def main(argv: list[str] | None = None) -> int: try: # Refuse a stale or unknown canonical before planning anything. status = check_canonical(canonical) - cells = plan(spec, args.formats) + formats = args.formats + if formats is None: + formats = build_formats(spec, load_manifest()["schema_version"], get_config()) + cells = plan(spec, formats) if args.dry_run: print(f" would export {slug} from {display_path(canonical)}: {', '.join(cells)}") n_exported += 1 continue - # A --format request replaces the spec's own list for this run only. + # A --format request replaces the install's formats for this run only. failed, skipped = [], [] - results = export_from_canonical(spec, canonical, args.formats, status=status, + results = export_from_canonical(spec, canonical, formats, status=status, on_unavailable=lambda failure, recorded: failed.append(failure), on_skip=skipped.append, retry_errors=args.retry_errors) except (KeyboardInterrupt, SystemExit): diff --git a/raincloud/pipeline/export/compare.py b/raincloud/pipeline/export/compare.py index d2fc130..c97d382 100644 --- a/raincloud/pipeline/export/compare.py +++ b/raincloud/pipeline/export/compare.py @@ -1,6 +1,13 @@ # SPDX-FileCopyrightText: 2026 Raincloud Maintainers # SPDX-License-Identifier: Apache-2.0 -"""Logical Arrow equality with lossless normalization and recursive float fidelity.""" +"""Logical Arrow equality with lossless normalization and recursive float fidelity. + +Where equality turns on representation rather than data, the rule is shared with +the Rust and JVM comparators through `sidecars/compare_cases` (pairs of files and +the verdict each must get): a union of exactly `null` and `T` is a nullable `T`; +zoned timestamps compare by instant, whatever zone labels them; an integer and +a scale-0 decimal holding the same values are equal. +""" from __future__ import annotations import numpy as np @@ -14,11 +21,65 @@ def _is_list(dtype: pa.DataType) -> bool: or pa.types.is_large_list_view(dtype)) +def _nullable_member(dtype: pa.DataType) -> int | None: + """The index of `T` in a union of exactly `null` and `T`, else None: such a + union is how some readers spell a nullable `T` (Avro's ["null", T]).""" + if pa.types.is_union(dtype) and dtype.num_fields == 2: + nulls = [pa.types.is_null(field.type) for field in dtype] + if nulls.count(True) == 1: + return nulls.index(False) + return None + + def _logical_type(dtype: pa.DataType) -> pa.DataType: - # Extension annotations and dictionary indices are physical representations. - while isinstance(dtype, pa.BaseExtensionType) or pa.types.is_dictionary(dtype): - dtype = dtype.storage_type if isinstance(dtype, pa.BaseExtensionType) else dtype.value_type - return dtype + # Extension annotations, dictionary indices and a null|T union are physical + # representations. + while True: + if isinstance(dtype, pa.BaseExtensionType): + dtype = dtype.storage_type + elif pa.types.is_dictionary(dtype): + dtype = dtype.value_type + elif (member := _nullable_member(dtype)) is not None: + dtype = dtype.field(member).type + else: + return dtype + + +def _has_nullable_union(dtype: pa.DataType) -> bool: + if _nullable_member(dtype) is not None: + return True + return any(_has_nullable_union(dtype.field(i).type) for i in range(dtype.num_fields)) + + +def _without_nullable_unions(array: pa.Array) -> pa.Array: + """`array` with every union of `null` and `T` replaced by the nullable `T` + it spells, at any depth of struct, list and map.""" + dtype = array.type + if not _has_nullable_union(dtype): + return array + member = _nullable_member(dtype) + if member is not None: + selected = pc.equal(array.type_codes, pa.scalar(dtype.type_codes[member], pa.int8())) + positions = (array.offsets if dtype.mode == "dense" + else pa.array(np.arange(len(array), dtype=np.int32))) + # A sparse member comes sliced with its union; a dense one is indexed by the offsets. + picked = array.field(member).take(pc.if_else(selected, positions, pa.scalar(None, positions.type))) + return _without_nullable_unions(picked) + validity = array.is_null() if array.null_count else None + if pa.types.is_struct(dtype): + children = [_without_nullable_unions(child) for child in array.flatten()] + return pa.StructArray.from_arrays(children, fields=[f.with_type(c.type) for f, c in zip(dtype, children)], + mask=validity) + if pa.types.is_map(dtype): + return pa.MapArray.from_arrays(array.offsets, _without_nullable_unions(array.keys), + _without_nullable_unions(array.items), mask=validity) + if pa.types.is_list(dtype) or pa.types.is_large_list(dtype): + factory = pa.LargeListArray if pa.types.is_large_list(dtype) else pa.ListArray + return factory.from_arrays(array.offsets, _without_nullable_unions(array.values), mask=validity) + if pa.types.is_fixed_size_list(dtype): + values = array.values.slice(array.offset * dtype.list_size, len(array) * dtype.list_size) + return pa.FixedSizeListArray.from_arrays(_without_nullable_unions(values), dtype.list_size, mask=validity) + return array def _compatible(got: pa.DataType, expected: pa.DataType) -> bool: @@ -51,7 +112,11 @@ def _compatible(got: pa.DataType, expected: pa.DataType) -> bool: if pa.types.is_run_end_encoded(got) and pa.types.is_run_end_encoded(expected): return _compatible(got.value_type, expected.value_type) if pa.types.is_timestamp(got) and pa.types.is_timestamp(expected): - return got.tz == expected.tz + # A zone labels instants; it is not data. Naive and zoned differ in kind. + return (got.tz is None) == (expected.tz is None) + if ((pa.types.is_integer(got) and pa.types.is_decimal(expected) and expected.scale == 0) + or (pa.types.is_decimal(got) and got.scale == 0 and pa.types.is_integer(expected))): + return True if (pa.types.is_fixed_size_binary(got) and pa.types.is_fixed_size_binary(expected) and got.byte_width != expected.byte_width): return False @@ -196,6 +261,7 @@ def values_equal(got: pa.Table, expected: pa.Table) -> tuple[bool, str]: return False, f"column {name!r}: incompatible logical types {got.column(name).type} vs {expected.column(name).type}" try: for g, e in _windows(got.column(name), expected.column(name)): + g, e = _without_nullable_unions(g), _without_nullable_unions(e) g, e = _materialize_views(g), _materialize_views(e) if pa.types.is_dictionary(g.type): g = g.dictionary_decode() diff --git a/raincloud/pipeline/export/exporters.py b/raincloud/pipeline/export/exporters.py index 9405f56..91e2e22 100644 --- a/raincloud/pipeline/export/exporters.py +++ b/raincloud/pipeline/export/exporters.py @@ -39,14 +39,20 @@ from ..discovery import VARIANT_EXT, has_variant from ..spec import ( + PARQUET_PAGE_INDEX_COLUMNS, + VORTEX_DATA_BLOCK_BYTES, + VORTEX_ROW_BLOCK_ROWS, + ParquetOptions, display_path, + parquet_options, + prepared_artifact, prepared_parquet, prepared_vortex, row_group_cap, row_group_probe_rows, row_group_target_bytes, row_group_target_encoded_bytes, - spec_field, + write_settings, ) from . import register from .base import Compliance, ExportResult, slug_from_canonical @@ -117,8 +123,47 @@ def _groups(reader, row_group: int, byte_target: int): yield pending, None +class UnsupportedOption(ValueError): + """A write option this writer's library cannot honour: the export fails, + and the failure is recorded, rather than the file being written another way.""" + + +def _leaf_paths(schema: pa.Schema) -> list[str]: + """The Parquet leaf column paths pyarrow writes `schema` as, in order.""" + sink = pa.BufferOutputStream() + pq.write_table(schema.empty_table(), sink) + written = pq.ParquetFile(pa.BufferReader(sink.getvalue())).schema + return [written.column(i).path for i in range(len(written))] + + +def _writer_options(options: ParquetOptions, schema: pa.Schema) -> dict: + """pyarrow's `ParquetWriter` arguments for `options`. An unset setting is + left out, so pyarrow's own default applies. + + pyarrow writes a page index for every column with statistics or for none, + so `page_index_columns` (page statistics for some columns, chunk statistics + for all) is refused. + """ + if options.page_index_columns is not None: + raise UnsupportedOption( + f"parquet@py cannot honour {PARQUET_PAGE_INDEX_COLUMNS}={options.page_index_columns}: pyarrow " + "writes a page index for every column that has statistics, or for none") + statistics: bool | list[str] = options.statistics + if options.statistics and options.statistics_columns is not None: + statistics = _leaf_paths(schema)[:options.statistics_columns] + kwargs = {"compression": options.compression, "write_statistics": statistics} + for name, value in (("compression_level", options.compression_level), + ("write_page_index", options.page_index), ("data_page_size", options.page_bytes), + ("max_rows_per_page", options.page_rows), ("use_dictionary", options.dictionary), + ("dictionary_pagesize_limit", options.dictionary_page_bytes), + ("write_page_checksum", options.page_checksums)): + if value is not None: + kwargs[name] = value + return kwargs + + def _write_parquet(canonical: Path, dest: Path, row_group: int, byte_target: int, - *, compression, stats) -> bool: + *, options: ParquetOptions) -> bool: """Stream the canonical into `dest` in the groups `_groups` cuts. Returns whether the byte ceiling closed EVERY group but the tail before its @@ -128,8 +173,7 @@ def _write_parquet(canonical: Path, dest: Path, row_group: int, byte_target: int byte_closed: list[bool] = [] # one flag per group, the tail left out with pa.ipc.open_file(str(canonical)) as reader: schema = reader.schema - with pq.ParquetWriter(dest, schema, compression=compression, - write_statistics=stats) as writer: + with pq.ParquetWriter(dest, schema, **_writer_options(options, schema)) as writer: for group, by_bytes in _groups(reader, row_group, byte_target): if by_bytes is not None: byte_closed.append(by_bytes) @@ -139,7 +183,7 @@ def _write_parquet(canonical: Path, dest: Path, row_group: int, byte_target: int return bool(byte_closed) and all(byte_closed) -def _probe_encoded(canonical: Path, want_rows: int, probe: Path, *, compression, stats) -> tuple[int, int]: +def _probe_encoded(canonical: Path, want_rows: int, probe: Path, *, options: ParquetOptions) -> tuple[int, int]: """Encode the real write's FIRST group at `want_rows` rows/group as ONE row group at `probe`; return (rows, encoded). The decoded-byte ceiling or the end of the file can make it shorter, as they would the real group.""" @@ -150,8 +194,7 @@ def _probe_encoded(canonical: Path, want_rows: int, probe: Path, *, compression, return 0, 0 try: table = pa.Table.from_batches(taken, schema=schema) - with pq.ParquetWriter(probe, schema, compression=compression, - write_statistics=stats) as writer: + with pq.ParquetWriter(probe, schema, **_writer_options(options, schema)) as writer: # One group of exactly this size. Left to its own default pyarrow # splits at ~1Mi rows, and the measurement would then describe a # group of THAT size -- useless, since dictionary and page overhead @@ -164,7 +207,7 @@ def _probe_encoded(canonical: Path, want_rows: int, probe: Path, *, compression, probe.unlink(missing_ok=True) -def _rows_for_encoded_target(canonical: Path, probe: Path, *, compression, stats, row_cap: int) -> int: +def _rows_for_encoded_target(canonical: Path, probe: Path, *, options: ParquetOptions, row_cap: int) -> int: """Rows per group that land near the encoded target — arrow-rs, approximated. Iterated, because bytes-per-row is not constant in the group size: a small @@ -172,15 +215,14 @@ def _rows_for_encoded_target(canonical: Path, probe: Path, *, compression, stats The probe file is written at `probe`, beside the destination. """ target = row_group_target_encoded_bytes() - rows, encoded = _probe_encoded(canonical, row_group_probe_rows(), probe, - compression=compression, stats=stats) + rows, encoded = _probe_encoded(canonical, row_group_probe_rows(), probe, options=options) if not rows or not encoded: return row_cap want = min(max(1, int(target * rows / encoded)), row_cap) for _ in range(2): if want <= rows: break - rows2, encoded2 = _probe_encoded(canonical, want, probe, compression=compression, stats=stats) + rows2, encoded2 = _probe_encoded(canonical, want, probe, options=options) if not rows2 or not encoded2: break if rows2 < want: @@ -229,12 +271,13 @@ def export(self, spec: dict, canonical: Path, dest: Path | None = None) -> Expor # multi-output transform. `spec` still supplies write.* opts + logging. slug = slug_from_canonical(canonical) - compression = spec_field(spec, "write.compression", "zstd") + # The recipe's compression and statistics, and the page knobs: the same + # options every Parquet lane is given (`spec.parquet_options`). + options = parquet_options(spec) # A CAP, not a target. arrow-rs's `max_row_group_row_count` and # parquet-java's `parquet.block.row.count.limit` are both effectively off # by default so that BYTES decide the group; absent here means uncapped. row_cap = row_group_cap(spec) - stats = spec_field(spec, "write.statistics", True) dest = dest or self.out_path(slug) dest.parent.mkdir(parents=True, exist_ok=True) @@ -261,19 +304,16 @@ def export(self, spec: dict, canonical: Path, dest: Path | None = None) -> Expor schema = reader.schema try: row_group = _rows_for_encoded_target( - canonical, tmp.with_suffix(".probe"), compression=compression, - stats=stats, row_cap=row_cap, + canonical, tmp.with_suffix(".probe"), options=options, row_cap=row_cap, ) - bytes_bound = _write_parquet(canonical, tmp, row_group, byte_target, - compression=compression, stats=stats) + bytes_bound = _write_parquet(canonical, tmp, row_group, byte_target, options=options) corrected = _corrected_rows(tmp, row_group, target_encoded, row_cap) # When the decoded-byte ceiling closed every group, more rows per # group are capped the same way and the second pass would rewrite # an identical file. if corrected is not None and not (bytes_bound and corrected > row_group): print(f" [row-groups] re-sizing {row_group:,} -> {corrected:,} rows/group") - _write_parquet(canonical, tmp, corrected, byte_target, - compression=compression, stats=stats) + _write_parquet(canonical, tmp, corrected, byte_target, options=options) tmp.replace(dest) finally: tmp.unlink(missing_ok=True) @@ -344,6 +384,15 @@ def export(self, spec: dict, canonical: Path, dest: Path | None = None) -> Expor raise BuildToolingMissing(f"{self.cell_id} {missing}") import vortex.io as vxio + settings = write_settings("vortex") + for field, var in (("row_block_rows", VORTEX_ROW_BLOCK_ROWS), ("data_block_bytes", VORTEX_DATA_BLOCK_BYTES)): + if settings[field] is not None: + raise UnsupportedOption(f"vortex@py cannot honour {var}={settings[field]}: vortex-data's " + "Python writer has no block size setting") + # Unset is `vxio.write`, the default options; compact is BtrBlocks' compact encodings. + write = (vxio.VortexWriteOptions.compact().write if settings["compact"] + else vxio.VortexWriteOptions.default().write if settings["compact"] is False else vxio.write) + # `with` closes the reader (RecordBatchFileReader is a context manager, # not a .close()-able) — vxio.write consumes the batch generator # synchronously inside the block, so the reader is done before exit. @@ -374,7 +423,7 @@ def batches(): tmp = tmp_path(dest) rbr = pa.RecordBatchReader.from_batches(schema, batches()) try: - vxio.write(rbr, str(tmp)) + write(rbr, str(tmp)) tmp.replace(dest) finally: tmp.unlink(missing_ok=True) @@ -395,5 +444,112 @@ def batches(): ) +def orc_storage_type(dtype: pa.DataType) -> pa.DataType: + """`dtype` as both ORC lanes store it: ORC has no unsigned integers and no + view types, so they widen to the next type that holds every value -- + uint8 -> int16, uint16 -> int32, uint32 -> int64, uint64 -> decimal(20, 0), + string/binary views -> their plain types -- at any depth. Always, not by the + data's range: a dataset's ORC schema must not change with its values. The + comparators read each back as the canonical's type (`sidecars/compare_cases`). + The Rust lane's `orc_storage_type` is the same rule.""" + widened = {pa.uint8(): pa.int16(), pa.uint16(): pa.int32(), pa.uint32(): pa.int64(), + pa.uint64(): pa.decimal128(20, 0), pa.string_view(): pa.string(), + pa.binary_view(): pa.binary()} + if dtype in widened: + return widened[dtype] + if pa.types.is_struct(dtype): + return pa.struct([f.with_type(orc_storage_type(f.type)) for f in dtype]) + if pa.types.is_map(dtype): + return pa.map_(dtype.key_field.with_type(orc_storage_type(dtype.key_type)), + dtype.item_field.with_type(orc_storage_type(dtype.item_type)), + keys_sorted=dtype.keys_sorted) + if pa.types.is_fixed_size_list(dtype): + return pa.list_(dtype.value_field.with_type(orc_storage_type(dtype.value_type)), dtype.list_size) + if pa.types.is_large_list(dtype): + return pa.large_list(dtype.value_field.with_type(orc_storage_type(dtype.value_type))) + if pa.types.is_list(dtype): + return pa.list_(dtype.value_field.with_type(orc_storage_type(dtype.value_type))) + return dtype + + +def _orc_writer_options(settings: dict) -> dict: + """pyarrow's `ORCWriter` arguments for the ORC write settings; an unset one + is left out, so pyarrow's default applies (zstd for the codec).""" + codec = settings["compression"] or "zstd" + kwargs = {"compression": "uncompressed" if codec == "none" else codec} + for name, value in (("compression_strategy", settings["compression_strategy"]), + ("stripe_size", settings["stripe_bytes"]), + ("compression_block_size", settings["compression_block_bytes"])): + if value is not None: + kwargs[name] = value + return kwargs + + +class OrcExporter: + """pyarrow ORC writer, the Apache ORC C++ library -- the `orc@py` cell. + + Streams the canonical's stored batches into one `ORCWriter`, with the ORC + write settings (`spec.FORMAT_SETTINGS["orc"]`): unset, zstd, since the API + makes the caller pick a codec (its default is none), and the library's own + stripe and compression block sizes and strategy. Unsigned integers and view types are widened + first (`orc_storage_type`); any other type the library does not write + (dictionaries, time, durations, ...) raises from it, and the build records + ORC unavailable for that dataset with its error. + """ + + format_id = "orc" + cell_id = "orc@py" + + def unavailable(self) -> str | None: + if importlib.util.find_spec("pyarrow._orc") is not None: + return None + return "needs a pyarrow built with ORC support (this platform's wheel has none)" + + def out_path(self, slug: str) -> Path: + return prepared_artifact(slug, "orc") + + def export(self, spec: dict, canonical: Path, dest: Path | None = None) -> ExportResult: + import pyarrow.orc as orc + + slug = slug_from_canonical(canonical) + dest = dest or self.out_path(slug) + dest.parent.mkdir(parents=True, exist_ok=True) + tmp = tmp_path(dest) + print(f"[export:orc@py] {display_path(dest)}") + with pa.ipc.open_file(str(canonical)) as reader: + schema = reader.schema + stored = pa.schema([f.with_type(orc_storage_type(f.type)) for f in schema], metadata=schema.metadata) + try: + writer = orc.ORCWriter(str(tmp), **_orc_writer_options(write_settings("orc"))) + try: + for i in range(reader.num_record_batches): + batch = pa.Table.from_batches([reader.get_batch(i)], schema=schema) + writer.write(batch if stored == schema else batch.cast(stored)) + if not reader.num_record_batches: + writer.write(stored.empty_table()) + finally: + writer.close() + tmp.replace(dest) + finally: + tmp.unlink(missing_ok=True) + + variant = has_variant(schema) + note = ("VARIANT column written as its shredded struct — ORC has no VARIANT type" + if variant else "") + roundtrip, why = read_back(self.cell_id, dest, canonical) + return ExportResult( + format_id=self.cell_id, + out_path=dest, + nbytes=dest.stat().st_size, + sha256=sha256_file(dest), + compliance=Compliance( + roundtrip=roundtrip, + variant_faithful=not variant, + note="; ".join(n for n in (why, note) if n), + ), + ) + + register(ParquetExporter()) register(VortexExporter()) +register(OrcExporter()) diff --git a/raincloud/pipeline/export/readers.py b/raincloud/pipeline/export/readers.py index 809daa5..b137c61 100644 --- a/raincloud/pipeline/export/readers.py +++ b/raincloud/pipeline/export/readers.py @@ -157,6 +157,15 @@ def vortex_batches(artifact: Path): return reader.schema, len(file), iter(reader) +def orc_batches(artifact: Path): + """(schema, rows, batches) of an ORC file read with pyarrow (the Apache ORC + C++ library), one stripe at a time.""" + import pyarrow.orc as orc + + file = orc.ORCFile(str(artifact)) + return file.schema, file.nrows, (file.read_stripe(i) for i in range(file.nstripes)) + + @contextmanager def canonical_batches(canonical: Path): """(schema, rows, batches) of the canonical Arrow IPC file, one stored batch @@ -288,6 +297,19 @@ def _run() -> Verdict: return _guarded(self.reader_id, _run) +class PyarrowOrcReader: + """In-process `orc@py` reader — pyarrow's ORC stripes vs the canonical's batches, streamed.""" + + reader_id = "orc@py" + formats = {"orc"} + + def read_conformance(self, artifact: Path, canonical: Path) -> Verdict: + def _run() -> Verdict: + return stream_verdict(self.reader_id, orc_batches(artifact), canonical) + + return _guarded(self.reader_id, _run) + + # --------------------------------------------------------------------------- # Sidecar reader (subprocess CLI) # --------------------------------------------------------------------------- @@ -449,8 +471,13 @@ def run_reader( # (via Hardwood) and `vortex@jni` (via vortex-jni). register_reader(PyarrowParquetReader()) register_reader(VortexPyReader()) +register_reader(PyarrowOrcReader()) register_reader(SidecarReader("parquet@rs", {"parquet"}, "raincloud-read-parquet-rs")) register_reader(SidecarReader("vortex@rs", {"vortex"}, "raincloud-read-vortex-rs")) register_reader(SidecarReader("parquet@java", {"parquet"}, "raincloud-read-parquet-java")) register_reader(SidecarReader("parquet@hardwood", {"parquet"}, "raincloud-read-parquet-hardwood")) register_reader(SidecarReader("vortex@jni", {"vortex"}, "raincloud-read-vortex-jni")) +register_reader(SidecarReader("orc@rs", {"orc"}, "raincloud-read-orc-rs")) +register_reader(SidecarReader("avro@rs", {"avro"}, "raincloud-read-avro-rs")) +register_reader(SidecarReader("avro@java", {"avro"}, "raincloud-read-avro-java")) +register_reader(SidecarReader("nimble@cpp", {"nimble"}, "raincloud-read-nimble-cpp")) diff --git a/raincloud/pipeline/export/sidecar.py b/raincloud/pipeline/export/sidecar.py index 8e39563..51916f4 100644 --- a/raincloud/pipeline/export/sidecar.py +++ b/raincloud/pipeline/export/sidecar.py @@ -83,10 +83,18 @@ import tempfile from pathlib import Path -from raincloud._cache import sha256_file +from raincloud._cache import EXT, sha256_file from raincloud._registry import SIDECAR_EXPORTERS -from ..spec import display_path, output_format_dir, row_group_cap, spec_field +from ..spec import ( + chosen_settings, + display_path, + output_format_dir, + row_group_cap, + setting_vars, + sidecar_settings, + spec_field, +) from . import register from .base import Compliance, ExportResult, slug_from_canonical from .bounded import export_timeout @@ -148,7 +156,7 @@ def __init__( self.cell_id = cell_id self.format_id = format_id self.binary = binary - self.ext = ext or format_id + self.ext = ext or EXT[format_id] def unavailable(self) -> str | None: if self._discover() is not None: @@ -164,18 +172,24 @@ def toolchain(self) -> dict[str, str]: exe = self._discover() found = shutil.which(exe) if exe else None return {"sidecar": self.binary, - **({"sidecar_sha256": sha256_file(Path(found))[:16]} if found else {})} + **({"sidecar_sha256": sha256_file(Path(found))[:16]} if found else {}), + **chosen_settings(self.format_id)} def _discover(self) -> str | None: """Resolve the reference-writer executable, or `None` if absent.""" return os.environ.get(_env_var(self.cell_id)) or shutil.which(self.binary) - @staticmethod - def _child_env(spec: dict) -> dict[str, str]: - """raincloud's environment, with the recipe's row cap when it declares one.""" + def _child_env(self, spec: dict) -> dict[str, str]: + """raincloud's environment, with the recipe's row cap when it declares one + and the format's write settings in the one form every sidecar reads + (`spec.sidecar_settings`): for Parquet the recipe's compression and + statistics, and each install setting only when it is set.""" env = dict(os.environ) if spec_field(spec, "write.row_group_size_rows"): env["RAINCLOUD_ROW_GROUP_MAX_ROWS"] = str(row_group_cap(spec)) + for var in setting_vars(self.format_id): + env.pop(var, None) + env.update(sidecar_settings(self.format_id, spec)) return env def _failure(self, dest: Path, reason: str, *, variant_faithful: bool = False) -> ExportResult: diff --git a/raincloud/pipeline/list_datasets.py b/raincloud/pipeline/list_datasets.py index d21f2b4..24c15cc 100644 --- a/raincloud/pipeline/list_datasets.py +++ b/raincloud/pipeline/list_datasets.py @@ -56,7 +56,7 @@ from typing import Any from raincloud._catalog import recorded_unavailable -from raincloud._formats import vortex_cells, vortex_skip_reason +from raincloud._formats import EXPORTED_FORMATS, vortex_cells, vortex_skip_reason from raincloud._suggest import hint from raincloud.exceptions import CatalogError @@ -76,7 +76,7 @@ iter_datasets, load_manifest, outputs_root, - prepared_arrow, + prepared_artifact, prepared_parquet, prepared_vortex, spec_field, @@ -227,8 +227,7 @@ def local_formats(spec: dict, manifest: dict) -> list[str]: """Formats of `spec` whose artifact is on this machine's disk, for the current schema_version. Presence only: `raincloud status` checks staleness.""" slug = spec["slug"] - paths = {"arrow": prepared_arrow, "parquet": prepared_parquet, "vortex": prepared_vortex} - return [fmt for fmt, path in paths.items() if path(slug, manifest).is_file()] + return [fmt for fmt in ("arrow", *EXPORTED_FORMATS) if prepared_artifact(slug, fmt, manifest).is_file()] def _filter_state_from_args(args) -> FilterState: diff --git a/raincloud/pipeline/publish.py b/raincloud/pipeline/publish.py index e4c390b..ddddf6e 100644 --- a/raincloud/pipeline/publish.py +++ b/raincloud/pipeline/publish.py @@ -23,7 +23,8 @@ import fsspec -from raincloud._cache import EXT, sha256_file +from raincloud._cache import sha256_file +from raincloud._formats import EXPORTED_FORMATS from raincloud._locking import locked from raincloud._resolve import artifact_key from raincloud._transport import filesystem_url @@ -38,10 +39,10 @@ class PublishMismatch(Exception): """On-disk artifact sha256 disagrees with the snapshot.""" -# Every artifact format the loader can fetch (`_cache.EXT`). The canonical -# .arrow.zstd leads because every exporter reads from it; it is published and -# sha-gated like the export formats. A format absent on disk is skipped. -_PUBLISH_FORMATS = ("arrow", *(f for f in EXT if f != "arrow")) +# Every artifact format the loader can fetch. The canonical .arrow.zstd leads +# because every exporter reads from it; it is published and sha-gated like the +# export formats. A format absent on disk is skipped. +_PUBLISH_FORMATS = ("arrow", *EXPORTED_FORMATS) def scrape_advisory_slugs(manifest, slugs): diff --git a/raincloud/pipeline/spec.py b/raincloud/pipeline/spec.py index 8dd500a..08254e7 100644 --- a/raincloud/pipeline/spec.py +++ b/raincloud/pipeline/spec.py @@ -18,6 +18,7 @@ import sys import tempfile from contextlib import contextmanager +from dataclasses import dataclass, replace from pathlib import Path from typing import Any, Iterator @@ -232,6 +233,265 @@ def row_group_probe_rows() -> int: return value +# Parquet write options every Parquet writer receives the same way. The install +# settings differ from the row-group knobs in one respect: UNSET means each +# writer's own default, not a raincloud figure, because the four libraries' +# defaults differ (arrow-rs and parquet-java write a page index, pyarrow and +# Hardwood do not) and no one figure leaves every lane's files as they are. +# Set, a setting reaches every lane, and a writer whose library cannot do what +# it asks fails that export rather than writing something else. +PARQUET_COMPRESSION = "RAINCLOUD_PARQUET_COMPRESSION" +PARQUET_STATISTICS = "RAINCLOUD_PARQUET_STATISTICS" +PARQUET_COMPRESSION_LEVEL = "RAINCLOUD_PARQUET_COMPRESSION_LEVEL" +PARQUET_STATISTICS_COLUMNS = "RAINCLOUD_PARQUET_STATISTICS_COLUMNS" +PARQUET_PAGE_INDEX = "RAINCLOUD_PARQUET_PAGE_INDEX" +PARQUET_PAGE_INDEX_COLUMNS = "RAINCLOUD_PARQUET_PAGE_INDEX_COLUMNS" +PARQUET_PAGE_BYTES = "RAINCLOUD_PARQUET_PAGE_BYTES" +PARQUET_PAGE_ROWS = "RAINCLOUD_PARQUET_PAGE_ROWS" +PARQUET_DICTIONARY = "RAINCLOUD_PARQUET_DICTIONARY" +PARQUET_DICTIONARY_PAGE_BYTES = "RAINCLOUD_PARQUET_DICTIONARY_PAGE_BYTES" +PARQUET_PAGE_CHECKSUMS = "RAINCLOUD_PARQUET_PAGE_CHECKSUMS" +PARQUET_CODECS = ("zstd", "snappy", "gzip", "lz4", "brotli", "none") +# The levels a codec takes, in every lane that sets one (arrow-rs's ranges). +CODEC_LEVELS = {"zstd": range(1, 23), "gzip": range(0, 10), "brotli": range(0, 12)} +_COUNT_LIMIT = (1 << 31) - 1 # parquet-java and Hardwood take an int +_TRUE, _FALSE = {"1", "true", "yes", "on"}, {"0", "false", "no", "off"} + + +def _env_switch(var: str) -> bool | None: + """An on/off setting: None when unset or empty, else one of 1/true/yes/on + or 0/false/no/off (any case); anything else raises, naming the variable.""" + raw = os.environ.get(var) + value = (raw or "").strip(_ASCII_SPACE).lower() + if not value: + return None + if value in _TRUE | _FALSE: + return value in _TRUE + raise ValueError(f"{var}={raw!r} is not a switch; give 1 or 0 (true/false, yes/no, on/off), " + "or leave it unset for each writer's own default") + + +def _env_limit(var: str) -> int | None: + """A size or count setting: None when unset (the writer's default); the + count grammar otherwise, where 0 or empty means no limit.""" + if os.environ.get(var) is None: + return None + value = _env_count(var, 0.0, limit=_COUNT_LIMIT) + return _COUNT_LIMIT if value is None else value + + +def _env_level(var: str) -> int | None: + """A compression level: None when unset or empty, else plain ASCII digits.""" + raw = os.environ.get(var) + value = (raw or "").strip(_ASCII_SPACE) + if not value: + return None + if not value.isascii() or not value.isdigit(): + raise ValueError(f"{var}={raw!r} is not a compression level; give a whole number such as 3") + return int(value) + + +def check_level(var: str, codec: str, level: int | None, levels: dict[str, range]) -> None: + """Refuse a level the codec does not take, in every lane alike.""" + if level is None: + return + allowed = levels.get(codec) + if allowed is None: + raise ValueError(f"{var}={level}: {codec} takes no compression level") + if level not in allowed: + raise ValueError(f"{var}={level} is outside {codec}'s levels {allowed.start}..{allowed.stop - 1}") + + +# (field, variable, reader): every install setting, declared once. The +# toolchain key is the field prefixed with "parquet_". +_PARQUET_SETTINGS = ( + ("compression_level", PARQUET_COMPRESSION_LEVEL, _env_level), + ("statistics_columns", PARQUET_STATISTICS_COLUMNS, _env_limit), + ("page_index", PARQUET_PAGE_INDEX, _env_switch), + ("page_index_columns", PARQUET_PAGE_INDEX_COLUMNS, _env_limit), + ("page_bytes", PARQUET_PAGE_BYTES, _env_limit), + ("page_rows", PARQUET_PAGE_ROWS, _env_limit), + ("dictionary", PARQUET_DICTIONARY, _env_switch), + ("dictionary_page_bytes", PARQUET_DICTIONARY_PAGE_BYTES, _env_limit), + ("page_checksums", PARQUET_PAGE_CHECKSUMS, _env_switch), +) +PARQUET_SETTING_VARS = tuple(var for _, var, _ in _PARQUET_SETTINGS) + + +@dataclass(frozen=True) +class ParquetOptions: + """What one dataset's Parquet file is written with, in every writer lane. + + `compression` and `statistics` are the recipe's `write.compression` and + `write.statistics`. The rest are install settings (`_PARQUET_SETTINGS`); + None leaves the writer's own default: + + - `compression_level`: the codec's level (zstd 1-22, gzip 0-9, brotli 0-11); + - `statistics_columns`: statistics only for the first N leaf columns; + - `page_index`: a ColumnIndex and OffsetIndex for every column chunk, or none; + - `page_index_columns`: page statistics (the ColumnIndex) only for the + first N leaf columns, every column keeping its chunk statistics; + - `page_bytes` / `page_rows`: the data page size target and row limit; + - `dictionary`: dictionary encoding on or off; + - `dictionary_page_bytes`: the dictionary page size limit; + - `page_checksums`: a CRC in every page header, or none. + + A count of `_COUNT_LIMIT` is no limit (`0` in the environment). + """ + compression: str = "zstd" + statistics: bool = True + compression_level: int | None = None + statistics_columns: int | None = None + page_index: bool | None = None + page_index_columns: int | None = None + page_bytes: int | None = None + page_rows: int | None = None + dictionary: bool | None = None + dictionary_page_bytes: int | None = None + page_checksums: bool | None = None + + def _set(self): + for field, var, _ in _PARQUET_SETTINGS: + value = getattr(self, field) + if value is not None: + yield field, var, str(int(value)) + + def chosen(self) -> dict[str, str]: + """The settings that are set, as a writer's toolchain records them: a + recorded failure is repeated only under the same options.""" + return {f"parquet_{field}": value for field, _, value in self._set()} + + def env(self) -> dict[str, str]: + """The options as a sidecar writer reads them from its environment.""" + return {PARQUET_COMPRESSION: self.compression, PARQUET_STATISTICS: str(int(self.statistics)), + **{var: value for _, var, value in self._set()}} + + +def parquet_page_options() -> ParquetOptions: + """The install settings from the environment, with the recipe fields at + their defaults.""" + return ParquetOptions(**{field: read(var) for field, var, read in _PARQUET_SETTINGS}) + + +def parquet_options(spec: dict) -> ParquetOptions: + """The Parquet write options for `spec`: its recipe's `write.compression` + and `write.statistics`, and the install settings from the environment. + Settings that contradict each other, or the recipe, are refused here, for + every lane at once.""" + compression = spec_field(spec, "write.compression", "zstd") + if compression not in PARQUET_CODECS: + raise ValueError(f"write.compression={compression!r} is not one of {', '.join(PARQUET_CODECS)}") + statistics = bool(spec_field(spec, "write.statistics", True)) + options = replace(parquet_page_options(), compression=compression, statistics=statistics) + check_level(PARQUET_COMPRESSION_LEVEL, compression, options.compression_level, CODEC_LEVELS) + if not statistics: + for var, value in ((PARQUET_PAGE_INDEX, options.page_index and 1), + (PARQUET_STATISTICS_COLUMNS, options.statistics_columns), + (PARQUET_PAGE_INDEX_COLUMNS, options.page_index_columns)): + if value: + raise ValueError(f"{var}={value} asks for statistics, but the recipe sets " + "write.statistics to false") + if options.page_index is False and options.page_index_columns is not None: + raise ValueError(f"{PARQUET_PAGE_INDEX_COLUMNS}={options.page_index_columns} asks for a page index, " + f"but {PARQUET_PAGE_INDEX}=0") + return options + + +# Write settings for the other exported formats, declared and read like the +# Parquet ones (above): one list per format, every writer of the format given +# the same values, and a writer whose library cannot honour a set one refusing +# it. These formats have no recipe fields, so an unset codec is what raincloud +# has always written, zstd; any other unset setting is the library's default. +ORC_COMPRESSION = "RAINCLOUD_ORC_COMPRESSION" +ORC_COMPRESSION_STRATEGY = "RAINCLOUD_ORC_COMPRESSION_STRATEGY" +ORC_STRIPE_BYTES = "RAINCLOUD_ORC_STRIPE_BYTES" +ORC_COMPRESSION_BLOCK_BYTES = "RAINCLOUD_ORC_COMPRESSION_BLOCK_BYTES" +AVRO_COMPRESSION = "RAINCLOUD_AVRO_COMPRESSION" +AVRO_COMPRESSION_LEVEL = "RAINCLOUD_AVRO_COMPRESSION_LEVEL" +AVRO_BLOCK_BYTES = "RAINCLOUD_AVRO_BLOCK_BYTES" +VORTEX_COMPACT = "RAINCLOUD_VORTEX_COMPACT" +VORTEX_ROW_BLOCK_ROWS = "RAINCLOUD_VORTEX_ROW_BLOCK_ROWS" +VORTEX_DATA_BLOCK_BYTES = "RAINCLOUD_VORTEX_DATA_BLOCK_BYTES" +ORC_CODECS = ("zstd", "snappy", "zlib", "lz4", "none") +AVRO_CODECS = ("zstd", "deflate", "snappy", "bzip2", "xz", "none") +AVRO_CODEC_LEVELS = {"zstd": range(1, 23), "deflate": range(0, 10), "xz": range(0, 10)} + + +def _env_choice(*choices: str): + """A reader for a setting that names one of `choices`: None when unset or empty.""" + def read(var: str) -> str | None: + raw = os.environ.get(var) + value = (raw or "").strip(_ASCII_SPACE).lower() + if not value: + return None + if value not in choices: + raise ValueError(f"{var}={raw!r} is not one of {', '.join(choices)}") + return value + return read + + +# (field, variable, reader) per format, as `_PARQUET_SETTINGS`. +FORMAT_SETTINGS = { + "orc": ( + ("compression", ORC_COMPRESSION, _env_choice(*ORC_CODECS)), + ("compression_strategy", ORC_COMPRESSION_STRATEGY, _env_choice("speed", "compression")), + ("stripe_bytes", ORC_STRIPE_BYTES, _env_limit), + ("compression_block_bytes", ORC_COMPRESSION_BLOCK_BYTES, _env_limit), + ), + "avro": ( + ("compression", AVRO_COMPRESSION, _env_choice(*AVRO_CODECS)), + ("compression_level", AVRO_COMPRESSION_LEVEL, _env_level), + ("block_bytes", AVRO_BLOCK_BYTES, _env_limit), + ), + "vortex": ( + ("compact", VORTEX_COMPACT, _env_switch), + ("row_block_rows", VORTEX_ROW_BLOCK_ROWS, _env_limit), + ("data_block_bytes", VORTEX_DATA_BLOCK_BYTES, _env_limit), + ), +} + + +def write_settings(fmt: str) -> dict[str, Any]: + """`fmt`'s install write settings from the environment, None where unset: + the values every writer of the format is given. Not Parquet's, which also + take the recipe (`parquet_options`).""" + settings = {field: read(var) for field, var, read in FORMAT_SETTINGS.get(fmt, ())} + if fmt == "avro": + check_level(AVRO_COMPRESSION_LEVEL, settings["compression"] or "zstd", settings["compression_level"], + AVRO_CODEC_LEVELS) + return settings + + +def _canonical(value) -> str: + return str(int(value)) if isinstance(value, bool | int) else str(value) + + +def setting_vars(fmt: str) -> tuple[str, ...]: + """Every environment variable a writer of `fmt` reads its settings from.""" + if fmt == "parquet": + return (PARQUET_COMPRESSION, PARQUET_STATISTICS, *PARQUET_SETTING_VARS) + return tuple(var for _, var, _ in FORMAT_SETTINGS.get(fmt, ())) + + +def chosen_settings(fmt: str) -> dict[str, str]: + """The install settings of `fmt` that are set, as a writer's toolchain + records them, keyed `_`.""" + if fmt == "parquet": + return parquet_page_options().chosen() + return {f"{fmt}_{field}": _canonical(value) for field, value in write_settings(fmt).items() + if value is not None} + + +def sidecar_settings(fmt: str, spec: dict) -> dict[str, str]: + """`fmt`'s settings as a sidecar writer reads them: each set one in one + canonical form (`1`/`0`, a whole number, a lower-case name).""" + if fmt == "parquet": + return parquet_options(spec).env() + settings = write_settings(fmt) + return {var: _canonical(settings[field]) for field, var, _ in FORMAT_SETTINGS.get(fmt, ()) + if settings[field] is not None} + + def max_decompressed_bytes() -> int | None: """Ceiling on a single in-memory decompression, from `RAINCLOUD_MAX_DECOMPRESSED_BYTES`. `0` (or empty) disables it. Default 4 GiB. @@ -503,6 +763,13 @@ def output_format_dir(slug: str, fmt: str = "parquet", return outputs_root(manifest) / slug / fmt +def prepared_artifact(slug: str, fmt: str, manifest: dict | None = None) -> Path: + """outputs/v{n}///. — the dataset's one file of an + artifact format (`_registry.FORMATS`), whichever writer made it.""" + from raincloud._cache import EXT + return output_format_dir(slug, fmt, manifest) / f"{slug}.{EXT[fmt]}" + + def prepared_parquet(slug: str, manifest: dict | None = None) -> Path: """outputs/v{n}//parquet/.parquet — the canonical prepared parquet.""" return output_format_dir(slug, "parquet", manifest) / f"{slug}.parquet" diff --git a/raincloud/pipeline/status.py b/raincloud/pipeline/status.py index e9a5672..60589bc 100644 --- a/raincloud/pipeline/status.py +++ b/raincloud/pipeline/status.py @@ -2,7 +2,9 @@ # SPDX-License-Identifier: Apache-2.0 """Report per-dataset state across the manifest. -For each DatasetSpec in sources.json, walk the filesystem and report: +For each DatasetSpec in sources.json, walk the filesystem and report what this +install builds and keeps (its `formats`, `keep_raw` and `keep_canonical` +settings): raw — the selected recipe generation's raw bytes (under the configured raw root, `/` or `/.recipes//`) present, and matching expected_bytes if the manifest declared one for a @@ -11,11 +13,16 @@ work — the recipe's scratch directory under the configured scratch root present (extract scratch — wiped by --clean-workdir) arrow — (schema_version 2+) the canonical outputs/v{n}//arrow/.arrow.zstd present - parquet — outputs/v{n}//parquet/.parquet present when the export - policy includes Parquet; row count vs expect.rows; stale (v2) when - older than the canonical Arrow - vortex — the Vortex export present when the export policy includes Vortex; - stale when older than its source (the canonical Arrow in v2) + parquet — outputs/v{n}//parquet/.parquet present when this install + builds Parquet; row count vs expect.rows; stale (v2) when older + than the canonical Arrow + vortex — the Vortex export present when this install builds Vortex; stale + when older than its source (the canonical Arrow in v2) + — likewise for each other format this install builds, as a column + when any dataset has it + +A raw download or canonical Arrow the install does not keep is reported, but +its absence leaves a dataset complete. A format a build measured unavailable at the current recipe (this install's build record, else the selected catalog's snapshot) shows `unavail` and counts @@ -45,13 +52,15 @@ import pyarrow.parquet as pq -from raincloud._formats import buildable_formats, vortex_cells +from raincloud._formats import EXPORTED_FORMATS, buildable_formats, vortex_cells, wanted_formats +from raincloud.config import get_config from .selection import SelectionError, select_specs from .spec import ( is_hydrated, load_manifest, prepared_arrow, + prepared_artifact, prepared_parquet, prepared_vortex, raw_slug_dir, @@ -111,12 +120,20 @@ def _parquet_path(spec: dict, m: dict) -> Path: def _arrow_status(spec: dict, m: dict) -> dict: + """The canonical Arrow: `expected` when this install keeps it.""" if m["schema_version"] < 2: # v1 catalogs only; remove when v1 bundles are no longer read. return {"expected": False} + expected = get_config().keep_canonical p = prepared_arrow(spec["slug"], m) if not p.is_file(): - return {"expected": True, "present": False} - return {"expected": True, "present": True, "bytes": p.stat().st_size} + return {"expected": expected, "present": False} + return {"expected": expected, "present": True, "bytes": p.stat().st_size} + + +def _builds_here(fmt: str, m: dict) -> bool: + """Whether this install builds `fmt` (its `formats` setting); a v1 catalog + predates the setting and builds what its recipes say.""" + return m["schema_version"] < 2 or fmt in wanted_formats(get_config()) def _measured(spec: dict, fmt: str, m: dict) -> dict | None: @@ -152,23 +169,32 @@ def _record() -> dict: return _RECORD["artifacts"] -def _parquet_status(spec: dict, m: dict, *, fast: bool) -> dict: - if "parquet" not in buildable_formats(spec, m["schema_version"]): +def _format_status(spec: dict, m: dict, fmt: str) -> dict: + """An exported format's state: `expected` (the dataset offers it and this + install builds it), `present` with its `bytes`, `stale` when older than the + canonical Arrow it was exported from, and `unavailable` (the measurement) + when a build measured its writer unable to produce it.""" + if fmt not in buildable_formats(spec, m["schema_version"]) or not _builds_here(fmt, m): return {"expected": False} - measured = _measured(spec, "parquet", m) + measured = _measured(spec, fmt, m) if measured is not None: return {"expected": True, "present": False, "unavailable": measured} - p = _parquet_path(spec, m) + p = prepared_artifact(spec["slug"], fmt, m) if not p.exists(): return {"expected": True, "present": False} info: dict[str, Any] = {"expected": True, "present": True, "bytes": p.stat().st_size} canonical = prepared_arrow(spec["slug"], m) if m["schema_version"] >= 2 else None if canonical is not None and canonical.exists() and canonical.stat().st_mtime > p.stat().st_mtime: info["stale"] = True # exported from an older canonical: a display hint, as for vortex - if fast: + return info + + +def _parquet_status(spec: dict, m: dict, *, fast: bool) -> dict: + info = _format_status(spec, m, "parquet") + if fast or not info.get("present"): return info try: - pf = pq.ParquetFile(p) + pf = pq.ParquetFile(_parquet_path(spec, m)) info["rows"] = pf.metadata.num_rows except Exception as e: info["error"] = f"{type(e).__name__}: {e}" @@ -188,7 +214,7 @@ def vortex_status(spec: dict, m: dict, *, source: Path | None = None, `vortex` and `source` default to the prepared paths; `source` is the canonical Arrow in schema_version 2 and the Parquet in v1. """ - if not vortex_cells(spec, m["schema_version"], m): + if not vortex_cells(spec, m["schema_version"], m) or not _builds_here("vortex", m): return {"opted_in": False} measured = _measured(spec, "vortex", m) if measured is not None: @@ -207,6 +233,12 @@ def vortex_status(spec: dict, m: dict, *, source: Path | None = None, return info +# Formats with a status of their own: Parquet also reports its footer's row +# count, Vortex keeps the `opted_in` shape the browser reads. Every other +# exported format is reported by `_format_status`. +_OTHER_FORMATS = tuple(fmt for fmt in EXPORTED_FORMATS if fmt not in ("parquet", "vortex")) + + def gather(spec: dict, m: dict, *, fast: bool) -> dict: return { "slug": spec["slug"], @@ -215,29 +247,32 @@ def gather(spec: dict, m: dict, *, fast: bool) -> dict: "arrow": _arrow_status(spec, m), "parquet": _parquet_status(spec, m, fast=fast), "vortex": vortex_status(spec, m), + **{fmt: _format_status(spec, m, fmt) for fmt in _OTHER_FORMATS}, } +def _wanted(state: dict) -> bool: + """A format the export policy includes, not measured unavailable.""" + return bool((state.get("expected") or state.get("opted_in")) and not state.get("unavailable")) + + def _is_incomplete(row: dict) -> bool: arrow = row["arrow"] parq = row["parquet"] - vrtx = row["vortex"] return bool( - not row["raw"].get("present") + (not row["raw"].get("present") and get_config().keep_raw) or row["raw"].get("bytes_expected") is not None or (arrow.get("expected") and not arrow.get("present")) - or (parq.get("expected") and not parq.get("unavailable") - and (not parq.get("present") or parq.get("stale"))) or parq.get("rows_expected") is not None or parq.get("error") - or (vrtx.get("opted_in") and not vrtx.get("unavailable") - and (not vrtx.get("present") or vrtx.get("stale"))) + or any(_wanted(row.get(fmt, {})) and (not row[fmt].get("present") or row[fmt].get("stale")) + for fmt in EXPORTED_FORMATS) ) # ---------- rendering ---------- -def _fmt_row(row: dict) -> tuple[str, ...]: +def _fmt_row(row: dict, rows: list[dict] | None = None) -> tuple[str, ...]: raw = row["raw"] if raw.get("error"): raw_cell = "err" @@ -249,7 +284,9 @@ def _fmt_row(row: dict) -> tuple[str, ...]: work_cell = "✓" if row["work"].get("present") else "·" arrow = row["arrow"] - arrow_cell = "n/a" if not arrow.get("expected") else ("✓" if arrow.get("present") else "·") + # A canonical the install does not keep can still be present: kept as the + # dataset's only file, or left by an earlier build. + arrow_cell = "✓" if arrow.get("present") else ("·" if arrow.get("expected") else "n/a") parq = row["parquet"] if not parq.get("expected"): @@ -269,24 +306,29 @@ def _fmt_row(row: dict) -> tuple[str, ...]: else: parq_cell = "✓" - v = row["vortex"] - if not v.get("opted_in"): - vrtx_cell = "n/a" - elif v.get("unavailable"): - vrtx_cell = "unavail" - elif not v.get("present"): - vrtx_cell = "·" - elif v.get("stale"): - vrtx_cell = "stale" - else: - vrtx_cell = "✓" + others = tuple(_presence_cell(row[fmt]) for fmt in ("vortex", *_shown(rows or [row]))) + return (row["slug"], raw_cell, work_cell, arrow_cell, parq_cell, *others) + + +def _presence_cell(state: dict) -> str: + if not (state.get("expected") or state.get("opted_in")): + return "n/a" + if state.get("unavailable"): + return "unavail" + if not state.get("present"): + return "·" + return "stale" if state.get("stale") else "✓" + - return row["slug"], raw_cell, work_cell, arrow_cell, parq_cell, vrtx_cell +def _shown(rows: list[dict]) -> tuple[str, ...]: + """The formats beyond parquet and vortex that some row's policy includes: + a column no dataset exports would be all n/a.""" + return tuple(fmt for fmt in _OTHER_FORMATS if any(r.get(fmt, {}).get("expected") for r in rows)) def render_table(rows: list[dict]) -> str: - headers = ("slug", "raw", "work", "arrow", "parquet", "vortex") - cells = [headers] + [_fmt_row(r) for r in rows] + headers = ("slug", "raw", "work", "arrow", "parquet", "vortex", *_shown(rows)) + cells = [headers] + [_fmt_row(r, rows) for r in rows] widths = [max(len(r[i]) for r in cells) for i in range(len(headers))] out = [] for i, r in enumerate(cells): @@ -312,10 +354,15 @@ def render_summary(rows: list[dict]) -> str: vrtx_opt = [r for r in rows if r["vortex"].get("opted_in")] vrtx_ok = sum(1 for r in vrtx_opt if r["vortex"].get("present") and not r["vortex"].get("stale")) arrow = f" · arrow {arrow_ok}/{len(arrow_exp)}" if arrow_exp else "" - unavailable = sum(1 for r in rows for fmt in ("parquet", "vortex") if r[fmt].get("unavailable")) + unavailable = sum(1 for r in rows for fmt in EXPORTED_FORMATS if r.get(fmt, {}).get("unavailable")) + others = "" + for fmt in _shown(rows): + expected = [r for r in rows if r[fmt].get("expected")] + ok = sum(1 for r in expected if r[fmt].get("present") and not r[fmt].get("stale")) + others += f" · {fmt} {ok}/{len(expected)}" return (f"\n{n} slugs · raw {raw_ok}/{n}{arrow} · parquet {parq_ok}/{len(parq_exp)}" f" · rows-match {rows_ok}/{len(parq_exp)}" - f" · vortex {vrtx_ok}/{len(vrtx_opt)}" + f" · vortex {vrtx_ok}/{len(vrtx_opt)}{others}" + (f" · {unavailable} measured unavailable" if unavailable else "")) diff --git a/raincloud/pipeline/validate_manifest.py b/raincloud/pipeline/validate_manifest.py index 0cd8ff1..e86846a 100644 --- a/raincloud/pipeline/validate_manifest.py +++ b/raincloud/pipeline/validate_manifest.py @@ -49,7 +49,7 @@ from collections import Counter, defaultdict from pathlib import Path -from raincloud._formats import WRITERS, export_formats, priority_shape_error +from raincloud._formats import EXPORTED_FORMATS, WRITERS, export_formats, priority_shape_error from raincloud.exceptions import CatalogError from .discovery import SHOWCASE_TIERS, TAG_VOCAB @@ -59,7 +59,7 @@ # which skips an unknown name: in a manifest it is a typo that would fall # through to the next writer unnoticed. `canonical` writes only the Arrow # spine, so it is no export writer. -EXPORT_WRITERS = {fmt: WRITERS[fmt] for fmt in ("parquet", "vortex")} +EXPORT_WRITERS = {fmt: WRITERS[fmt] for fmt in EXPORTED_FORMATS} def _schema_path(): @@ -332,10 +332,10 @@ def _cross_checks(manifest: dict) -> tuple[list[str], list[str]]: # comes from export.priority. formats = (d.get("export") or {}).get("formats") or [] # a list: see _field_errors for fmt in formats: - if fmt not in ("parquet", "vortex"): + if fmt not in EXPORTED_FORMATS: hint = " (name the format; export.priority picks the writer)" if isinstance(fmt, str) and "@" in fmt else "" errors.append(f"{slug}: export.formats entry {fmt!r} is not an exported format " - f"(parquet, vortex){hint}") + f"({', '.join(EXPORTED_FORMATS)}){hint}") if d.get("derive"): continue fetch = d.get("fetch") or {} diff --git a/scripts/test_native_readers.py b/scripts/test_native_readers.py index eabd999..800fb6f 100644 --- a/scripts/test_native_readers.py +++ b/scripts/test_native_readers.py @@ -72,9 +72,11 @@ def run(args, **kwargs): no_vortex_target = target / "no-vortex" run(["cargo", "build", *client, "--no-default-features", "--target-dir", no_vortex_target]) run(["cargo", "build", "--locked", "--manifest-path", "sidecars/rust/Cargo.toml", - "--bin", "parquet-read", "--bin", "vortex-read", "--bin", "parquet-write", "--bin", "vortex-write"]) + "--bin", "parquet-read", "--bin", "vortex-read", "--bin", "parquet-write", "--bin", "vortex-write", + "--bin", "orc-read", "--bin", "orc-write", "--bin", "avro-read", "--bin", "avro-write"]) lib = target / "debug" - for binary in ("parquet-read", "vortex-read", "parquet-write", "vortex-write"): + for binary in ("parquet-read", "vortex-read", "parquet-write", "vortex-write", "orc-read", "orc-write", + "avro-read", "avro-write"): if not os.access(lib / binary, os.X_OK): raise RuntimeError(f"required conformance reader is not executable: {lib / binary}") if not env.get("JAVA_HOME") and shutil.which("java"): @@ -82,11 +84,13 @@ def run(args, **kwargs): # parquet-hardwood builds on a Java 21 toolchain (Gradle provisions one if none is # installed); its launchers default to that JDK when JAVA_HOME names an older one. run(["bash", "sidecars/java/gradlew", "-p", "sidecars/java", ":parquet-java:installDist", - ":parquet-hardwood:installDist", ":vortex-jni-reader:installDist", "--no-daemon"]) + ":parquet-hardwood:installDist", ":vortex-jni-reader:installDist", ":avro-java:installDist", + "--no-daemon"]) install = root / "sidecars/java" parquet_java = install / "parquet-java/build/install/raincloud-export-parquet-java/bin" hardwood = install / "parquet-hardwood/build/install/raincloud-export-parquet-hardwood/bin" vortex_jni = install / "vortex-jni-reader/build/install/raincloud-read-vortex-jni/bin" + avro_java = install / "avro-java/build/install/raincloud-export-avro-java/bin" env.update( RAINCLOUD_SIDECAR_VORTEX_RS=str(lib / "vortex-write"), RAINCLOUD_SIDECAR_PARQUET_JAVA=str(parquet_java / "raincloud-export-parquet-java"), @@ -95,6 +99,8 @@ def run(args, **kwargs): RAINCLOUD_READER_PARQUET_HARDWOOD=str(hardwood / "raincloud-read-parquet-hardwood"), RAINCLOUD_SIDECAR_VORTEX_JNI=str(vortex_jni / "raincloud-export-vortex-jni"), RAINCLOUD_READER_VORTEX_JNI=str(vortex_jni / "raincloud-read-vortex-jni"), + RAINCLOUD_SIDECAR_AVRO_JAVA=str(avro_java / "raincloud-export-avro-java"), + RAINCLOUD_READER_AVRO_JAVA=str(avro_java / "raincloud-read-avro-java"), ) (root / ".tmp").mkdir(exist_ok=True) with tempfile.TemporaryDirectory(prefix="raincloud-reader-", dir=root / ".tmp") as temp: @@ -111,7 +117,12 @@ def run(args, **kwargs): RAINCLOUD_NATIVE_LIBRARY_NO_VORTEX=str(no_vortex_target / "debug/libraincloud_reader.so"), RAINCLOUD_READER_PARQUET_RS=str(lib / "parquet-read"), RAINCLOUD_SIDECAR_PARQUET_RS=str(lib / "parquet-write"), - RAINCLOUD_READER_VORTEX_RS=str(lib / "vortex-read")) + RAINCLOUD_READER_VORTEX_RS=str(lib / "vortex-read"), + RAINCLOUD_SIDECAR_ORC_RS=str(lib / "orc-write"), + RAINCLOUD_READER_ORC_RS=str(lib / "orc-read"), + RAINCLOUD_SIDECAR_AVRO_RS=str(lib / "avro-write"), + # nimble@cpp is built by sidecars/nimble/build.sh, not here: absent. + RAINCLOUD_READER_AVRO_RS=str(lib / "avro-read")) run(["cargo", "test", *client]) run(["cargo", "test", *client, "--no-default-features", "--target-dir", no_vortex_target]) run([sys.executable, "-m", "pytest", *PYTEST_MODULES, "-q"]) diff --git a/sidecars/README.md b/sidecars/README.md index feec13d..3d7acb7 100644 --- a/sidecars/README.md +++ b/sidecars/README.md @@ -7,15 +7,15 @@ these programs during ordinary reads. ## Which writer a build uses -A dataset has one file per format; `export.formats` says which formats, never -which writer. For each format a build takes the first writer that is installed, +A dataset has one file per format; the install's `formats` setting says which +formats a build writes, never which writer. For each format a build takes the first writer that is installed, from the first of these that names one: 1. the recipe's `export.priority` (a list for every format, or a map such as `{"parquet": ["rs", "py"]}`), 2. the catalog's `export_priority`, 3. `RAINCLOUD_EXPORT_PRIORITY` (e.g. `rs,py`), -4. the built-in order `py, rs, java, canonical` (`canonical` writes only the +4. the built-in order `py, rs, java, cpp, canonical` (`canonical` writes only the canonical Arrow IPC file, so it is the last resort for the `arrow` format). A writer that is not installed falls through to the next one in that order, so a @@ -44,8 +44,45 @@ export RAINCLOUD_SIDECAR_PARQUET_RS="$RAINCLOUD_TOOLS_ROOT/rust/bin/parquet-writ export RAINCLOUD_READER_PARQUET_RS="$RAINCLOUD_TOOLS_ROOT/rust/bin/parquet-read" export RAINCLOUD_SIDECAR_VORTEX_RS="$RAINCLOUD_TOOLS_ROOT/rust/bin/vortex-write" export RAINCLOUD_READER_VORTEX_RS="$RAINCLOUD_TOOLS_ROOT/rust/bin/vortex-read" +export RAINCLOUD_SIDECAR_ORC_RS="$RAINCLOUD_TOOLS_ROOT/rust/bin/orc-write" +export RAINCLOUD_READER_ORC_RS="$RAINCLOUD_TOOLS_ROOT/rust/bin/orc-read" +export RAINCLOUD_SIDECAR_AVRO_RS="$RAINCLOUD_TOOLS_ROOT/rust/bin/avro-write" +export RAINCLOUD_READER_AVRO_RS="$RAINCLOUD_TOOLS_ROOT/rust/bin/avro-read" ``` +The ORC lane (`orc@rs`) is orc-rust, pinned exactly in `sidecars/rust/Cargo.toml`. +It writes zstd with orc-rust's default stripe size, and panics on a type it does +not write (anything but signed integers, floats, strings, binary, booleans, +`date32` and timestamps); the panic is its report, and the build records ORC +unavailable for that dataset. The Python lane (`orc@py`, pyarrow's Apache ORC C++ +library) needs no sidecar. + +Avro has two lanes and no Python one (pyarrow reads and writes no Avro): `avro@rs`, +arrow-avro (released with arrow-rs, pinned with it), and `avro@java`, Arrow Java's own Avro +adapter over Apache Avro's Java implementation (the `avro-java` project). Both write a +zstandard object container file, one block per canonical batch (Rust) or Avro's own +block size (Java), with the same fixed sync marker, `raincloud-avro01`: each library +otherwise draws one at random, so a rebuild would change the file's sha256. Avro Java +takes the marker as an argument; arrow-avro offers no way to choose it, so the Rust lane +overwrites the marker it drew in place, after the header and after each block, leaving +every other byte arrow-avro's. Nothing converts a column for either library. Arrow +Java 19.0.0's adapter reads with its legacy mapping (the only one its public API +offers), which decodes a nullable Avro field into a sparse union of `null` and the +value type; every comparator reads such a union as the nullable column it spells. + +Where equality turns on representation rather than data, the three comparators share +one rule set, declared once in [`compare_cases/`](compare_cases/): pairs of Arrow files +and the verdict each must get, which every lane's tests read (`generate.py` writes +them). A union of exactly `null` and `T` is a nullable `T`; zoned timestamps compare by +instant whatever zone labels them (naive and zoned differ); an integer and a scale-0 +decimal holding the same values are equal. + +Both ORC lanes widen what ORC cannot hold before writing, always rather than by the +data's range, so a dataset's ORC schema never changes with its values: uint8 → int16, +uint16 → int32, uint32 → int64, uint64 → decimal(20, 0), and view types to their plain +types. The comparators read each back as the canonical's type. orc-rust 0.9.0 writes +no decimals, so a uint64 column is still unavailable in `orc@rs`. + Build Java distributions with JDK 17 and the pinned submodule. The parquet-hardwood project builds on Java 21, because Hardwood's jar targets it. If a JDK a project needs is not installed, Gradle's toolchain resolver provisions one, @@ -54,7 +91,7 @@ which may download it: ```bash git submodule update --init --recursive bash sidecars/java/gradlew -p sidecars/java :parquet-java:installDist \ - :parquet-hardwood:installDist :vortex-jni-reader:installDist + :parquet-hardwood:installDist :vortex-jni-reader:installDist :avro-java:installDist ``` Copy each complete directory from the corresponding project's `build/install/` @@ -69,6 +106,8 @@ these environment variables at the copied launchers: | `RAINCLOUD_READER_PARQUET_HARDWOOD` | `raincloud-export-parquet-hardwood` | `bin/raincloud-read-parquet-hardwood` | | `RAINCLOUD_SIDECAR_VORTEX_JNI` | `raincloud-read-vortex-jni` | `bin/raincloud-export-vortex-jni` | | `RAINCLOUD_READER_VORTEX_JNI` | `raincloud-read-vortex-jni` | `bin/raincloud-read-vortex-jni` | +| `RAINCLOUD_SIDECAR_AVRO_JAVA` | `raincloud-export-avro-java` | `bin/raincloud-export-avro-java` | +| `RAINCLOUD_READER_AVRO_JAVA` | `raincloud-export-avro-java` | `bin/raincloud-read-avro-java` | The `vortex-jni-reader` project holds the `vortex@jni` writer as well as its reader, so its one distribution carries both launchers. @@ -113,6 +152,97 @@ read. A recipe's `write.row_group_size_rows` wins over so the build passes that cap to them as `RAINCLOUD_ROW_GROUP_MAX_ROWS` in their environment. +The Parquet writers also share one set of write options +(`raincloud/pipeline/spec.py::ParquetOptions`, which documents each), which the build +passes to a sidecar in this form: + +| variable | value | from | +|---|---|---| +| `RAINCLOUD_PARQUET_COMPRESSION` | `zstd`, `snappy`, `gzip`, `lz4` (LZ4_RAW), `brotli` or `none` | the recipe's `write.compression` | +| `RAINCLOUD_PARQUET_STATISTICS` | `1` or `0` | the recipe's `write.statistics` | +| `RAINCLOUD_PARQUET_COMPRESSION_LEVEL` | a whole number in the codec's range: zstd 1-22, gzip 0-9, brotli 0-11 | the install, only when set | +| `RAINCLOUD_PARQUET_STATISTICS_COLUMNS` | N: statistics only for the first N leaf columns | the install, only when set | +| `RAINCLOUD_PARQUET_PAGE_INDEX` | `1` or `0`: a ColumnIndex and OffsetIndex for every column chunk, or neither | the install, only when set | +| `RAINCLOUD_PARQUET_PAGE_INDEX_COLUMNS` | N: page statistics only for the first N leaf columns, chunk statistics for all | the install, only when set | +| `RAINCLOUD_PARQUET_PAGE_BYTES` | data page size target | the install, only when set | +| `RAINCLOUD_PARQUET_PAGE_ROWS` | data page row limit | the install, only when set | +| `RAINCLOUD_PARQUET_DICTIONARY` | `1` or `0`: dictionary encoding, or PLAIN | the install, only when set | +| `RAINCLOUD_PARQUET_DICTIONARY_PAGE_BYTES` | dictionary page size limit | the install, only when set | +| `RAINCLOUD_PARQUET_PAGE_CHECKSUMS` | `1` or `0`: a CRC in every page header, or none | the install, only when set | + +An unset variable is the lane's library default (zstd and statistics on, for the first +two). Counts use the count grammar, `0` meaning no limit (every column, for the two +column counts). Switches read `1/true/yes/on` and `0/false/no/off` in any case, empty as +unset, in every lane. Settings that contradict each other or the recipe (a page index +with statistics off; `PAGE_INDEX=0` with `PAGE_INDEX_COLUMNS`; a level for snappy, lz4 +or none, or out of the codec's range) are refused for every lane before any writer runs. +A lane refuses, as a failed round-trip naming the variable, a setting its library +cannot honour: + +| | py (pyarrow 24) | rs (arrow-rs 59.2) | java (parquet-arrow-java 0.3.0) | hardwood (1.1.0.Beta1) | +|---|---|---|---|---| +| codec | all | all | not `brotli` | all | +| compression level | yes | yes | yes | no | +| statistics off | yes | yes | yes | no | +| statistics, first N columns | yes | yes | yes | no | +| page index on | yes | yes | yes | no | +| page index off | yes | yes | no, with statistics on | yes | +| page index, first N columns | no: all or none | yes | no | no | +| page bytes | yes | yes | yes | yes | +| page rows | yes | yes, checked every 1,024 values | yes | no | +| dictionary on / off | yes | yes | yes | yes (off is PLAIN) | +| dictionary page bytes | yes | yes | yes | no | +| page checksums on | yes | no | yes (its default) | yes (always) | +| page checksums off | yes (its default) | yes (always) | yes | no | + +Each library measures a page its own way, so the same `RAINCLOUD_PARQUET_PAGE_BYTES` +does not give identical pages in every lane. The java column's page index gaps are +parquet-java's: it writes a page index for every column that has statistics. + +The ORC, Avro and Vortex writers read their own settings the same way +(`spec.FORMAT_SETTINGS`): the build passes each set one, in canonical form, under +the name in `AGENTS.md`'s table, and a lane refuses what its library cannot do. With +none set, every lane writes exactly what it wrote before (the codec is zstd). + +| setting | honoured by | refused by | +|---|---|---| +| `RAINCLOUD_ORC_COMPRESSION` (zstd, snappy, zlib, lz4, none) | orc@py, orc@rs | | +| `RAINCLOUD_ORC_COMPRESSION_STRATEGY` | orc@py | orc@rs (no strategy) | +| `RAINCLOUD_ORC_STRIPE_BYTES` | orc@py, orc@rs (each measures a stripe its own way) | | +| `RAINCLOUD_ORC_COMPRESSION_BLOCK_BYTES` | orc@py (multiples of 64 KiB only), orc@rs | | +| `RAINCLOUD_AVRO_COMPRESSION` (zstd, deflate, snappy, bzip2, xz, none) | avro@rs, avro@java | | +| `RAINCLOUD_AVRO_COMPRESSION_LEVEL` | avro@java | avro@rs (arrow-avro has no level) | +| `RAINCLOUD_AVRO_BLOCK_BYTES` | avro@java (Avro's sync interval) | avro@rs (one block per batch) | +| `RAINCLOUD_VORTEX_COMPACT` | vortex@py, vortex@rs (BtrBlocks compact, within the session's editions) | vortex@jni (no write strategy) | +| `RAINCLOUD_VORTEX_ROW_BLOCK_ROWS`, `_DATA_BLOCK_BYTES` | vortex@rs | vortex@py, vortex@jni (no block settings) | + +## The Nimble lane + +Nimble has one implementation, Meta's C++ (facebookincubator/nimble), with no releases or +packages, so `nimble@cpp` is built from source. [`sidecars/nimble/build.sh`](nimble/build.sh) +builds the lane's two binaries, `raincloud-export-nimble-cpp` and `raincloud-read-nimble-cpp`, +inside a Nimble checkout at a pinned commit: upstream plus build fixes for a +current Linux toolchain (a host `liburing.h` that Folly mistakes for its own, GCC 16's +``, two Velox libraries a minimal build links but never declares), kept on the +`raincloud` branch of a Nimble fork, [mprammer/nimble](https://github.com/mprammer/nimble). +A cold build needs the network, about 4 GB on disk and several minutes, so CI does not +build it and records the lane as absent. + +```bash +git clone --branch raincloud --recurse-submodules https://github.com/mprammer/nimble /path/to/nimble +RAINCLOUD_NIMBLE_SRC=/path/to/nimble sidecars/nimble/build.sh /srv/raincloud-tools/nimble +export RAINCLOUD_SIDECAR_NIMBLE_CPP=/srv/raincloud-tools/nimble/bin/raincloud-export-nimble-cpp +export RAINCLOUD_READER_NIMBLE_CPP=/srv/raincloud-tools/nimble/bin/raincloud-read-nimble-cpp +``` + +The binaries are C++ over upstream Nimble's `VeloxWriter` (default options) and +`VeloxReader` (the file read as the type it records), and they link `nimble-ffi`, a +member of the Rust crate, which runs the sidecar contract as every Rust lane does: it +reads the canonical with arrow-rs, hands its batches to the writer in memory through the +Arrow C stream interface, takes the read-back the same way, compares and reports. Velox's +own Arrow bridge imports and exports the batches; nothing converts a column. The build +links both halves against the host's libzstd, so the binary carries one zstd. + ## How each lane judges a round trip All three comparators check column names and nested shape first, then values. A @@ -126,14 +256,17 @@ representation changes they can judge: | integer width or signedness | pass if the values are exact | pass if the values are exact | pass if exact (BigInteger) | | float width (incl. half) | pass only if reversible | pass only if reversible | compares decoded values bitwise, so the same pass/fail | | timestamp unit, same timezone | pass if the instant survives | pass if the instant survives | compares the instant: pass or fail | -| timestamp timezone changed or dropped | fail | fail | fail | +| timestamp timezone relabelled (both zoned) | pass if the instants match | pass if the instants match | pass if the instants match | +| timestamp timezone dropped or added | fail | fail | fail | | decimal precision/scale | reversible cast | reversible cast | skip (gap) | +| integer ↔ scale-0 decimal | pass if the values are exact | pass if the values are exact | pass if exact | | date32 / date64 | reversible cast | reversible cast | skip (gap) | | time32 / time64, duration unit | reversible cast | reversible cast | skip (gap) | | dictionary ↔ plain, top level | pass | pass | decoded to values, then compared | | dictionary inside a nested column | pass | pass | skip (gap) | | struct with duplicate child names | compared | compared | skip (gap) | -| union ↔ non-union | fail | fail | skip (gap) | +| union of `null` and `T` ↔ `T` | compared as `T` | compared as `T` | compared as `T` | +| any other union ↔ non-union | fail | fail | fail | | fixed-size ↔ variable binary/list | reversible cast | reversible cast | compared by value | A JVM `skip` means the JVM lane is unmeasured for that cell, never that the @@ -213,12 +346,12 @@ writes no `ARROW:schema` footer. What it cannot carry is reported, never guessed `"unsupported type"`, no file. SECOND times and date64 are written (as MILLIS and days) but read back in another unit, a comparator gap, so the round trip is unmeasured. A timezone other than UTC - cannot be kept: Parquet records only "adjusted to UTC", so the self-verify - fails as a timezone change. + is not kept (Parquet records only "adjusted to UTC"), but the instants are, and + zoned timestamps compare by instant. - The reader reports `INT96`, `INTERVAL`, a key-only map, a repeated field outside a `LIST` or `MAP`, and a layer layout it does not expect as a - comparator gap (`skip`). It bundles zstd, snappy and lz4 but not brotli, which - needs a per-platform native library; no raincloud writer uses it. + comparator gap (`skip`). It bundles zstd, snappy, lz4 and brotli (brotli4j, with + its native library for Linux and macOS on x86-64 and aarch64). - Hardwood 1.1.0.Beta1 fails to read a page header whose statistics are longer than its first 1 KiB read of the header: it raises "Malformed Parquet metadata" where it means to read further (fixed after the release by hardwood bdecd568, diff --git a/sidecars/compare_cases/cases.json b/sidecars/compare_cases/cases.json new file mode 100644 index 0000000..3b53c4f --- /dev/null +++ b/sidecars/compare_cases/cases.json @@ -0,0 +1,80 @@ +{ + "_about": "Comparison rules every lane shares; see generate.py. Each case is .got.arrow (column `c`, what a reader returned) and .expected.arrow (the canonical).", + "cases": [ + { + "name": "union-null-int-sparse", + "verdict": "equal" + }, + { + "name": "union-null-int-dense", + "verdict": "equal" + }, + { + "name": "union-int-null-order", + "verdict": "equal" + }, + { + "name": "union-null-string", + "verdict": "equal" + }, + { + "name": "union-null-int-values-differ", + "verdict": "differ" + }, + { + "name": "union-null-int-nulls-differ", + "verdict": "differ" + }, + { + "name": "union-inside-struct", + "verdict": "equal" + }, + { + "name": "union-two-members-is-not-nullable", + "verdict": "differ" + }, + { + "name": "tz-utc-vs-offset", + "verdict": "equal" + }, + { + "name": "tz-other-zone-same-instants", + "verdict": "equal" + }, + { + "name": "tz-etc-utc", + "verdict": "equal" + }, + { + "name": "tz-same-zone-values-differ", + "verdict": "differ" + }, + { + "name": "tz-naive-vs-zoned", + "verdict": "differ" + }, + { + "name": "decimal-scale0-holds-uint64", + "verdict": "equal", + "gap": true + }, + { + "name": "decimal-scale0-values-differ", + "verdict": "differ", + "gap": true + }, + { + "name": "decimal-scale2-is-not-an-integer", + "verdict": "differ", + "gap": true + }, + { + "name": "int-width", + "verdict": "equal" + }, + { + "name": "string-view", + "verdict": "equal" + } + ] +} diff --git a/sidecars/compare_cases/decimal-scale0-holds-uint64.expected.arrow b/sidecars/compare_cases/decimal-scale0-holds-uint64.expected.arrow new file mode 100644 index 0000000..e30d358 Binary files /dev/null and b/sidecars/compare_cases/decimal-scale0-holds-uint64.expected.arrow differ diff --git a/sidecars/compare_cases/decimal-scale0-holds-uint64.got.arrow b/sidecars/compare_cases/decimal-scale0-holds-uint64.got.arrow new file mode 100644 index 0000000..89dcf83 Binary files /dev/null and b/sidecars/compare_cases/decimal-scale0-holds-uint64.got.arrow differ diff --git a/sidecars/compare_cases/decimal-scale0-values-differ.expected.arrow b/sidecars/compare_cases/decimal-scale0-values-differ.expected.arrow new file mode 100644 index 0000000..89c67c2 Binary files /dev/null and b/sidecars/compare_cases/decimal-scale0-values-differ.expected.arrow differ diff --git a/sidecars/compare_cases/decimal-scale0-values-differ.got.arrow b/sidecars/compare_cases/decimal-scale0-values-differ.got.arrow new file mode 100644 index 0000000..678b3fa Binary files /dev/null and b/sidecars/compare_cases/decimal-scale0-values-differ.got.arrow differ diff --git a/sidecars/compare_cases/decimal-scale2-is-not-an-integer.expected.arrow b/sidecars/compare_cases/decimal-scale2-is-not-an-integer.expected.arrow new file mode 100644 index 0000000..a3a5e27 Binary files /dev/null and b/sidecars/compare_cases/decimal-scale2-is-not-an-integer.expected.arrow differ diff --git a/sidecars/compare_cases/decimal-scale2-is-not-an-integer.got.arrow b/sidecars/compare_cases/decimal-scale2-is-not-an-integer.got.arrow new file mode 100644 index 0000000..70516bf Binary files /dev/null and b/sidecars/compare_cases/decimal-scale2-is-not-an-integer.got.arrow differ diff --git a/sidecars/compare_cases/generate.py b/sidecars/compare_cases/generate.py new file mode 100644 index 0000000..aaa3da9 --- /dev/null +++ b/sidecars/compare_cases/generate.py @@ -0,0 +1,119 @@ +# SPDX-FileCopyrightText: 2026 Raincloud Maintainers +# SPDX-License-Identifier: Apache-2.0 +"""Write the comparison cases every lane's comparator is held to. + +The comparators (Python `raincloud/pipeline/export/compare.py`, Rust +`sidecars/rust/src/lib.rs`, JVM `LogicalCompare`) decide when a file read back +holds the canonical's data. Where that decision turns on representation rather +than data -- a zone label, a union spelling of nullability, an integer held as a +scale-0 decimal -- the rule is declared here once, as pairs of Arrow IPC files +and the verdict each pair must get, and every lane's tests read the same files. + +`verdict` is `equal` or `differ`. `gap: true` lets a comparator that cannot +represent the type report a gap instead (the JVM comparator has no decimals); +no comparator may ever give the opposite verdict. + +Run from the repository root to regenerate `cases.json` and `*.arrow`: + python sidecars/compare_cases/generate.py +""" +from __future__ import annotations + +import datetime +import decimal +import json +from pathlib import Path + +import pyarrow as pa + +HERE = Path(__file__).resolve().parent +UTC = datetime.timezone.utc + + +def _union(values, member: pa.DataType, *, mode: str = "sparse", null_first: bool = True) -> pa.Array: + """A union of exactly `null` and `member`: a None selects the null member.""" + null_code, member_code = (0, 1) if null_first else (1, 0) + codes = pa.array([null_code if v is None else member_code for v in values], pa.int8()) + member_values = pa.array(values, member) + if mode == "sparse": + children = [pa.nulls(len(values)), member_values] + if not null_first: + children.reverse() + return pa.UnionArray.from_sparse(codes, children, ["null", "value"] if null_first else ["value", "null"], + [0, 1]) + present = [v for v in values if v is not None] + offsets, n_null, n_member = [], 0, 0 + for v in values: + if v is None: + offsets.append(n_null) + n_null += 1 + else: + offsets.append(n_member) + n_member += 1 + children = [pa.nulls(n_null), pa.array(present, member)] + if not null_first: + children.reverse() + return pa.UnionArray.from_dense(codes, pa.array(offsets, pa.int32()), children, + ["null", "value"] if null_first else ["value", "null"], [0, 1]) + + +def _ts(values, tz): + return pa.array([None if v is None else datetime.datetime(2024, 1, 1, v, tzinfo=UTC) for v in values], + pa.timestamp("us", tz)) + + +def cases(): + ints = [1, None, 3] + yield "union-null-int-sparse", "equal", _union(ints, pa.int32()), pa.array(ints, pa.int32()) + yield "union-null-int-dense", "equal", _union(ints, pa.int32(), mode="dense"), pa.array(ints, pa.int32()) + yield "union-int-null-order", "equal", _union(ints, pa.int32(), null_first=False), pa.array(ints, pa.int32()) + yield "union-null-string", "equal", _union(["a", None, "c"], pa.string()), pa.array(["a", None, "c"]) + yield "union-null-int-values-differ", "differ", _union([1, None, 4], pa.int32()), pa.array(ints, pa.int32()) + yield "union-null-int-nulls-differ", "differ", _union([1, 2, 3], pa.int32()), pa.array(ints, pa.int32()) + nested = _union(ints, pa.int64()) + yield ("union-inside-struct", "equal", + pa.StructArray.from_arrays([nested], ["x"]), pa.StructArray.from_arrays([pa.array(ints, pa.int64())], ["x"])) + two = pa.UnionArray.from_sparse(pa.array([0, 1, 0], pa.int8()), + [pa.array([1, 2, 3], pa.int32()), pa.array(["a", "b", "c"])], ["i", "s"], [0, 1]) + yield "union-two-members-is-not-nullable", "differ", two, pa.array([1, 2, 3], pa.int32()) + + hours = [1, None, 3] + yield "tz-utc-vs-offset", "equal", _ts(hours, "+00:00"), _ts(hours, "UTC") + yield "tz-other-zone-same-instants", "equal", _ts(hours, "America/New_York"), _ts(hours, "UTC") + yield "tz-etc-utc", "equal", _ts(hours, "Etc/UTC"), _ts(hours, "UTC") + yield "tz-same-zone-values-differ", "differ", _ts([1, None, 4], "+00:00"), _ts(hours, "UTC") + yield "tz-naive-vs-zoned", "differ", _ts(hours, None), _ts(hours, "UTC") + + big = [0, None, 2**64 - 1] + as_decimal = pa.array([None if v is None else decimal.Decimal(v) for v in big], pa.decimal128(20, 0)) + yield "decimal-scale0-holds-uint64", "equal", as_decimal, pa.array(big, pa.uint64()), True + yield ("decimal-scale0-values-differ", "differ", + pa.array([decimal.Decimal(1), None, decimal.Decimal(2)], pa.decimal128(20, 0)), + pa.array([0, None, 2], pa.uint64()), True) + yield ("decimal-scale2-is-not-an-integer", "differ", + pa.array([decimal.Decimal("0.00"), None, decimal.Decimal("2.00")], pa.decimal128(20, 2)), + pa.array([0, None, 2], pa.int64()), True) + + yield "int-width", "equal", pa.array(ints, pa.int64()), pa.array(ints, pa.int8()) + yield "string-view", "equal", pa.array(["a", None], pa.string_view()), pa.array(["a", None]) + + +def _write(path: Path, array: pa.Array) -> None: + table = pa.table({"c": array}) + with pa.OSFile(str(path), "wb") as sink, pa.ipc.new_file(sink, table.schema) as writer: + writer.write_table(table) + + +def main(): + listed = [] + for name, verdict, got, expected, *gap in cases(): + _write(HERE / f"{name}.got.arrow", got) + _write(HERE / f"{name}.expected.arrow", expected) + listed.append({"name": name, "verdict": verdict, **({"gap": True} if gap and gap[0] else {})}) + (HERE / "cases.json").write_text(json.dumps({ + "_about": "Comparison rules every lane shares; see generate.py. Each case is .got.arrow " + "(column `c`, what a reader returned) and .expected.arrow (the canonical).", + "cases": listed}, indent=1) + "\n") + + +if __name__ == "__main__": + main() diff --git a/sidecars/compare_cases/int-width.expected.arrow b/sidecars/compare_cases/int-width.expected.arrow new file mode 100644 index 0000000..28b3754 Binary files /dev/null and b/sidecars/compare_cases/int-width.expected.arrow differ diff --git a/sidecars/compare_cases/int-width.got.arrow b/sidecars/compare_cases/int-width.got.arrow new file mode 100644 index 0000000..1d6472a Binary files /dev/null and b/sidecars/compare_cases/int-width.got.arrow differ diff --git a/sidecars/compare_cases/string-view.expected.arrow b/sidecars/compare_cases/string-view.expected.arrow new file mode 100644 index 0000000..227dc6b Binary files /dev/null and b/sidecars/compare_cases/string-view.expected.arrow differ diff --git a/sidecars/compare_cases/string-view.got.arrow b/sidecars/compare_cases/string-view.got.arrow new file mode 100644 index 0000000..b86b9ff Binary files /dev/null and b/sidecars/compare_cases/string-view.got.arrow differ diff --git a/sidecars/compare_cases/tz-etc-utc.expected.arrow b/sidecars/compare_cases/tz-etc-utc.expected.arrow new file mode 100644 index 0000000..dcd1217 Binary files /dev/null and b/sidecars/compare_cases/tz-etc-utc.expected.arrow differ diff --git a/sidecars/compare_cases/tz-etc-utc.got.arrow b/sidecars/compare_cases/tz-etc-utc.got.arrow new file mode 100644 index 0000000..8fa8cd3 Binary files /dev/null and b/sidecars/compare_cases/tz-etc-utc.got.arrow differ diff --git a/sidecars/compare_cases/tz-naive-vs-zoned.expected.arrow b/sidecars/compare_cases/tz-naive-vs-zoned.expected.arrow new file mode 100644 index 0000000..dcd1217 Binary files /dev/null and b/sidecars/compare_cases/tz-naive-vs-zoned.expected.arrow differ diff --git a/sidecars/compare_cases/tz-naive-vs-zoned.got.arrow b/sidecars/compare_cases/tz-naive-vs-zoned.got.arrow new file mode 100644 index 0000000..33044c3 Binary files /dev/null and b/sidecars/compare_cases/tz-naive-vs-zoned.got.arrow differ diff --git a/sidecars/compare_cases/tz-other-zone-same-instants.expected.arrow b/sidecars/compare_cases/tz-other-zone-same-instants.expected.arrow new file mode 100644 index 0000000..dcd1217 Binary files /dev/null and b/sidecars/compare_cases/tz-other-zone-same-instants.expected.arrow differ diff --git a/sidecars/compare_cases/tz-other-zone-same-instants.got.arrow b/sidecars/compare_cases/tz-other-zone-same-instants.got.arrow new file mode 100644 index 0000000..df2d6f4 Binary files /dev/null and b/sidecars/compare_cases/tz-other-zone-same-instants.got.arrow differ diff --git a/sidecars/compare_cases/tz-same-zone-values-differ.expected.arrow b/sidecars/compare_cases/tz-same-zone-values-differ.expected.arrow new file mode 100644 index 0000000..dcd1217 Binary files /dev/null and b/sidecars/compare_cases/tz-same-zone-values-differ.expected.arrow differ diff --git a/sidecars/compare_cases/tz-same-zone-values-differ.got.arrow b/sidecars/compare_cases/tz-same-zone-values-differ.got.arrow new file mode 100644 index 0000000..e14170d Binary files /dev/null and b/sidecars/compare_cases/tz-same-zone-values-differ.got.arrow differ diff --git a/sidecars/compare_cases/tz-utc-vs-offset.expected.arrow b/sidecars/compare_cases/tz-utc-vs-offset.expected.arrow new file mode 100644 index 0000000..dcd1217 Binary files /dev/null and b/sidecars/compare_cases/tz-utc-vs-offset.expected.arrow differ diff --git a/sidecars/compare_cases/tz-utc-vs-offset.got.arrow b/sidecars/compare_cases/tz-utc-vs-offset.got.arrow new file mode 100644 index 0000000..a12d717 Binary files /dev/null and b/sidecars/compare_cases/tz-utc-vs-offset.got.arrow differ diff --git a/sidecars/compare_cases/union-inside-struct.expected.arrow b/sidecars/compare_cases/union-inside-struct.expected.arrow new file mode 100644 index 0000000..2943679 Binary files /dev/null and b/sidecars/compare_cases/union-inside-struct.expected.arrow differ diff --git a/sidecars/compare_cases/union-inside-struct.got.arrow b/sidecars/compare_cases/union-inside-struct.got.arrow new file mode 100644 index 0000000..bdc024b Binary files /dev/null and b/sidecars/compare_cases/union-inside-struct.got.arrow differ diff --git a/sidecars/compare_cases/union-int-null-order.expected.arrow b/sidecars/compare_cases/union-int-null-order.expected.arrow new file mode 100644 index 0000000..b0b65be Binary files /dev/null and b/sidecars/compare_cases/union-int-null-order.expected.arrow differ diff --git a/sidecars/compare_cases/union-int-null-order.got.arrow b/sidecars/compare_cases/union-int-null-order.got.arrow new file mode 100644 index 0000000..08fd883 Binary files /dev/null and b/sidecars/compare_cases/union-int-null-order.got.arrow differ diff --git a/sidecars/compare_cases/union-null-int-dense.expected.arrow b/sidecars/compare_cases/union-null-int-dense.expected.arrow new file mode 100644 index 0000000..b0b65be Binary files /dev/null and b/sidecars/compare_cases/union-null-int-dense.expected.arrow differ diff --git a/sidecars/compare_cases/union-null-int-dense.got.arrow b/sidecars/compare_cases/union-null-int-dense.got.arrow new file mode 100644 index 0000000..7f3f15b Binary files /dev/null and b/sidecars/compare_cases/union-null-int-dense.got.arrow differ diff --git a/sidecars/compare_cases/union-null-int-nulls-differ.expected.arrow b/sidecars/compare_cases/union-null-int-nulls-differ.expected.arrow new file mode 100644 index 0000000..b0b65be Binary files /dev/null and b/sidecars/compare_cases/union-null-int-nulls-differ.expected.arrow differ diff --git a/sidecars/compare_cases/union-null-int-nulls-differ.got.arrow b/sidecars/compare_cases/union-null-int-nulls-differ.got.arrow new file mode 100644 index 0000000..162d1e3 Binary files /dev/null and b/sidecars/compare_cases/union-null-int-nulls-differ.got.arrow differ diff --git a/sidecars/compare_cases/union-null-int-sparse.expected.arrow b/sidecars/compare_cases/union-null-int-sparse.expected.arrow new file mode 100644 index 0000000..b0b65be Binary files /dev/null and b/sidecars/compare_cases/union-null-int-sparse.expected.arrow differ diff --git a/sidecars/compare_cases/union-null-int-sparse.got.arrow b/sidecars/compare_cases/union-null-int-sparse.got.arrow new file mode 100644 index 0000000..4102711 Binary files /dev/null and b/sidecars/compare_cases/union-null-int-sparse.got.arrow differ diff --git a/sidecars/compare_cases/union-null-int-values-differ.expected.arrow b/sidecars/compare_cases/union-null-int-values-differ.expected.arrow new file mode 100644 index 0000000..b0b65be Binary files /dev/null and b/sidecars/compare_cases/union-null-int-values-differ.expected.arrow differ diff --git a/sidecars/compare_cases/union-null-int-values-differ.got.arrow b/sidecars/compare_cases/union-null-int-values-differ.got.arrow new file mode 100644 index 0000000..80c5c8d Binary files /dev/null and b/sidecars/compare_cases/union-null-int-values-differ.got.arrow differ diff --git a/sidecars/compare_cases/union-null-string.expected.arrow b/sidecars/compare_cases/union-null-string.expected.arrow new file mode 100644 index 0000000..4f4589a Binary files /dev/null and b/sidecars/compare_cases/union-null-string.expected.arrow differ diff --git a/sidecars/compare_cases/union-null-string.got.arrow b/sidecars/compare_cases/union-null-string.got.arrow new file mode 100644 index 0000000..e294a51 Binary files /dev/null and b/sidecars/compare_cases/union-null-string.got.arrow differ diff --git a/sidecars/compare_cases/union-two-members-is-not-nullable.expected.arrow b/sidecars/compare_cases/union-two-members-is-not-nullable.expected.arrow new file mode 100644 index 0000000..4a9592d Binary files /dev/null and b/sidecars/compare_cases/union-two-members-is-not-nullable.expected.arrow differ diff --git a/sidecars/compare_cases/union-two-members-is-not-nullable.got.arrow b/sidecars/compare_cases/union-two-members-is-not-nullable.got.arrow new file mode 100644 index 0000000..cbe1364 Binary files /dev/null and b/sidecars/compare_cases/union-two-members-is-not-nullable.got.arrow differ diff --git a/sidecars/java/avro-java/build.gradle.kts b/sidecars/java/avro-java/build.gradle.kts new file mode 100644 index 0000000..d5f8d24 --- /dev/null +++ b/sidecars/java/avro-java/build.gradle.kts @@ -0,0 +1,81 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +// +// `avro@java` WRITE + READ conformance lanes via Arrow Java's own Avro adapter +// (org.apache.arrow:arrow-avro, the same arrowVersion as every JVM lane) over Apache +// Avro's Java implementation. ONE subproject, TWO launch scripts sharing one lib/: +// raincloud-export-avro-java (write the canonical as an Avro object container file, +// then self-verify by reading it back) and raincloud-read-avro-java (read an Avro file, +// compare LOGICALLY to the canonical). +plugins { + application +} + +repositories { + mavenCentral() +} + +val arrowVersion: String by project +val junitVersion: String by project +val slf4jVersion: String by project +val zstdJniVersion: String by project +val snappyJavaVersion: String by project +val xzVersion: String by project + +dependencies { + implementation(project(":conformance-common")) + + // Arrow Java's Avro adapter; it brings Apache Avro (1.12.1 with Arrow 19.0.0). + implementation("org.apache.arrow:arrow-avro:$arrowVersion") + // Avro's zstandard, snappy and xz codecs, which Avro declares optional (deflate is the + // JDK's, bzip2 commons-compress, a dependency of Avro's own). + runtimeOnly("com.github.luben:zstd-jni:$zstdJniVersion") + runtimeOnly("org.xerial.snappy:snappy-java:$snappyJavaVersion") + runtimeOnly("org.tukaani:xz:$xzVersion") + + // Off-heap allocator impl + silence SLF4J. + runtimeOnly("org.apache.arrow:arrow-memory-netty:$arrowVersion") + runtimeOnly("org.slf4j:slf4j-nop:$slf4jVersion") + + testImplementation(platform("org.junit:junit-bom:$junitVersion")) + testImplementation("org.junit.jupiter:junit-jupiter") + testRuntimeOnly("org.junit.platform:junit-platform-launcher") +} + +java { + toolchain { languageVersion.set(JavaLanguageVersion.of(17)) } +} + +val sidecarJvmArgs = listOf( + "--add-opens=java.base/java.nio=ALL-UNNAMED", + "--enable-native-access=ALL-UNNAMED", +) + +tasks.test { + useJUnitPlatform() + jvmArgs(sidecarJvmArgs) +} + +application { + mainClass.set("dev.raincloud.sidecar.avrojava.ConformanceWriter") + applicationName = "raincloud-export-avro-java" + applicationDefaultJvmArgs = sidecarJvmArgs +} + +// Second entry point: the READ lane, added into the same distribution's bin/. +val readerStartScripts = tasks.register("readerStartScripts") { + dependsOn(tasks.named("jar")) + mainClass.set("dev.raincloud.sidecar.avrojava.ConformanceReader") + applicationName = "raincloud-read-avro-java" + outputDir = layout.buildDirectory.dir("scriptsReader").get().asFile + classpath = files(tasks.named("jar")) + configurations.runtimeClasspath.get() + defaultJvmOpts = sidecarJvmArgs +} + +distributions { + named("main") { + contents { + from(readerStartScripts) { into("bin") } + } + } +} diff --git a/sidecars/java/avro-java/src/main/java/dev/raincloud/sidecar/avrojava/AvroArrowIo.java b/sidecars/java/avro-java/src/main/java/dev/raincloud/sidecar/avrojava/AvroArrowIo.java new file mode 100644 index 0000000..d0c54f3 --- /dev/null +++ b/sidecars/java/avro-java/src/main/java/dev/raincloud/sidecar/avrojava/AvroArrowIo.java @@ -0,0 +1,228 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +package dev.raincloud.sidecar.avrojava; + +import dev.raincloud.sidecar.common.BatchSource; +import dev.raincloud.sidecar.common.MaterializedTable; +import dev.raincloud.sidecar.common.WriteSettings; +import java.io.BufferedInputStream; +import java.io.BufferedOutputStream; +import java.io.ByteArrayOutputStream; +import java.io.IOException; +import java.io.InputStream; +import java.io.OutputStream; +import java.io.SequenceInputStream; +import java.nio.ByteBuffer; +import java.nio.channels.SeekableByteChannel; +import java.nio.charset.StandardCharsets; +import java.nio.file.Files; +import java.nio.file.Path; +import java.nio.file.StandardOpenOption; +import java.util.Enumeration; +import org.apache.arrow.adapter.avro.ArrowToAvroUtils; +import org.apache.arrow.adapter.avro.AvroToArrow; +import org.apache.arrow.adapter.avro.AvroToArrowConfig; +import org.apache.arrow.adapter.avro.AvroToArrowConfigBuilder; +import org.apache.arrow.adapter.avro.AvroToArrowVectorIterator; +import org.apache.arrow.adapter.avro.producers.CompositeAvroProducer; +import org.apache.arrow.compression.CommonsCompressionFactory; +import org.apache.arrow.memory.BufferAllocator; +import org.apache.arrow.vector.VectorSchemaRoot; +import org.apache.arrow.vector.dictionary.DictionaryProvider; +import org.apache.arrow.vector.ipc.ArrowFileReader; +import org.apache.arrow.vector.types.pojo.Schema; +import org.apache.avro.file.CodecFactory; +import org.apache.avro.file.DataFileStream; +import org.apache.avro.file.DataFileWriter; +import org.apache.avro.generic.GenericDatumReader; +import org.apache.avro.generic.GenericDatumWriter; +import org.apache.avro.io.BinaryEncoder; +import org.apache.avro.io.DecoderFactory; +import org.apache.avro.io.EncoderFactory; + +/** + * Arrow ⇆ Avro object container files through Arrow Java's Avro adapter and Apache + * Avro's Java implementation. + * + *

Writing: the adapter maps the canonical's schema to an Avro schema and encodes each + * row ({@link ArrowToAvroUtils}); Avro's {@link DataFileWriter} frames the rows into a + * zstandard container file. Reading: Avro's {@link DataFileStream} yields each block's + * decoded bytes and the adapter decodes them into Arrow vectors ({@link AvroToArrow}). + * Nothing here converts a column: a type the adapter does not map is its error. + */ +final class AvroArrowIo { + private AvroArrowIo() {} + + /** + * The sync marker of every Avro file raincloud writes. An object container file + * separates its blocks with a 16-byte marker the writer chooses, and Avro draws it at + * random by default, so the same canonical would give a file with a different sha256 on + * every build. The Rust lane writes the same marker. + */ + static final byte[] SYNC_MARKER = "raincloud-avro01".getBytes(StandardCharsets.US_ASCII); + + static final String AVRO_COMPRESSION = "RAINCLOUD_AVRO_COMPRESSION"; + static final String AVRO_COMPRESSION_LEVEL = "RAINCLOUD_AVRO_COMPRESSION_LEVEL"; + static final String AVRO_BLOCK_BYTES = "RAINCLOUD_AVRO_BLOCK_BYTES"; + + /** + * The Avro codec the write settings ask for ({@code spec.FORMAT_SETTINGS["avro"]}): unset, zstandard + * at Avro's default level, as raincloud has always written; a level only for zstandard, + * deflate and xz, the codecs that take one. + */ + static CodecFactory codec(java.util.function.Function env) { + String codec = WriteSettings.choice(AVRO_COMPRESSION, env.apply(AVRO_COMPRESSION), + "zstd", "deflate", "snappy", "bzip2", "xz", "none"); + Integer level = WriteSettings.level(AVRO_COMPRESSION_LEVEL, env.apply(AVRO_COMPRESSION_LEVEL)); + switch (codec == null ? "zstd" : codec) { + case "zstd": + return CodecFactory.zstandardCodec(level == null ? CodecFactory.DEFAULT_ZSTANDARD_LEVEL : level); + case "deflate": + return CodecFactory.deflateCodec(level == null ? CodecFactory.DEFAULT_DEFLATE_LEVEL : level); + case "xz": + return CodecFactory.xzCodec(level == null ? CodecFactory.DEFAULT_XZ_LEVEL : level); + default: + if (level != null) { + throw new IllegalArgumentException(AVRO_COMPRESSION_LEVEL + "=" + level + ": " + codec + + " takes no compression level"); + } + return "snappy".equals(codec) ? CodecFactory.snappyCodec() + : "bzip2".equals(codec) ? CodecFactory.bzip2Codec() : CodecFactory.nullCodec(); + } + } + + /** Write the canonical {@code input} to {@code output}; on failure no {@code output} remains. */ + static void writeAvro(Path input, Path output, BufferAllocator allocator) throws Exception { + try { + write(input, output, allocator); + } catch (Exception | Error e) { + Files.deleteIfExists(output); + throw e; + } + } + + private static void write(Path input, Path output, BufferAllocator allocator) throws Exception { + try (SeekableByteChannel channel = Files.newByteChannel(input, StandardOpenOption.READ); + ArrowFileReader reader = + new ArrowFileReader(channel, allocator, CommonsCompressionFactory.INSTANCE)) { + VectorSchemaRoot root = reader.getVectorSchemaRoot(); + // The reader is the dictionary provider: the adapter writes a dictionary column + // as its values (or an enum), by its own rules. + DictionaryProvider dictionaries = reader; + org.apache.avro.Schema schema = + ArrowToAvroUtils.createAvroSchema(root.getSchema().getFields(), dictionaries); + try (OutputStream out = new BufferedOutputStream(Files.newOutputStream(output)); + DataFileWriter writer = new DataFileWriter<>(new GenericDatumWriter<>(schema))) { + writer.setCodec(codec(System::getenv)); + Integer blockBytes = WriteSettings.count(AVRO_BLOCK_BYTES, System.getenv(AVRO_BLOCK_BYTES)); + if (blockBytes != null) { + // Avro's sync interval: a block closes once it holds about this many bytes. + writer.setSyncInterval(blockBytes); + } + writer.create(schema, out, SYNC_MARKER); + ByteArrayOutputStream row = new ByteArrayOutputStream(); + BinaryEncoder encoder = EncoderFactory.get().directBinaryEncoder(row, null); + while (reader.loadNextBatch()) { + // A producer per batch, over that batch's vectors. Arrow Java 19.0.0's + // CompositeAvroProducer.resetProducerVectors, meant for this, binds every + // producer to the root's first vector (its loop never advances its index). + CompositeAvroProducer producer = + ArrowToAvroUtils.createCompositeProducer(root.getFieldVectors(), dictionaries); + for (int i = 0; i < root.getRowCount(); i++) { + row.reset(); + producer.produce(encoder); + encoder.flush(); + writer.appendEncoded(ByteBuffer.wrap(row.toByteArray())); + } + } + } + } + } + + /** The batches of the Avro file at {@code input}, decoded by the adapter. */ + static BatchSource openAvro(Path input, BufferAllocator allocator) throws IOException { + InputStream in = new BufferedInputStream(Files.newInputStream(input)); + DataFileStream file; + try { + file = new DataFileStream<>(in, new GenericDatumReader<>()); + } catch (IOException | RuntimeException e) { + in.close(); + throw e; + } + org.apache.avro.Schema avroSchema = file.getSchema(); + DictionaryProvider.MapDictionaryProvider dictionaries = new DictionaryProvider.MapDictionaryProvider(); + AvroToArrowConfig config = + new AvroToArrowConfigBuilder(allocator).setProvider(dictionaries).build(); + // The blocks' decoded bytes, one after another: the rows, as the adapter reads them. + Enumeration blocks = new Enumeration<>() { + @Override + public boolean hasMoreElements() { + return file.hasNext(); + } + + @Override + public InputStream nextElement() { + try { + ByteBuffer block = file.nextBlock(); + return new java.io.ByteArrayInputStream( + block.array(), block.arrayOffset() + block.position(), block.remaining()); + } catch (IOException e) { + throw new java.io.UncheckedIOException(e); + } + } + }; + AvroToArrowVectorIterator rows; + VectorSchemaRoot first; + try { + rows = AvroToArrow.avroToArrowIterator(avroSchema, + DecoderFactory.get().binaryDecoder(new SequenceInputStream(blocks), null), config); + first = rows.hasNext() ? rows.next() : null; + } catch (IOException | RuntimeException e) { + file.close(); + throw e; + } + // The schema of the vectors the adapter decodes into, which is what it reads the file + // as. With no rows there are none, and its schema mapping says. + Schema schema = first != null ? first.getSchema() : AvroToArrow.avroToAvroSchema(avroSchema, config); + return new BatchSource() { + private VectorSchemaRoot pending = first; + + @Override + public MaterializedTable empty() { + MaterializedTable.Builder builder = new MaterializedTable.Builder(); + builder.initialize(schema, dictionaries); + return builder.build(); + } + + @Override + public MaterializedTable next() { + VectorSchemaRoot next = pending; + pending = null; + if (next == null) { + if (!rows.hasNext()) { + return null; + } + next = rows.next(); + } + try (VectorSchemaRoot batch = next) { + MaterializedTable.Builder builder = new MaterializedTable.Builder(); + builder.initialize(schema, dictionaries); + builder.appendBatch(batch, dictionaries); + return builder.build(); + } + } + + @Override + public void close() throws IOException { + try { + if (pending != null) { + pending.close(); + } + rows.close(); + } finally { + file.close(); + } + } + }; + } +} diff --git a/sidecars/java/avro-java/src/main/java/dev/raincloud/sidecar/avrojava/ConformanceReader.java b/sidecars/java/avro-java/src/main/java/dev/raincloud/sidecar/avrojava/ConformanceReader.java new file mode 100644 index 0000000..1502e8c --- /dev/null +++ b/sidecars/java/avro-java/src/main/java/dev/raincloud/sidecar/avrojava/ConformanceReader.java @@ -0,0 +1,22 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +package dev.raincloud.sidecar.avrojava; + +import dev.raincloud.sidecar.common.ReaderMain; + +/** + * {@code avro@java} READ-conformance sidecar (see {@link ReaderMain} for the CLI contract): + * + *
{@code
+ *   raincloud-read-avro-java --input  --canonical  --report 
+ * }
+ * + * Reads an Avro object container file through Apache Avro's Java implementation and Arrow + * Java's Avro adapter ({@link AvroArrowIo#openAvro}) and compares LOGICALLY to the canonical. + */ +public final class ConformanceReader { + public static void main(String[] args) { + ReaderMain.run(ConformanceWriter.CELL, "artifact.avro", args, AvroArrowIo::openAvro, + t -> false); + } +} diff --git a/sidecars/java/avro-java/src/main/java/dev/raincloud/sidecar/avrojava/ConformanceWriter.java b/sidecars/java/avro-java/src/main/java/dev/raincloud/sidecar/avrojava/ConformanceWriter.java new file mode 100644 index 0000000..dae4fbb --- /dev/null +++ b/sidecars/java/avro-java/src/main/java/dev/raincloud/sidecar/avrojava/ConformanceWriter.java @@ -0,0 +1,42 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +package dev.raincloud.sidecar.avrojava; + +import dev.raincloud.sidecar.common.BatchSource; +import dev.raincloud.sidecar.common.CanonicalReader; +import dev.raincloud.sidecar.common.LogicalCompare; +import dev.raincloud.sidecar.common.Verdict; +import dev.raincloud.sidecar.common.WriterMain; +import java.nio.file.Path; +import org.apache.arrow.memory.BufferAllocator; + +/** + * {@code avro@java} WRITE-conformance sidecar (see {@link WriterMain} for the CLI contract + * and the report): + * + *
{@code
+ *   raincloud-export-avro-java --input  --output  --report 
+ * }
+ * + * Writes the canonical as a zstandard Avro object container file through Arrow Java's Avro + * adapter ({@link AvroArrowIo#writeAvro}), then self-verifies by reading it back through the + * same adapter. Avro has no VARIANT type, so a VARIANT column is never kept. A type the + * adapter refuses is the implementation's failure, measured, never a comparator gap: the + * adapter is what this lane measures. + */ +public final class ConformanceWriter { + static final String CELL = "avro@java"; + + private static Verdict selfVerify(Path input, Path output, BufferAllocator allocator) throws Exception { + try (BatchSource expected = CanonicalReader.open(input, allocator); + BatchSource got = AvroArrowIo.openAvro(output, allocator)) { + return LogicalCompare.compare(CELL, got, expected); + } + } + + public static void main(String[] args) { + WriterMain.run(CELL, (output, variants, allocator) -> "Avro has no VARIANT type", args, + AvroArrowIo::writeAvro, ConformanceWriter::selfVerify, + t -> false); + } +} diff --git a/sidecars/java/avro-java/src/test/java/dev/raincloud/sidecar/avrojava/AvroArrowIoTest.java b/sidecars/java/avro-java/src/test/java/dev/raincloud/sidecar/avrojava/AvroArrowIoTest.java new file mode 100644 index 0000000..c9fd389 --- /dev/null +++ b/sidecars/java/avro-java/src/test/java/dev/raincloud/sidecar/avrojava/AvroArrowIoTest.java @@ -0,0 +1,91 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +package dev.raincloud.sidecar.avrojava; + +import static org.junit.jupiter.api.Assertions.assertArrayEquals; +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertNull; + +import dev.raincloud.sidecar.common.BatchSource; +import dev.raincloud.sidecar.common.MaterializedTable; +import java.nio.channels.FileChannel; +import java.nio.file.Files; +import java.nio.file.Path; +import java.nio.file.StandardOpenOption; +import java.util.List; +import org.apache.arrow.memory.BufferAllocator; +import org.apache.arrow.memory.RootAllocator; +import org.apache.arrow.vector.BigIntVector; +import org.apache.arrow.vector.VectorSchemaRoot; +import org.apache.arrow.vector.ipc.ArrowFileWriter; +import org.apache.arrow.vector.types.pojo.ArrowType; +import org.apache.arrow.vector.types.pojo.Field; +import org.apache.arrow.vector.types.pojo.FieldType; +import org.apache.arrow.vector.types.pojo.Schema; +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.io.TempDir; + +class AvroArrowIoTest { + @TempDir + Path dir; + + /** A canonical of one non-nullable bigint column, in two batches. */ + private Path canonical(BufferAllocator allocator) throws Exception { + Path path = dir.resolve("source.arrow"); + Schema schema = new Schema(List.of(new Field("x", FieldType.notNullable(new ArrowType.Int(64, true)), null))); + try (VectorSchemaRoot root = VectorSchemaRoot.create(schema, allocator); + FileChannel channel = FileChannel.open(path, StandardOpenOption.CREATE, StandardOpenOption.WRITE); + ArrowFileWriter writer = new ArrowFileWriter(root, null, channel)) { + writer.start(); + BigIntVector x = (BigIntVector) root.getVector("x"); + for (int batch = 0; batch < 2; batch++) { + x.allocateNew(3); + for (int i = 0; i < 3; i++) { + x.set(i, batch * 3 + i); + } + root.setRowCount(3); + writer.writeBatch(); + } + writer.end(); + } + return path; + } + + @Test + void writesTheSameBytesWithTheFixedMarkerAndReadsThemBack() throws Exception { + try (BufferAllocator allocator = new RootAllocator()) { + Path source = canonical(allocator); + Path first = dir.resolve("first.avro"); + Path second = dir.resolve("second.avro"); + AvroArrowIo.writeAvro(source, first, allocator); + AvroArrowIo.writeAvro(source, second, allocator); + byte[] bytes = Files.readAllBytes(first); + assertArrayEquals(bytes, Files.readAllBytes(second)); + assertEquals("raincloud-avro01", new String(bytes, bytes.length - 16, 16, "US-ASCII")); + + try (BatchSource got = AvroArrowIo.openAvro(first, allocator)) { + long rows = 0; + for (MaterializedTable batch = got.next(); batch != null; batch = got.next()) { + rows += batch.rowCount; + } + assertEquals(6, rows); + assertNull(got.next()); + } + } + } + + @Test + void aFailedWriteLeavesNoFile() throws Exception { + try (BufferAllocator allocator = new RootAllocator()) { + Path missing = dir.resolve("missing.arrow"); + Path output = dir.resolve("out.avro"); + try { + AvroArrowIo.writeAvro(missing, output, allocator); + } catch (Exception expected) { + // the canonical does not exist + } + assertFalse(Files.exists(output)); + } + } +} diff --git a/sidecars/java/conformance-common/build.gradle.kts b/sidecars/java/conformance-common/build.gradle.kts index 3dbc0b5..36b8d00 100644 --- a/sidecars/java/conformance-common/build.gradle.kts +++ b/sidecars/java/conformance-common/build.gradle.kts @@ -24,7 +24,8 @@ dependencies { api("org.apache.arrow:arrow-compression:$arrowVersion") api("org.apache.arrow:arrow-memory-core:$arrowVersion") - // The knob grammar is tested against the shared sidecars/knob_cases.json, hence Jackson. + // The knob grammar and comparator are tested against the shared sidecars/knob_cases.json + // and sidecars/compare_cases/cases.json, hence Jackson. testImplementation("com.fasterxml.jackson.core:jackson-databind:$jacksonVersion") testImplementation(platform("org.junit:junit-bom:$junitVersion")) testImplementation("org.junit.jupiter:junit-jupiter") @@ -38,6 +39,7 @@ java { } val knobCases = rootDir.resolve("../knob_cases.json") +val compareCases = rootDir.resolve("../compare_cases") tasks.test { useJUnitPlatform() @@ -45,4 +47,6 @@ tasks.test { jvmArgs("--add-opens=java.base/java.nio=ALL-UNNAMED") inputs.file(knobCases) systemProperty("raincloud.knobCases", knobCases.absolutePath) + inputs.dir(compareCases) + systemProperty("raincloud.compareCases", compareCases.absolutePath) } diff --git a/sidecars/java/conformance-common/src/main/java/dev/raincloud/sidecar/common/LogicalCompare.java b/sidecars/java/conformance-common/src/main/java/dev/raincloud/sidecar/common/LogicalCompare.java index af4a90a..3f1e90e 100644 --- a/sidecars/java/conformance-common/src/main/java/dev/raincloud/sidecar/common/LogicalCompare.java +++ b/sidecars/java/conformance-common/src/main/java/dev/raincloud/sidecar/common/LogicalCompare.java @@ -3,6 +3,7 @@ package dev.raincloud.sidecar.common; import java.io.IOException; +import java.math.BigDecimal; import java.util.Arrays; import java.util.HashSet; import java.util.List; @@ -34,11 +35,16 @@ * boxed kind; INT compares exact signed/unsigned integers; FLOAT compares the * decoded values' {@code doubleToRawLongBits} (matches Rust's bitwise NaN / * signed-zero semantics; half floats are decoded from their raw bits first); - * TIMESTAMP of the same timezone compares the INSTANT across differing units - * (e.g. the SECOND→MILLIS promotion Parquet forces); timestamps of differing - * timezone, and different families, are a genuine mismatch (fail); + * TIMESTAMP compares the INSTANT across differing units (e.g. the + * SECOND→MILLIS promotion Parquet forces) and, when both are zoned, across + * zone labels; naive vs zoned, and different families, are a genuine + * mismatch (fail); + *
  • an INT and a scale-0 DECIMAL compare as exact integers; + *
  • a union of exactly Null and T (sparse or dense, at any depth) is a + * nullable T, as Arrow Java's Avro adapter spells Avro's ["null", T]; + * any other union against a non-union is a mismatch; *
  • a type pair the comparator can't confidently judge (differing - * DECIMAL/DATE/TIME/DURATION/NESTED encodings) → the column is a + * DECIMAL/DATE/TIME/DURATION/NESTED encodings, union vs union) → the column is a * comparator gap: the run yields {@code skip} with a distinct note, * never a false {@code pass} and never a {@code fail} for OUR limitation. * @@ -215,9 +221,37 @@ private static boolean hasDuplicateNames(List children) { return false; } + /** + * {@code field} with every union of exactly Null and T replaced by T. Such a + * union's {@code getObject} is already T's value (null for the Null member), so + * only the type needs unwrapping. Callers keep the outer name. + */ + private static Field logical(Field field) { + while (field.getType() instanceof ArrowType.Union && field.getChildren().size() == 2) { + Field a = field.getChildren().get(0), b = field.getChildren().get(1); + boolean aNull = a.getType() instanceof ArrowType.Null, bNull = b.getType() instanceof ArrowType.Null; + if (aNull == bNull) { + break; + } + field = aNull ? b : a; + } + return field; + } + + private static boolean isScaleZeroDecimal(ArrowType t) { + return t instanceof ArrowType.Decimal && ((ArrowType.Decimal) t).getScale() == 0; + } + + private static boolean intAndScaleZeroDecimal(ArrowType a, ArrowType b) { + return (a instanceof ArrowType.Int && isScaleZeroDecimal(b)) + || (isScaleZeroDecimal(a) && b instanceof ArrowType.Int); + } + // Check shape before looking at values: null parents and empty containers still // have schemas. List element names are representation details; struct names are not. private static Compatibility compatibility(Field expected, Field got) { + expected = logical(expected); + got = logical(got); if (expected.getDictionary() != null || got.getDictionary() != null) { return Compatibility.GAP; } @@ -236,7 +270,8 @@ private static Compatibility compatibility(Field expected, Field got) { } if (family(et) == Family.NESTED || family(gt) == Family.NESTED) { if (!sameContainer(et, gt)) { - return et instanceof ArrowType.Union || gt instanceof ArrowType.Union + // Two unions box by type id, unjudged; a union vs a non-union is another type. + return et instanceof ArrowType.Union && gt instanceof ArrowType.Union ? Compatibility.GAP : Compatibility.MISMATCH; } List ec = expected.getChildren(), gc = got.getChildren(); @@ -272,13 +307,16 @@ private static Compatibility compatibility(Field expected, Field got) { } return result; } + if (intAndScaleZeroDecimal(et, gt)) { + return Compatibility.MATCH; + } if (family(et) != family(gt)) { return Compatibility.MISMATCH; } - // A timezone is part of a timestamp's meaning: dropping or changing it is a real - // difference, as in the Python and Rust comparators, not a unit representation. - if (et instanceof ArrowType.Timestamp && !Objects.equals( - ((ArrowType.Timestamp) et).getTimezone(), ((ArrowType.Timestamp) gt).getTimezone())) { + // A zone labels UTC instants; it is not data. Naive and zoned differ in kind, as in + // the Python and Rust comparators. + if (et instanceof ArrowType.Timestamp && (((ArrowType.Timestamp) et).getTimezone() == null) + != (((ArrowType.Timestamp) gt).getTimezone() == null)) { return Compatibility.MISMATCH; } return isGap(et, gt) ? Compatibility.GAP : Compatibility.MATCH; @@ -288,6 +326,8 @@ private static boolean cellEquals(Field expected, Object e, Field got, Object g) if (e == null || g == null) { return e == g; } + expected = logical(expected); + got = logical(got); ArrowType et = expected.getType(), gt = got.getType(); if (family(et) != Family.NESTED && family(gt) != Family.NESTED) { return cellEquals(et, e, gt, g); @@ -335,7 +375,7 @@ private static boolean cellEquals(Field expected, Object e, Field got, Object g) /** * True when a same-family (et, gt) pair is a comparator gap (can't confidently * judge). Only reached from {@link #compatibility}, after cross-family pairs and - * timezone differences have already been ruled a mismatch. + * naive-vs-zoned timestamps have already been ruled a mismatch. */ private static boolean isGap(ArrowType et, ArrowType gt) { if (et.equals(gt)) { @@ -364,6 +404,9 @@ private static boolean cellEquals(ArrowType et, Object e, ArrowType gt, Object g if (et.equals(gt)) { return deepEquals(e, g); } + if (intAndScaleZeroDecimal(et, gt)) { + return exactInteger(et, e).compareTo(exactInteger(gt, g)) == 0; + } Family fe = family(et); Family fg = family(gt); if (fe != fg) { @@ -387,9 +430,16 @@ private static boolean cellEquals(ArrowType et, Object e, ArrowType gt, Object g } } - // Compares two timestamps of the same timezone but possibly differing units. No-tz timestamps - // box as LocalDateTime already at their wall-clock instant (equal instants compare equal - // regardless of source unit); tz timestamps box as a raw long, so normalize both to nanoseconds. + // An INT or scale-0 DECIMAL cell (boxed as a BigDecimal) as an exact number. + private static BigDecimal exactInteger(ArrowType t, Object v) { + return t instanceof ArrowType.Int + ? new BigDecimal(ArrowValues.intAsExact((ArrowType.Int) t, v)) : (BigDecimal) v; + } + + // Compares two timestamps, both naive or both zoned (any zones), possibly of differing units. + // No-tz timestamps box as LocalDateTime already at their wall-clock instant (equal instants + // compare equal regardless of source unit); tz timestamps box as a raw long UTC instant, so + // normalize both to nanoseconds. private static boolean timestampEquals(ArrowType.Timestamp et, Object e, ArrowType.Timestamp gt, Object g) { if (e instanceof java.time.LocalDateTime && g instanceof java.time.LocalDateTime) { return e.equals(g); diff --git a/sidecars/java/conformance-common/src/main/java/dev/raincloud/sidecar/common/MaterializedTable.java b/sidecars/java/conformance-common/src/main/java/dev/raincloud/sidecar/common/MaterializedTable.java index d742829..cb4f686 100644 --- a/sidecars/java/conformance-common/src/main/java/dev/raincloud/sidecar/common/MaterializedTable.java +++ b/sidecars/java/conformance-common/src/main/java/dev/raincloud/sidecar/common/MaterializedTable.java @@ -11,6 +11,7 @@ import org.apache.arrow.vector.dictionary.Dictionary; import org.apache.arrow.vector.dictionary.DictionaryEncoder; import org.apache.arrow.vector.dictionary.DictionaryProvider; +import org.apache.arrow.vector.types.pojo.ArrowType; import org.apache.arrow.vector.types.pojo.DictionaryEncoding; import org.apache.arrow.vector.types.pojo.Field; import org.apache.arrow.vector.types.pojo.FieldType; @@ -170,7 +171,7 @@ private void appendColumn(int c, String name, FieldVector values, boolean decode /** Type, dictionary encoding and children agree recursively (names are not compared). */ private static boolean sameShape(Field recorded, Field actual) { - if (!recorded.getType().equals(actual.getType()) + if (!sameType(recorded.getType(), actual.getType()) || !Objects.equals(recorded.getDictionary(), actual.getDictionary()) || recorded.getChildren().size() != actual.getChildren().size()) { return false; @@ -183,6 +184,15 @@ private static boolean sameShape(Field recorded, Field actual) { return true; } + // UnionVector.getField() re-derives type ids from its members' minor types, so a + // union read from IPC reports other ids than its schema; mode and members decide. + private static boolean sameType(ArrowType recorded, ArrowType actual) { + if (recorded instanceof ArrowType.Union && actual instanceof ArrowType.Union) { + return ((ArrowType.Union) recorded).getMode() == ((ArrowType.Union) actual).getMode(); + } + return recorded.equals(actual); + } + /** The path of the first dictionary-encoded descendant of {@code field}, or null. */ private static String nestedDictionary(Field field, String path) { for (Field child : field.getChildren()) { diff --git a/sidecars/java/conformance-common/src/main/java/dev/raincloud/sidecar/common/ParquetKnobs.java b/sidecars/java/conformance-common/src/main/java/dev/raincloud/sidecar/common/ParquetKnobs.java new file mode 100644 index 0000000..99e483f --- /dev/null +++ b/sidecars/java/conformance-common/src/main/java/dev/raincloud/sidecar/common/ParquetKnobs.java @@ -0,0 +1,92 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +package dev.raincloud.sidecar.common; + +import java.util.Set; +import java.util.function.Function; + +/** + * The Parquet write options every Parquet lane is given the same way + * ({@code raincloud/pipeline/spec.py::ParquetOptions}, which documents each, and + * {@code ParquetOptions} in the Rust lane). The sidecar never sees the recipe: + * {@code SidecarExporter} passes its {@code write.compression} and {@code write.statistics} as + * {@link #COMPRESSION} and {@link #STATISTICS}, and each install setting only when it is set. An + * unset setting ({@code null} here) leaves the writer library's own default; a lane whose library + * cannot do what a set one asks refuses it ({@link #unsupported}) rather than writing something + * else. A count of {@link #NO_LIMIT} is no limit ({@code 0} in the environment). + */ +public record ParquetKnobs(String compression, Integer compressionLevel, boolean statistics, + Integer statisticsColumns, Boolean pageIndex, Integer pageIndexColumns, Integer pageBytes, + Integer pageRows, Boolean dictionary, Integer dictionaryPageBytes, Boolean pageChecksums) { + + public static final String COMPRESSION = "RAINCLOUD_PARQUET_COMPRESSION"; + public static final String COMPRESSION_LEVEL = "RAINCLOUD_PARQUET_COMPRESSION_LEVEL"; + public static final String STATISTICS = "RAINCLOUD_PARQUET_STATISTICS"; + public static final String STATISTICS_COLUMNS = "RAINCLOUD_PARQUET_STATISTICS_COLUMNS"; + public static final String PAGE_INDEX = "RAINCLOUD_PARQUET_PAGE_INDEX"; + public static final String PAGE_INDEX_COLUMNS = "RAINCLOUD_PARQUET_PAGE_INDEX_COLUMNS"; + public static final String PAGE_BYTES = "RAINCLOUD_PARQUET_PAGE_BYTES"; + public static final String PAGE_ROWS = "RAINCLOUD_PARQUET_PAGE_ROWS"; + public static final String DICTIONARY = "RAINCLOUD_PARQUET_DICTIONARY"; + public static final String DICTIONARY_PAGE_BYTES = "RAINCLOUD_PARQUET_DICTIONARY_PAGE_BYTES"; + public static final String PAGE_CHECKSUMS = "RAINCLOUD_PARQUET_PAGE_CHECKSUMS"; + public static final Set CODECS = Set.of("zstd", "snappy", "gzip", "lz4", "brotli", "none"); + /** What 0 (no limit) means for a count. */ + public static final int NO_LIMIT = WriteSettings.NO_LIMIT; + + /** Every option unset: zstd, statistics on, each library's own defaults. */ + public static final ParquetKnobs DEFAULT = new ParquetKnobs("zstd", null, true, null, null, null, null, null, + null, null, null); + + /** The options from the process environment. */ + public static ParquetKnobs fromEnv() { + return from(System::getenv); + } + + /** The options from {@code env}, a variable lookup that returns null when unset. */ + public static ParquetKnobs from(Function env) { + String rawCodec = env.apply(COMPRESSION); + String compression = rawCodec == null ? "zstd" : rawCodec.strip(); + if (!CODECS.contains(compression)) { + throw new IllegalArgumentException(COMPRESSION + "='" + rawCodec + + "' is not one of zstd, snappy, gzip, lz4, brotli, none"); + } + Boolean statistics = toggle(STATISTICS, env.apply(STATISTICS)); + boolean withStatistics = statistics == null || statistics; + ParquetKnobs knobs = new ParquetKnobs(compression, level(env.apply(COMPRESSION_LEVEL)), withStatistics, + count(STATISTICS_COLUMNS, env.apply(STATISTICS_COLUMNS)), toggle(PAGE_INDEX, env.apply(PAGE_INDEX)), + count(PAGE_INDEX_COLUMNS, env.apply(PAGE_INDEX_COLUMNS)), count(PAGE_BYTES, env.apply(PAGE_BYTES)), + count(PAGE_ROWS, env.apply(PAGE_ROWS)), toggle(DICTIONARY, env.apply(DICTIONARY)), + count(DICTIONARY_PAGE_BYTES, env.apply(DICTIONARY_PAGE_BYTES)), + toggle(PAGE_CHECKSUMS, env.apply(PAGE_CHECKSUMS))); + if (!withStatistics) { + if (Boolean.TRUE.equals(knobs.pageIndex) || knobs.statisticsColumns != null + || knobs.pageIndexColumns != null) { + throw new IllegalArgumentException("a page index or per-column statistics ask for statistics, but " + + STATISTICS + " is 0"); + } + } + if (Boolean.FALSE.equals(knobs.pageIndex) && knobs.pageIndexColumns != null) { + throw new IllegalArgumentException(PAGE_INDEX_COLUMNS + " asks for a page index, but " + PAGE_INDEX + + " is 0"); + } + return knobs; + } + + private static Boolean toggle(String var, String raw) { + return WriteSettings.toggle(var, raw); + } + + private static Integer count(String var, String raw) { + return WriteSettings.count(var, raw); + } + + private static Integer level(String raw) { + return WriteSettings.level(COMPRESSION_LEVEL, raw); + } + + /** Refuse an option this lane's library cannot honour, naming the lane, the setting and why. */ + public static IllegalArgumentException unsupported(String lane, String var, Object value, String why) { + return WriteSettings.unsupported(lane, var, value, why); + } +} diff --git a/sidecars/java/conformance-common/src/main/java/dev/raincloud/sidecar/common/WriteSettings.java b/sidecars/java/conformance-common/src/main/java/dev/raincloud/sidecar/common/WriteSettings.java new file mode 100644 index 0000000..e4622a6 --- /dev/null +++ b/sidecars/java/conformance-common/src/main/java/dev/raincloud/sidecar/common/WriteSettings.java @@ -0,0 +1,73 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +package dev.raincloud.sidecar.common; + +import java.util.Arrays; +import java.util.Locale; + +/** + * How every lane reads a write setting from its environment ({@code raincloud/pipeline/spec.py}, + * which declares them all, and the Rust lane): unset or empty is {@code null}, the library's + * default; anything malformed is refused naming the variable. The build passes each set setting + * in one canonical form, so a sidecar run by hand reads the same grammar. + */ +public final class WriteSettings { + private WriteSettings() {} + + /** What 0 (no limit) means for a size or count: the libraries take an int. */ + public static final int NO_LIMIT = Integer.MAX_VALUE; + + /** An on/off setting: 1/true/yes/on or 0/false/no/off, in any case. */ + public static Boolean toggle(String var, String raw) { + String value = raw == null ? "" : raw.strip().toLowerCase(Locale.ROOT); + switch (value) { + case "": + return null; + case "1": case "true": case "yes": case "on": + return true; + case "0": case "false": case "no": case "off": + return false; + default: + throw new IllegalArgumentException(var + "='" + raw + + "' is not a switch; give 1 or 0 (true/false, yes/no, on/off)"); + } + } + + /** A size or count, in the count grammar ({@link Knobs#count}); 0 is {@link #NO_LIMIT}. */ + public static Integer count(String var, String raw) { + if (raw == null) { + return null; + } + return (int) Math.min(Knobs.count(var, raw, NO_LIMIT, NO_LIMIT), NO_LIMIT); + } + + /** A compression level: plain ASCII digits. */ + public static Integer level(String var, String raw) { + String value = raw == null ? "" : raw.strip(); + if (value.isEmpty()) { + return null; + } + if (!value.chars().allMatch(c -> c >= '0' && c <= '9') || value.length() > 9) { + throw new IllegalArgumentException(var + "='" + raw + + "' is not a compression level; give a whole number such as 3"); + } + return Integer.parseInt(value); + } + + /** One of {@code choices}, in any case. */ + public static String choice(String var, String raw, String... choices) { + String value = raw == null ? "" : raw.strip().toLowerCase(Locale.ROOT); + if (value.isEmpty()) { + return null; + } + if (!Arrays.asList(choices).contains(value)) { + throw new IllegalArgumentException(var + "='" + raw + "' is not one of " + String.join(", ", choices)); + } + return value; + } + + /** Refuse a setting this lane's library cannot honour, naming the lane, the setting and why. */ + public static IllegalArgumentException unsupported(String lane, String var, Object value, String why) { + return new IllegalArgumentException(lane + " cannot honour " + var + "=" + value + ": " + why); + } +} diff --git a/sidecars/java/conformance-common/src/test/java/dev/raincloud/sidecar/common/CompareCasesTest.java b/sidecars/java/conformance-common/src/test/java/dev/raincloud/sidecar/common/CompareCasesTest.java new file mode 100644 index 0000000..d36753f --- /dev/null +++ b/sidecars/java/conformance-common/src/test/java/dev/raincloud/sidecar/common/CompareCasesTest.java @@ -0,0 +1,56 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +package dev.raincloud.sidecar.common; + +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertTrue; + +import java.io.IOException; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.ArrayList; +import java.util.List; + +import org.apache.arrow.memory.RootAllocator; +import org.apache.arrow.vector.ipc.ArrowFileReader; +import org.junit.jupiter.api.Test; + +import com.fasterxml.jackson.databind.JsonNode; +import com.fasterxml.jackson.databind.ObjectMapper; + +/** The representation rules every lane's comparator shares (sidecars/compare_cases). */ +class CompareCasesTest { + + @Test + void followsTheSharedCases() throws IOException { + // The same files pytest and cargo test read: one verdict per case in every lane. + Path dir = Path.of(System.getProperty("raincloud.compareCases")); + JsonNode table = new ObjectMapper().readTree(dir.resolve("cases.json").toFile()); + List wrong = new ArrayList<>(); + int cases = 0; + try (RootAllocator allocator = new RootAllocator()) { + for (JsonNode c : table.get("cases")) { + String name = c.get("name").asText(); + String want = "equal".equals(c.get("verdict").asText()) ? "pass" : "fail"; + boolean gap = c.path("gap").asBoolean(false); + Verdict v; + try (BatchSource got = open(dir.resolve(name + ".got.arrow"), allocator); + BatchSource expected = open(dir.resolve(name + ".expected.arrow"), allocator)) { + v = LogicalCompare.compare(name, got, expected); + } + // A gap case may stay unmeasured, never take the opposite verdict. + if (!v.status.equals(want) && !(gap && "skip".equals(v.status))) { + wrong.add(name + ": want " + want + (gap ? " or skip" : "") + ", got " + v.status + + " (" + v.note + "; " + v.detail + ")"); + } + cases++; + } + } + assertFalse(cases == 0, "no cases in " + dir); + assertTrue(wrong.isEmpty(), String.join("\n", wrong)); + } + + private static BatchSource open(Path file, RootAllocator allocator) throws IOException { + return BatchSource.of(new ArrowFileReader(Files.newByteChannel(file), allocator)); + } +} diff --git a/sidecars/java/conformance-common/src/test/java/dev/raincloud/sidecar/common/ParquetKnobsTest.java b/sidecars/java/conformance-common/src/test/java/dev/raincloud/sidecar/common/ParquetKnobsTest.java new file mode 100644 index 0000000..e27dd87 --- /dev/null +++ b/sidecars/java/conformance-common/src/test/java/dev/raincloud/sidecar/common/ParquetKnobsTest.java @@ -0,0 +1,51 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +package dev.raincloud.sidecar.common; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertThrows; +import static org.junit.jupiter.api.Assertions.assertTrue; + +import java.util.Map; + +import org.junit.jupiter.api.Test; + +/** The Parquet write options as every JVM Parquet lane reads them. */ +class ParquetKnobsTest { + + private static ParquetKnobs from(Map vars) { + return ParquetKnobs.from(name -> vars.get(name.replace("RAINCLOUD_PARQUET_", ""))); + } + + @Test + void unsetIsEachLibrarysDefault() { + assertEquals(ParquetKnobs.DEFAULT, from(Map.of())); + assertEquals(ParquetKnobs.DEFAULT, from(Map.of("PAGE_INDEX", " ", "COMPRESSION_LEVEL", ""))); + } + + @Test + void readsWhatThePythonLaneWrites() { + ParquetKnobs knobs = from(Map.of("COMPRESSION", "gzip", "COMPRESSION_LEVEL", "9", "STATISTICS", "1", + "STATISTICS_COLUMNS", "100", "PAGE_INDEX", " On ", "PAGE_INDEX_COLUMNS", "10", "PAGE_BYTES", "4096", + "PAGE_ROWS", "0", "DICTIONARY", "off", "DICTIONARY_PAGE_BYTES", "65536")); + assertEquals(new ParquetKnobs("gzip", 9, true, 100, true, 10, 4096, ParquetKnobs.NO_LIMIT, false, 65536, + null), knobs); + assertEquals(true, from(Map.of("PAGE_CHECKSUMS", "yes")).pageChecksums()); + } + + @Test + void refusesWhatNoLaneCanRead() { + Map, String> cases = Map.of( + Map.of("PAGE_INDEX", "maybe"), "is not a switch", + Map.of("COMPRESSION", "lzo"), "is not one of", + Map.of("PAGE_BYTES", "1MiB"), "is not a number", + Map.of("COMPRESSION_LEVEL", "-1"), "is not a compression level", + Map.of("PAGE_INDEX", "1", "STATISTICS", "0"), "ask for statistics", + Map.of("STATISTICS_COLUMNS", "5", "STATISTICS", "0"), "ask for statistics", + Map.of("PAGE_INDEX", "0", "PAGE_INDEX_COLUMNS", "5"), "asks for a page index"); + cases.forEach((vars, error) -> { + IllegalArgumentException e = assertThrows(IllegalArgumentException.class, () -> from(vars)); + assertTrue(e.getMessage().contains(error), vars + ": " + e.getMessage()); + }); + } +} diff --git a/sidecars/java/gradle.properties b/sidecars/java/gradle.properties index 4ff25d5..c25a5f5 100644 --- a/sidecars/java/gradle.properties +++ b/sidecars/java/gradle.properties @@ -23,6 +23,12 @@ junitVersion=5.10.2 slf4jVersion=2.0.18 jacksonVersion=2.21.0 hardwoodVersion=1.1.0.Beta1 +# The avro@java lanes' zstandard codec: Avro declares zstd-jni optional, so the version +# is Avro's own pin (avro-parent 1.12.1, the Avro that arrow-avro $arrowVersion brings). +zstdJniVersion=1.5.7-4 +# Avro 1.12.1's own versions of its optional snappy and xz codecs (avro-parent pom). +snappyJavaVersion=1.1.10.8 +xzVersion=1.10 # Where Gradle looks for JDKs before provisioning one: the variables actions/setup-java # exports on CI (unset elsewhere, which Gradle ignores), so a CI job that installs 17 and diff --git a/sidecars/java/parquet-arrow-java b/sidecars/java/parquet-arrow-java index a54c89b..9f98247 160000 --- a/sidecars/java/parquet-arrow-java +++ b/sidecars/java/parquet-arrow-java @@ -1 +1 @@ -Subproject commit a54c89b27290ae9025cf6f8bc2df0d92dc7a39ed +Subproject commit 9f98247cf2129dcd71e8e755c04b330cf7ed62d7 diff --git a/sidecars/java/parquet-hardwood/build.gradle.kts b/sidecars/java/parquet-hardwood/build.gradle.kts index fd0d859..3dbc353 100644 --- a/sidecars/java/parquet-hardwood/build.gradle.kts +++ b/sidecars/java/parquet-hardwood/build.gradle.kts @@ -31,13 +31,18 @@ dependencies { // Hardwood's BOM pins its optional codec libraries, which a consumer must declare to // get them: zstd (this lane writes zstd), and snappy/lz4 so the reader opens Parquet - // other writers compress that way (pyarrow's default is snappy). Brotli is not bundled: - // it needs a per-platform native artifact, and no raincloud writer produces it. + // other writers compress that way (pyarrow's default is snappy), and brotli4j, with its + // native library for the platforms raincloud builds on, since RAINCLOUD_PARQUET_COMPRESSION + // can ask for Brotli. implementation(platform("dev.hardwood:hardwood-bom:$hardwoodVersion")) implementation("dev.hardwood:hardwood-core") runtimeOnly("com.github.luben:zstd-jni") runtimeOnly("org.xerial.snappy:snappy-java") runtimeOnly("at.yawk.lz4:lz4-java") + runtimeOnly("com.aayushatharva.brotli4j:brotli4j") + for (platform in listOf("linux-x86_64", "linux-aarch64", "osx-x86_64", "osx-aarch64")) { + runtimeOnly("com.aayushatharva.brotli4j:native-$platform:1.23.0") + } // Off-heap allocator impl for the Arrow side + silence Arrow's SLF4J warning. runtimeOnly("org.apache.arrow:arrow-memory-netty:$arrowVersion") diff --git a/sidecars/java/parquet-hardwood/src/main/java/dev/raincloud/sidecar/hardwood/HardwoodWriter.java b/sidecars/java/parquet-hardwood/src/main/java/dev/raincloud/sidecar/hardwood/HardwoodWriter.java index c8f1778..acc126f 100644 --- a/sidecars/java/parquet-hardwood/src/main/java/dev/raincloud/sidecar/hardwood/HardwoodWriter.java +++ b/sidecars/java/parquet-hardwood/src/main/java/dev/raincloud/sidecar/hardwood/HardwoodWriter.java @@ -62,10 +62,12 @@ import dev.hardwood.metadata.SchemaElement; import dev.hardwood.schema.FileSchema; import dev.hardwood.writer.ColumnBatch; +import dev.hardwood.writer.ColumnEncoding; import dev.hardwood.writer.ColumnWriter; import dev.hardwood.writer.ParquetFileWriter; import dev.hardwood.writer.WriterConfig; import dev.raincloud.sidecar.common.Knobs; +import dev.raincloud.sidecar.common.ParquetKnobs; import dev.raincloud.sidecar.common.VariantFidelity; /** @@ -107,19 +109,90 @@ private HardwoodWriter() {} * than the other lanes'; there is no Hardwood setting that measures encoded bytes.

    */ public static WriterConfig writerConfig(String maxRows, String targetEncodedBytes) { - return WriterConfig.builder() - .codec(CompressionCodec.ZSTD) + return writerConfig(maxRows, targetEncodedBytes, ParquetKnobs.DEFAULT); + } + + /** + * {@link #writerConfig(String, String)} with the Parquet options every lane is given. + * + *

    Hardwood 1.1.0.Beta1's {@code WriterConfig} has a codec, a page size target and a + * per-column encoding, and nothing else these settings ask for: no compression level, no page + * row limit, no dictionary page size, no statistics switch (it always writes column-chunk + * statistics), no page index (it writes none) and no way to leave out page checksums (it + * always writes them). A setting that needs one of those is refused rather than written + * another way. Dictionaries off is {@code ColumnEncoding.PLAIN}, the encoding the other + * libraries fall back to.

    + */ + public static WriterConfig writerConfig(String maxRows, String targetEncodedBytes, ParquetKnobs knobs) { + String lane = "parquet@hardwood"; + if (knobs.compressionLevel() != null) { + throw ParquetKnobs.unsupported(lane, ParquetKnobs.COMPRESSION_LEVEL, knobs.compressionLevel(), + "Hardwood has no compression level"); + } + if (!knobs.statistics() || knobs.statisticsColumns() != null) { + throw ParquetKnobs.unsupported(lane, + knobs.statistics() ? ParquetKnobs.STATISTICS_COLUMNS : ParquetKnobs.STATISTICS, + knobs.statistics() ? knobs.statisticsColumns() : 0, + "Hardwood always writes column-chunk statistics for every column"); + } + if (Boolean.TRUE.equals(knobs.pageIndex()) || knobs.pageIndexColumns() != null) { + throw ParquetKnobs.unsupported(lane, + knobs.pageIndexColumns() != null ? ParquetKnobs.PAGE_INDEX_COLUMNS : ParquetKnobs.PAGE_INDEX, + knobs.pageIndexColumns() != null ? knobs.pageIndexColumns() : 1, + "Hardwood writes no page index"); + } + if (knobs.pageRows() != null) { + throw ParquetKnobs.unsupported(lane, ParquetKnobs.PAGE_ROWS, knobs.pageRows(), + "Hardwood has no page row limit"); + } + if (knobs.dictionaryPageBytes() != null) { + throw ParquetKnobs.unsupported(lane, ParquetKnobs.DICTIONARY_PAGE_BYTES, knobs.dictionaryPageBytes(), + "Hardwood has no dictionary page size limit"); + } + if (Boolean.FALSE.equals(knobs.pageChecksums())) { + throw ParquetKnobs.unsupported(lane, ParquetKnobs.PAGE_CHECKSUMS, 0, + "Hardwood writes a checksum in every page header"); + } + WriterConfig.Builder builder = WriterConfig.builder() + .codec(codec(knobs.compression())) .rowGroupTargetRows(Knobs.count(Knobs.MAX_ROWS, maxRows, Knobs.DEFAULT_MAX_ROWS, Long.MAX_VALUE)) .rowGroupBufferTargetBytes(Knobs.count(Knobs.TARGET_ENCODED_BYTES, targetEncodedBytes, - Knobs.DEFAULT_TARGET_ENCODED_BYTES, Long.MAX_VALUE)) - .build(); + Knobs.DEFAULT_TARGET_ENCODED_BYTES, Long.MAX_VALUE)); + if (knobs.pageBytes() != null) { + builder.pageTargetBytes(knobs.pageBytes()); + } + if (Boolean.FALSE.equals(knobs.dictionary())) { + builder.encoding(ColumnEncoding.PLAIN); + } + return builder.build(); + } + + private static CompressionCodec codec(String codec) { + switch (codec) { + case "zstd": + return CompressionCodec.ZSTD; + case "snappy": + return CompressionCodec.SNAPPY; + case "gzip": + return CompressionCodec.GZIP; + case "lz4": + // pyarrow's "lz4" and arrow-rs's LZ4_RAW, not the deprecated Hadoop LZ4. + return CompressionCodec.LZ4_RAW; + case "brotli": + return CompressionCodec.BROTLI; + case "none": + return CompressionCodec.UNCOMPRESSED; + default: + throw new IllegalArgumentException("unknown codec " + codec); + } } - /** Stream the canonical into a zstd Parquet with the knobs from the environment. */ + /** Stream the canonical into a Parquet with the knobs from the environment. */ public static void writeParquet(Path canonical, Path output, BufferAllocator allocator) throws IOException { // Resolved before touching the output, so a bad knob leaves nothing behind. writeParquet(canonical, output, allocator, - writerConfig(System.getenv(Knobs.MAX_ROWS), System.getenv(Knobs.TARGET_ENCODED_BYTES))); + writerConfig(System.getenv(Knobs.MAX_ROWS), System.getenv(Knobs.TARGET_ENCODED_BYTES), + ParquetKnobs.fromEnv())); } /** diff --git a/sidecars/java/parquet-hardwood/src/test/java/dev/raincloud/sidecar/hardwood/HardwoodIoTest.java b/sidecars/java/parquet-hardwood/src/test/java/dev/raincloud/sidecar/hardwood/HardwoodIoTest.java index 5e388c5..6dd34b3 100644 --- a/sidecars/java/parquet-hardwood/src/test/java/dev/raincloud/sidecar/hardwood/HardwoodIoTest.java +++ b/sidecars/java/parquet-hardwood/src/test/java/dev/raincloud/sidecar/hardwood/HardwoodIoTest.java @@ -68,6 +68,7 @@ import dev.raincloud.sidecar.common.BatchSource; import dev.raincloud.sidecar.common.CanonicalReader; import dev.raincloud.sidecar.common.LogicalCompare; +import dev.raincloud.sidecar.common.ParquetKnobs; import dev.raincloud.sidecar.common.Verdict; /** The Arrow ⇆ Hardwood bridge: types, nested shapes, row groups, failure atomicity. */ @@ -321,6 +322,35 @@ private static void fill(VectorSchemaRoot root, int from, int to) { root.setRowCount(to - from); } + /** Options as the environment would give them: alternating setting names (after + * {@code RAINCLOUD_PARQUET_}) and values. */ + private static ParquetKnobs knobs(String... pairs) { + java.util.Map vars = new java.util.HashMap<>(); + for (int i = 0; i < pairs.length; i += 2) { + vars.put("RAINCLOUD_PARQUET_" + pairs[i], pairs[i + 1]); + } + return ParquetKnobs.from(vars::get); + } + + @Test + void parquetKnobsHardwoodCannotHonourAreRefused() { + for (ParquetKnobs knobs : new ParquetKnobs[] { + knobs("STATISTICS", "0"), knobs("PAGE_INDEX", "1"), knobs("PAGE_ROWS", "1000"), + knobs("COMPRESSION_LEVEL", "3"), knobs("STATISTICS_COLUMNS", "10"), + knobs("PAGE_INDEX_COLUMNS", "10"), knobs("DICTIONARY_PAGE_BYTES", "65536"), + knobs("PAGE_CHECKSUMS", "0")}) { + IllegalArgumentException e = assertThrows(IllegalArgumentException.class, + () -> HardwoodWriter.writerConfig(null, null, knobs)); + assertTrue(e.getMessage().startsWith("parquet@hardwood cannot honour RAINCLOUD_PARQUET_"), + e.getMessage()); + } + // What it can: every codec, no page index, a page size, dictionaries off, checksums on. + for (String codec : ParquetKnobs.CODECS) { + HardwoodWriter.writerConfig(null, null, knobs("COMPRESSION", codec, "PAGE_INDEX", "0", + "PAGE_BYTES", "4096", "DICTIONARY", "0", "PAGE_CHECKSUMS", "1")); + } + } + @Test void aMalformedKnobIsRefusedNamingTheVariable() { IllegalArgumentException e = assertThrows(IllegalArgumentException.class, diff --git a/sidecars/java/parquet-java/build.gradle.kts b/sidecars/java/parquet-java/build.gradle.kts index 7c3c4c8..353a6af 100644 --- a/sidecars/java/parquet-java/build.gradle.kts +++ b/sidecars/java/parquet-java/build.gradle.kts @@ -30,7 +30,7 @@ dependencies { // Composite-build substituted from ./parquet-arrow-java (see settings.gradle.kts), // so the submodule commit is the pin and this version is nominal. - implementation("dev.spiraldb.parquet.arrow:parquet-arrow-core:0.2.0") + implementation("dev.spiraldb.parquet.arrow:parquet-arrow-core:0.3.0") // Off-heap allocator impl for reading the canonical + silence SLF4J. runtimeOnly("org.apache.arrow:arrow-memory-netty:$arrowVersion") diff --git a/sidecars/java/parquet-java/src/main/java/dev/raincloud/sidecar/parquetjava/ParquetArrowIo.java b/sidecars/java/parquet-java/src/main/java/dev/raincloud/sidecar/parquetjava/ParquetArrowIo.java index 6b89811..5d67650 100644 --- a/sidecars/java/parquet-java/src/main/java/dev/raincloud/sidecar/parquetjava/ParquetArrowIo.java +++ b/sidecars/java/parquet-java/src/main/java/dev/raincloud/sidecar/parquetjava/ParquetArrowIo.java @@ -8,6 +8,7 @@ import java.nio.file.Path; import java.nio.file.StandardOpenOption; import java.util.HashSet; +import java.util.List; import java.util.Set; import org.apache.arrow.compression.CommonsCompressionFactory; @@ -15,6 +16,7 @@ import org.apache.arrow.vector.ipc.ArrowFileReader; import org.apache.arrow.vector.types.pojo.Schema; import org.apache.parquet.ParquetReadOptions; +import org.apache.parquet.column.ColumnDescriptor; import org.apache.parquet.conf.PlainParquetConfiguration; import org.apache.parquet.hadoop.ParquetFileReader; import org.apache.parquet.io.LocalInputFile; @@ -23,6 +25,8 @@ import dev.raincloud.sidecar.common.BatchSource; import dev.raincloud.sidecar.common.Knobs; +import dev.raincloud.sidecar.common.ParquetKnobs; +import dev.spiraldb.parquet.arrow.ArrowSchemaToParquet; import dev.spiraldb.parquet.arrow.Compression; import dev.spiraldb.parquet.arrow.CoreCompressionCodecFactory; import dev.spiraldb.parquet.arrow.ParquetArrow; @@ -76,15 +80,94 @@ static long rowGroupTargetEncodedBytes(String raw) { * rows in a group.

    */ static WriteOptions writeOptions(String maxRows, String targetEncodedBytes) { - return WriteOptions.builder() - .compression(Compression.ZSTD) + return writeOptions(maxRows, targetEncodedBytes, ParquetKnobs.DEFAULT); + } + + /** + * {@link #writeOptions(String, String)} with the Parquet options every lane is given. + * + *

    parquet-java writes a page index for every column whenever statistics are on, so + * leaving it out, or limiting it to some columns, is refused; so is Brotli, which + * parquet-arrow-java does not write. {@code RAINCLOUD_PARQUET_STATISTICS_COLUMNS} needs the + * file's leaf columns, so {@link #writeParquet(Path, Path, BufferAllocator)} applies it + * ({@link #limitStatistics}) once the canonical's schema is open.

    + */ + static WriteOptions writeOptions(String maxRows, String targetEncodedBytes, ParquetKnobs knobs) { + return optionsBuilder(maxRows, targetEncodedBytes, knobs).build(); + } + + private static WriteOptions.Builder optionsBuilder(String maxRows, String targetEncodedBytes, ParquetKnobs knobs) { + String lane = "parquet@java"; + if (knobs.pageIndexColumns() != null) { + throw ParquetKnobs.unsupported(lane, ParquetKnobs.PAGE_INDEX_COLUMNS, knobs.pageIndexColumns(), + "parquet-java writes a page index for every column that has statistics"); + } + if (Boolean.FALSE.equals(knobs.pageIndex()) && knobs.statistics()) { + throw ParquetKnobs.unsupported(lane, ParquetKnobs.PAGE_INDEX, 0, + "parquet-java writes a page index whenever statistics are on"); + } + WriteOptions.Builder builder = WriteOptions.builder() + .compression(compression(knobs.compression())) + .statisticsEnabled(knobs.statistics()) .maxRowGroupRows(rowGroupMaxRows(maxRows)) - .targetRowGroupBytes(rowGroupTargetEncodedBytes(targetEncodedBytes)) - .build(); + .targetRowGroupBytes(rowGroupTargetEncodedBytes(targetEncodedBytes)); + if (knobs.compressionLevel() != null) { + builder.compressionLevel(knobs.compressionLevel()); + } + if (knobs.pageBytes() != null) { + builder.pageSizeBytes(knobs.pageBytes()); + } + if (knobs.pageRows() != null) { + builder.pageRowLimit(knobs.pageRows()); + } + if (knobs.dictionary() != null) { + builder.parquetDictionaryEnabled(knobs.dictionary()); + } + if (knobs.dictionaryPageBytes() != null) { + builder.dictionaryPageSizeBytes(knobs.dictionaryPageBytes()); + } + if (knobs.pageChecksums() != null) { + builder.pageChecksums(knobs.pageChecksums()); + } + return builder; + } + + /** The write options for the canonical's schema. */ + @FunctionalInterface + private interface OptionsFor { + WriteOptions apply(Schema schema) throws IOException; + } + + /** Statistics off for every Parquet leaf column of {@code schema} past the first {@code columns}. */ + static WriteOptions.Builder limitStatistics(WriteOptions.Builder builder, Schema schema, int columns) + throws IOException { + List leaves = ArrowSchemaToParquet.toParquet(schema).getColumns(); + for (int i = columns; i < leaves.size(); i++) { + builder.columnStatisticsEnabled(String.join(".", leaves.get(i).getPath()), false); + } + return builder; + } + + private static Compression compression(String codec) { + switch (codec) { + case "zstd": + return Compression.ZSTD; + case "snappy": + return Compression.SNAPPY; + case "gzip": + return Compression.GZIP; + case "none": + return Compression.UNCOMPRESSED; + case "lz4": + return Compression.LZ4_RAW; + default: + throw ParquetKnobs.unsupported("parquet@java", ParquetKnobs.COMPRESSION, codec, + "parquet-arrow-java writes no brotli"); + } } /** - * Streams the canonical Arrow IPC file straight into a zstd Parquet (no intermediate copy). + * Streams the canonical Arrow IPC file straight into a Parquet (no intermediate copy). * *

    Atomic w.r.t. failure: any failed write — an exception such as {@link * dev.spiraldb.parquet.arrow.UnsupportedParquetTypeException} mid-stream, or an @@ -94,20 +177,29 @@ static WriteOptions writeOptions(String maxRows, String targetEncodedBytes) { public static void writeParquet(Path canonicalArrow, Path output, BufferAllocator allocator) throws IOException { // Resolved before touching the output, so a bad knob leaves nothing behind. - writeParquet(canonicalArrow, output, allocator, - writeOptions(System.getenv(Knobs.MAX_ROWS), System.getenv(Knobs.TARGET_ENCODED_BYTES))); + ParquetKnobs knobs = ParquetKnobs.fromEnv(); + WriteOptions.Builder options = + optionsBuilder(System.getenv(Knobs.MAX_ROWS), System.getenv(Knobs.TARGET_ENCODED_BYTES), knobs); + options.build(); + writeParquet(canonicalArrow, output, allocator, schema -> knobs.statisticsColumns() == null + ? options.build() : limitStatistics(options, schema, knobs.statisticsColumns()).build()); } /** {@link #writeParquet(Path, Path, BufferAllocator)} with resolved options. */ static void writeParquet(Path canonicalArrow, Path output, BufferAllocator allocator, WriteOptions options) throws IOException { + writeParquet(canonicalArrow, output, allocator, schema -> options); + } + + private static void writeParquet(Path canonicalArrow, Path output, BufferAllocator allocator, + OptionsFor optionsFor) throws IOException { Files.deleteIfExists(output); // parquet-arrow-java's writer is create-new boolean written = false; try (SeekableByteChannel channel = Files.newByteChannel(canonicalArrow, StandardOpenOption.READ); ArrowFileReader input = new ArrowFileReader(channel, allocator, CommonsCompressionFactory.INSTANCE)) { Schema schema = input.getVectorSchemaRoot().getSchema(); - try (ParquetArrowWriter writer = ParquetArrow.writer(schema).options(options).build(output)) { + try (ParquetArrowWriter writer = ParquetArrow.writer(schema).options(optionsFor.apply(schema)).build(output)) { writer.writeAll(input); writer.finish(); } diff --git a/sidecars/java/parquet-java/src/test/java/dev/raincloud/sidecar/parquetjava/ParquetArrowIoTest.java b/sidecars/java/parquet-java/src/test/java/dev/raincloud/sidecar/parquetjava/ParquetArrowIoTest.java index fad9239..b5884ba 100644 --- a/sidecars/java/parquet-java/src/test/java/dev/raincloud/sidecar/parquetjava/ParquetArrowIoTest.java +++ b/sidecars/java/parquet-java/src/test/java/dev/raincloud/sidecar/parquetjava/ParquetArrowIoTest.java @@ -33,9 +33,12 @@ import dev.raincloud.sidecar.common.CanonicalReader; import dev.raincloud.sidecar.common.LogicalCompare; import dev.raincloud.sidecar.common.MaterializedTable; +import dev.raincloud.sidecar.common.ParquetKnobs; import dev.raincloud.sidecar.common.Verdict; +import dev.spiraldb.parquet.arrow.Compression; import dev.spiraldb.parquet.arrow.ParquetArrow; import dev.spiraldb.parquet.arrow.ParquetArrowReader; +import dev.spiraldb.parquet.arrow.WriteOptions; /** Direct coverage for the Arrow⇆Parquet hop the {@code parquet@java} lane depends on. */ class ParquetArrowIoTest { @@ -162,4 +165,49 @@ void rowGroupKnobs_reachTheWriter() throws IOException { assertEquals(20, rowGroups(capped)); assertEquals(20_000, rows(small)); } + + // ---- Parquet options (their grammar is ParquetKnobsTest's, in conformance-common) ---- + + /** Options as the environment would give them: alternating setting names (after + * {@code RAINCLOUD_PARQUET_}) and values. */ + private static ParquetKnobs knobs(String... pairs) { + java.util.Map vars = new java.util.HashMap<>(); + for (int i = 0; i < pairs.length; i += 2) { + vars.put("RAINCLOUD_PARQUET_" + pairs[i], pairs[i + 1]); + } + return ParquetKnobs.from(vars::get); + } + + @Test + void parquetKnobs_reachTheWriter() { + WriteOptions options = ParquetArrowIo.writeOptions(null, null, knobs("COMPRESSION", "gzip", + "PAGE_BYTES", "4096", "PAGE_ROWS", "1000", "DICTIONARY", "0", "DICTIONARY_PAGE_BYTES", "65536", + "PAGE_CHECKSUMS", "0", "PAGE_INDEX", "1")); + assertEquals(Compression.GZIP, options.compression()); + assertEquals(4096, options.pageSizeBytes()); + assertEquals(1000, options.pageRowLimit()); + assertFalse(options.parquetDictionaryEnabled()); + assertEquals(65536, options.dictionaryPageSizeBytes()); + assertFalse(options.pageChecksums()); + assertFalse(ParquetArrowIo.writeOptions(null, null, knobs("STATISTICS", "0")).statisticsEnabled()); + WriteOptions leveled = ParquetArrowIo.writeOptions(null, null, knobs("COMPRESSION_LEVEL", "9")); + assertEquals(9, leveled.compressionLevel().getAsInt()); + assertEquals(Compression.LZ4_RAW, ParquetArrowIo.writeOptions(null, null, knobs("COMPRESSION", "lz4")).compression()); + // Unset leaves parquet-arrow-java's defaults. + WriteOptions defaults = WriteOptions.builder().build(); + WriteOptions unset = ParquetArrowIo.writeOptions(null, null, ParquetKnobs.DEFAULT); + assertEquals(defaults.pageSizeBytes(), unset.pageSizeBytes()); + assertEquals(defaults.pageRowLimit(), unset.pageRowLimit()); + assertEquals(defaults.pageChecksums(), unset.pageChecksums()); + } + + @Test + void parquetKnobs_theLibraryCannotHonourAreRefused() { + for (ParquetKnobs knobs : new ParquetKnobs[] { + knobs("COMPRESSION", "brotli"), knobs("PAGE_INDEX", "0"), knobs("PAGE_INDEX_COLUMNS", "10")}) { + IllegalArgumentException e = assertThrows(IllegalArgumentException.class, + () -> ParquetArrowIo.writeOptions(null, null, knobs)); + assertTrue(e.getMessage().startsWith("parquet@java cannot honour RAINCLOUD_PARQUET_"), e.getMessage()); + } + } } diff --git a/sidecars/java/settings.gradle.kts b/sidecars/java/settings.gradle.kts index 2781549..7b38971 100644 --- a/sidecars/java/settings.gradle.kts +++ b/sidecars/java/settings.gradle.kts @@ -18,6 +18,7 @@ include("conformance-common") include("vortex-jni-reader") include("parquet-java") include("parquet-hardwood") +include("avro-java") // parquet-arrow-java (git submodule) supplies the Hadoop-free Arrow⇆Parquet bridge the // parquet@java lane hops through (arrow → parquet-arrow-java → parquet-java). Consumed as a diff --git a/sidecars/java/vortex-jni-reader/src/main/java/dev/raincloud/sidecar/vortexjni/ConformanceWriter.java b/sidecars/java/vortex-jni-reader/src/main/java/dev/raincloud/sidecar/vortexjni/ConformanceWriter.java index bdb0954..e939e2a 100644 --- a/sidecars/java/vortex-jni-reader/src/main/java/dev/raincloud/sidecar/vortexjni/ConformanceWriter.java +++ b/sidecars/java/vortex-jni-reader/src/main/java/dev/raincloud/sidecar/vortexjni/ConformanceWriter.java @@ -32,6 +32,7 @@ import dev.raincloud.sidecar.common.CanonicalReader; import dev.raincloud.sidecar.common.LogicalCompare; import dev.raincloud.sidecar.common.Verdict; +import dev.raincloud.sidecar.common.WriteSettings; import dev.raincloud.sidecar.common.WriterMain; import dev.vortex.api.Session; import dev.vortex.api.VortexWriter; @@ -85,8 +86,27 @@ static int run(String[] args, WriterMain.SelfVerify verify) { ConformanceWriter::writeVortex, verify, t -> false); } + /** + * The Vortex write settings ({@code spec.FORMAT_SETTINGS["vortex"]}) this lane cannot honour: + * vortex-jni 0.86.1's writer takes only object-store options, no write strategy, so compact + * encodings and block sizes are refused rather than written with the defaults. + */ + static void refuseSettings(java.util.function.Function env) { + String compact = "RAINCLOUD_VORTEX_COMPACT"; + if (Boolean.TRUE.equals(WriteSettings.toggle(compact, env.apply(compact)))) { + throw WriteSettings.unsupported(CELL, compact, 1, "vortex-jni has no write strategy"); + } + for (String var : new String[] {"RAINCLOUD_VORTEX_ROW_BLOCK_ROWS", "RAINCLOUD_VORTEX_DATA_BLOCK_BYTES"}) { + Integer value = WriteSettings.count(var, env.apply(var)); + if (value != null) { + throw WriteSettings.unsupported(CELL, var, value, "vortex-jni has no block size setting"); + } + } + } + /** Stream the canonical into {@code output}; a failed write removes what it left. */ static void writeVortex(Path canonical, Path output, BufferAllocator allocator) throws IOException { + refuseSettings(System::getenv); Files.deleteIfExists(output); NativeLoader.loadJni(); boolean written = false; diff --git a/sidecars/nimble/CMakeLists.txt b/sidecars/nimble/CMakeLists.txt new file mode 100644 index 0000000..9381779 --- /dev/null +++ b/sidecars/nimble/CMakeLists.txt @@ -0,0 +1,41 @@ +# SPDX-FileCopyrightText: 2026 Raincloud Maintainers +# SPDX-License-Identifier: Apache-2.0 +# +# The nimble@cpp lane's binaries, grafted onto the pinned Nimble checkout's own CMake project. +# +# Not a standalone project: `build.sh` injects it into Nimble's `project()` call through +# `-DCMAKE_PROJECT_Nimble_INCLUDE`, so the binaries compile with exactly the flags, +# dependencies and targets Nimble itself is built with, and passes RAINCLOUD_NIMBLE_FFI, the +# Rust static library that runs the sidecar contract (sidecars/rust/nimble-ffi). None of +# Nimble's or Velox's targets exist yet at that point, so the targets are declared at the +# end of Nimble's top-level directory. + +set(RAINCLOUD_NIMBLE_DIR "${CMAKE_CURRENT_LIST_DIR}") +if(NOT RAINCLOUD_NIMBLE_FFI) + message(FATAL_ERROR "RAINCLOUD_NIMBLE_FFI must name libraincloud_nimble_ffi.a (see build.sh)") +endif() + +function(raincloud_nimble_targets) + find_package(Threads REQUIRED) + add_library(raincloud_nimble_lane STATIC "${RAINCLOUD_NIMBLE_DIR}/nimble_lane.cpp") + target_link_libraries( + raincloud_nimble_lane + PUBLIC + nimble_velox_writer + nimble_velox_reader + velox_arrow_bridge + velox_file + velox_vector + velox_memory + Folly::folly + ) + foreach(binary export read) + add_executable(raincloud-${binary}-nimble-cpp "${RAINCLOUD_NIMBLE_DIR}/${binary}_main.cpp") + target_link_libraries( + raincloud-${binary}-nimble-cpp + PRIVATE raincloud_nimble_lane "${RAINCLOUD_NIMBLE_FFI}" Threads::Threads ${CMAKE_DL_LIBS} m + ) + endforeach() +endfunction() + +cmake_language(DEFER CALL raincloud_nimble_targets) diff --git a/sidecars/nimble/build.sh b/sidecars/nimble/build.sh new file mode 100755 index 0000000..1749220 --- /dev/null +++ b/sidecars/nimble/build.sh @@ -0,0 +1,159 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: 2026 Raincloud Maintainers +# SPDX-License-Identifier: Apache-2.0 +# +# Build the `nimble@cpp` lane's binaries, raincloud-export-nimble-cpp and +# raincloud-read-nimble-cpp, inside a pinned Nimble checkout. Nimble has no releases and no +# packages, so it is built from source: Nimble at the commit below (upstream plus +# host-compatibility build fixes), its Velox and OpenZL submodules, the Folly and other +# dependencies Velox fetches, and four small libraries bootstrapped here. The binaries link +# raincloud_nimble_ffi (sidecars/rust/nimble-ffi), the Rust library that runs the sidecar +# contract, built here too. +# +# RAINCLOUD_NIMBLE_SRC= sidecars/nimble/build.sh +# +# RAINCLOUD_NIMBLE_SRC is a clone of the Nimble fork (https://github.com/mprammer/nimble, branch +# `raincloud`) at `nimble_pin`, with its submodules +# initialized; it is never edited. ROOT holds everything the build creates (about 4 GB) and +# must be on a disk, not tmpfs, and outside the repository and the checkout: +# +# downloads/ the pinned tarballs, sha256-verified +# bootstrap/ their sources and build trees deps/ their install prefix +# nimble/ the Nimble build tree (Velox's fetched dependencies land in its _deps/) +# cargo/ the Rust library's build tree +# bin/ the two binaries and nimble-cpp.provenance, written last +# +# A cold build needs the network, cargo and several minutes (about 8 at 8 jobs). Other +# environment: RAINCLOUD_NIMBLE_JOBS (default: half the CPUs); RAINCLOUD_BUILD_LOCK, a file +# the heavy steps take a shared flock on (waiting up to an hour), for a machine whose users +# coordinate long jobs that way (unset: no lock). Point RAINCLOUD_SIDECAR_NIMBLE_CPP and RAINCLOUD_READER_NIMBLE_CPP at the two +# binaries in ROOT/bin to use them. +set -euo pipefail + +# The Nimble commit: upstream acead744 (2026-08-07) and the build fixes on top of it. The +# submodule commits are the ones it records, checked against its gitlinks. +nimble_pin=b8ebbcbbbc5df5a6e9ea4988ec7a4090a52ecfb4 +velox_pin=351b0a72f446c75e0bd0ad179063a7a02f8894fc +openzl_pin=6b48fa4868160ed1e5c78ac422639615dd0dcf28 + +# Libraries the host lacks and Velox does not fetch: name, version, URL, sha256. libdwarf +# because Nimble calls Folly's symbolizer, which Folly builds only with it; FlatBuffers is +# Nimble's metadata runtime. +deps=( + "double-conversion 3.3.1 https://codeload.github.com/google/double-conversion/tar.gz/refs/tags/v3.3.1 fe54901055c71302dcdc5c3ccbe265a6c191978f3761ce1414d0895d6b0ea90e" + "fmt 11.2.0 https://codeload.github.com/fmtlib/fmt/tar.gz/refs/tags/11.2.0 bc23066d87ab3168f27cef3e97d545fa63314f5c79df5ea444d41d56f962c6af" + "libdwarf 0.12.0 https://github.com/davea42/libdwarf-code/releases/download/v0.12.0/libdwarf-0.12.0.tar.xz 444dc1c5176f04d3ebc50341552a8b2ea6c334f8f1868a023a740ace0e6eae9f" + "flatbuffers 25.12.19 https://codeload.github.com/google/flatbuffers/tar.gz/refs/tags/v25.12.19 f81c3162b1046fe8b84b9a0dbdd383e24fdbcf88583b9cb6028f90d04d90696a" +) + +die() { echo "build.sh: $*" >&2; exit 1; } +step() { echo; echo "==> $(date -u +%Y-%m-%dT%H:%M:%SZ) $*"; } + +# Every option before the lock file; -o so a daemon the command starts does not keep the lock. +build_lock=${RAINCLOUD_BUILD_LOCK:-} +quiet() { + local status=0 + if [[ -n $build_lock ]]; then + flock -E 75 -s -o -w 3600 "$build_lock" "$@" || status=$? + [[ $status -ne 75 ]] || die "timed out waiting for $build_lock to run: $*" + else + "$@" || status=$? + fi + [[ $status -eq 0 ]] || die "exit status $status from: $*" +} + +for tool in cmake ninja cc c++ cargo pkg-config git curl sha256sum tar xz; do + command -v "$tool" >/dev/null || die "missing prerequisite: $tool" +done +[[ $# -eq 1 && -n ${RAINCLOUD_NIMBLE_SRC:-} ]] || + die "usage: RAINCLOUD_NIMBLE_SRC= $0 " +here=$(cd "$(dirname "$0")" && pwd -P) +repo=$(cd "$here/../.." && pwd -P) +src=$(realpath -e -- "$RAINCLOUD_NIMBLE_SRC") || die "RAINCLOUD_NIMBLE_SRC=$RAINCLOUD_NIMBLE_SRC does not exist" +root=$(realpath -m -- "$1") +for tree in "$repo" "$src"; do + [[ $root != "$tree" && $root != "$tree"/* && $tree != "$root"/* ]] || die "ROOT $root overlaps $tree" +done +existing=$root +while [[ ! -e $existing ]]; do existing=$(dirname "$existing"); done +case $(stat -f -c %T -- "$existing") in tmpfs | ramfs) die "ROOT $root is on tmpfs; use a disk" ;; esac +jobs=${RAINCLOUD_NIMBLE_JOBS:-$(($(nproc) / 2))} + +step "checking the Nimble checkout $src" +[[ $(git -C "$src" rev-parse HEAD) == "$nimble_pin" ]] || die "$src is not at $nimble_pin" +gitlinks=$(git -C "$src" ls-tree HEAD velox openzl) +grep -qF "commit $velox_pin"$'\t'velox <<<"$gitlinks" || die "the pin does not record velox at $velox_pin" +grep -qF "commit $openzl_pin"$'\t'openzl <<<"$gitlinks" || die "the pin does not record openzl at $openzl_pin" +[[ $(git -C "$src/velox" rev-parse HEAD 2>/dev/null) == "$velox_pin" ]] || + die "velox is not checked out at $velox_pin (git -C $src submodule update --init --recursive)" +[[ $(git -C "$src/openzl" rev-parse HEAD 2>/dev/null) == "$openzl_pin" ]] || die "openzl is not checked out at $openzl_pin" +[[ -z $(git -C "$src" status --porcelain --untracked-files=no) ]] || die "$src has tracked modifications" + +mkdir -p "$root/downloads" "$root/bootstrap" "$root/deps" "$root/bin" +compilers=(-DCMAKE_C_COMPILER="$(command -v cc)" -DCMAKE_CXX_COMPILER="$(command -v c++)") +for dep in "${deps[@]}"; do + read -r name version url sha256 <<<"$dep" + flags=(-DCMAKE_POLICY_VERSION_MINIMUM=3.5 -DBUILD_SHARED_LIBS=OFF -DCMAKE_POSITION_INDEPENDENT_CODE=ON) + case $name in + double-conversion) flags+=(-DBUILD_TESTING=OFF) ;; + fmt) flags+=(-DFMT_TEST=OFF -DFMT_DOC=OFF) ;; + libdwarf) flags+=(-DBUILD_NON_SHARED=ON -DBUILD_SHARED=OFF -DPIC_ALWAYS=ON -DBUILD_DWARFDUMP=OFF) ;; + flatbuffers) flags+=(-DFLATBUFFERS_BUILD_TESTS=OFF) ;; + esac + tarball=$root/downloads/$name-$version.archive + stamp=$root/deps/.installed-$name + key=$(printf '%s\n' "$dep" "${flags[@]}" "${compilers[@]}" | sha256sum | cut -d' ' -f1) + if [[ ! -f $tarball ]]; then + step "downloading $name $version" + curl -fsSL --retry 3 -o "$tarball.part" "$url" + mv "$tarball.part" "$tarball" + fi + echo "$sha256 $tarball" | sha256sum -c --quiet - || die "$tarball does not match its pinned sha256" + [[ -f $stamp && $(<"$stamp") == "$key" ]] && continue + step "building $name $version" + rm -rf "$root/bootstrap/$name-$version" "$root/bootstrap/$name-build" + tar -xf "$tarball" -C "$root/bootstrap" + quiet cmake -S "$root/bootstrap/$name-$version" -B "$root/bootstrap/$name-build" -G Ninja \ + -DCMAKE_BUILD_TYPE=Release "${compilers[@]}" -DCMAKE_INSTALL_PREFIX="$root/deps" "${flags[@]}" + quiet cmake --build "$root/bootstrap/$name-build" -j "$jobs" + quiet cmake --install "$root/bootstrap/$name-build" + echo "$key" >"$stamp" +done + +step "building raincloud_nimble_ffi" +# Against the host's libzstd, which the C++ side links too: one zstd in the binary. +ffi=$root/cargo/release/libraincloud_nimble_ffi.a +quiet env ZSTD_SYS_USE_PKG_CONFIG=1 CARGO_TARGET_DIR="$root/cargo" \ + cargo build --release --locked --manifest-path "$repo/sidecars/rust/Cargo.toml" -p raincloud-nimble-ffi +[[ -f $ffi ]] || die "the cargo build did not produce $ffi" + +step "configuring Nimble into $root/nimble" +quiet cmake -S "$src" -B "$root/nimble" -G Ninja -DCMAKE_BUILD_TYPE=Release "${compilers[@]}" \ + -DCMAKE_PREFIX_PATH="$root/deps" \ + -DVELOX_ENABLE_GEO=OFF -DVELOX_BUILD_TESTING=OFF -DBUILD_TESTING=OFF \ + -DCMAKE_POLICY_VERSION_MINIMUM=3.5 -Dglog_SOURCE=BUNDLED \ + -DCMAKE_FIND_PACKAGE_TARGETS_GLOBAL=TRUE -DCMAKE_DISABLE_FIND_PACKAGE_LibUring=TRUE \ + -DVELOX_MONO_LIBRARY=OFF -DBUILD_SHARED_LIBS=OFF \ + -DCMAKE_PROJECT_Nimble_INCLUDE="$here/CMakeLists.txt" -DRAINCLOUD_NIMBLE_FFI="$ffi" + +binaries=(raincloud-export-nimble-cpp raincloud-read-nimble-cpp) +step "building ${binaries[*]}" +rm -f "$root/bin/nimble-cpp.provenance" +for binary in "${binaries[@]}"; do rm -f "$root/bin/$binary"; done +quiet cmake --build "$root/nimble" --target "${binaries[@]}" -j "$jobs" +for binary in "${binaries[@]}"; do + [[ -x $root/nimble/$binary ]] || die "the build did not produce $root/nimble/$binary" + cp "$root/nimble/$binary" "$root/bin/$binary.part" + mv "$root/bin/$binary.part" "$root/bin/$binary" +done +{ + echo "nimble $nimble_pin" + echo "velox $velox_pin" + echo "openzl $openzl_pin" + for dep in "${deps[@]}"; do read -r name version _ _ <<<"$dep"; echo "$name $version"; done + echo "compiler $(c++ --version | head -n 1)" + echo "rust $(rustc --version)" + echo "sources $(cd "$here" && sha256sum ./*.cpp ./*.h CMakeLists.txt build.sh | sha256sum | cut -c1-16)" + for binary in "${binaries[@]}"; do echo "sha256 $binary $(sha256sum "$root/bin/$binary" | cut -d' ' -f1)"; done +} >"$root/bin/nimble-cpp.provenance" +echo "built ${binaries[*]} in $root/bin" diff --git a/sidecars/nimble/export_main.cpp b/sidecars/nimble/export_main.cpp new file mode 100644 index 0000000..82ecda5 --- /dev/null +++ b/sidecars/nimble/export_main.cpp @@ -0,0 +1,8 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +// raincloud-export-nimble-cpp: raincloud's WRITE sidecar contract, over upstream Nimble. +#include "nimble_lane.h" + +int main(int argc, char** argv) { + return raincloud_nimble_write_main(argc, argv, nimble_lane_write, nimble_lane_read); +} diff --git a/sidecars/nimble/nimble_lane.cpp b/sidecars/nimble/nimble_lane.cpp new file mode 100644 index 0000000..0c3b94e --- /dev/null +++ b/sidecars/nimble/nimble_lane.cpp @@ -0,0 +1,176 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +// +// The nimble@cpp lane's two callbacks over upstream Nimble, for the sidecar contract the +// Rust library `raincloud_nimble_ffi` (sidecars/rust/nimble-ffi) runs: write an Arrow C +// stream to a Nimble file with `nimble::VeloxWriter` at default options, and read a Nimble +// file back as an Arrow C stream with `nimble::VeloxReader`, as the type the file records. +// Arrow crosses into Velox through Velox's own Arrow bridge; nothing converts a column, and +// a type the bridge or Nimble does not take is the callback's error. + +#include "nimble_lane.h" + +#include "dwio/nimble/velox/VeloxReader.h" +#include "dwio/nimble/writer/VeloxWriter.h" +#include "velox/common/base/VeloxException.h" +#include "velox/common/file/LocalFile.h" +#include "velox/common/memory/Memory.h" +#include "velox/vector/ComplexVector.h" +#include "velox/vector/arrow/Bridge.h" + +#include +#include +#include +#include +#include +#include + +namespace velox = facebook::velox; +namespace nimble = facebook::nimble; + +namespace { + +// Rows per batch the reader returns. Batch boundaries are not compared. +constexpr uint64_t kReadBatchRows = 64 * 1024; + +struct Pools { + std::shared_ptr root; + std::shared_ptr leaf; +}; + +const Pools& pools() { + static const Pools pools = [] { + velox::memory::initializeMemoryManager(velox::memory::MemoryManager::Options{}); + Pools made; + made.root = velox::memory::memoryManager()->addRootPool("nimble-lane"); + made.leaf = made.root->addLeafChild("leaf"); + return made; + }(); + return pools; +} + +// An exception's message, for the Rust side: a VeloxException's reason only (what() adds +// the source, code, context and a stack trace). +std::string describe(const std::exception& e) { + if (const auto* velox = dynamic_cast(&e)) return velox->message(); + return e.what(); +} + +int failed(char* error, size_t error_len, const std::string& message) { + if (error_len > 0) { + std::strncpy(error, message.c_str(), error_len - 1); + error[error_len - 1] = '\0'; + } + return 1; +} + +template +struct Released { + T value{}; + ~Released() { + if (value.release != nullptr) value.release(&value); + } +}; + +std::string streamError(ArrowArrayStream* stream, const char* what) { + const char* said = stream->get_last_error(stream); + return std::string(what) + (said != nullptr ? std::string(": ") + said : std::string()); +} + +// The stream `read` hands back: VeloxReader's batches, exported batch by batch. +struct ReadStream { + std::unique_ptr file; + std::unique_ptr reader; + std::string error; +}; + +int readSchema(ArrowArrayStream* stream, ArrowSchema* out) { + auto* self = static_cast(stream->private_data); + try { + velox::exportToArrow(velox::BaseVector::create(self->reader->type(), 0, pools().leaf.get()), *out); + return 0; + } catch (const std::exception& e) { + self->error = describe(e); + return EIO; + } +} + +int readNext(ArrowArrayStream* stream, ArrowArray* out) { + auto* self = static_cast(stream->private_data); + try { + velox::VectorPtr batch; + if (!self->reader->next(kReadBatchRows, batch)) { + out->release = nullptr; // end of stream + return 0; + } + // Every batch must match the schema get_schema declared, which is the plain type: a + // reader may hand back constant (an all-null column) or dictionary vectors, which + // Velox would export as run-end-encoded or dictionary arrays. Flat at every depth. + velox::BaseVector::flattenVector(batch); + velox::exportToArrow(batch, *out, pools().leaf.get()); + return 0; + } catch (const std::exception& e) { + self->error = describe(e); + return EIO; + } +} + +const char* readError(ArrowArrayStream* stream) { + auto* self = static_cast(stream->private_data); + return self->error.empty() ? nullptr : self->error.c_str(); +} + +void readRelease(ArrowArrayStream* stream) { + delete static_cast(stream->private_data); + stream->release = nullptr; +} + +} // namespace + +extern "C" int nimble_lane_write(ArrowArrayStream* input, const char* output, char* error, + size_t error_len) { + Released stream; + stream.value = *input; // ours now: released here, whatever happens + input->release = nullptr; + try { + Released schema; + if (stream.value.get_schema(&stream.value, &schema.value) != 0) { + throw std::runtime_error(streamError(&stream.value, "the canonical's schema")); + } + const auto type = velox::importFromArrow(schema.value); + nimble::VeloxWriter writer( + type, + std::make_unique(output, false, /*shouldThrowOnFileAlreadyExists=*/false), + *pools().root, nimble::VeloxWriterOptions{}); + while (true) { + Released array; + if (stream.value.get_next(&stream.value, &array.value) != 0) { + throw std::runtime_error(streamError(&stream.value, "a canonical batch")); + } + if (array.value.release == nullptr) break; // end of stream + // A view over the batch's buffers, which outlive the write that encodes them. + writer.write(velox::importFromArrowAsViewer(schema.value, array.value, pools().leaf.get())); + } + writer.close(); + return 0; + } catch (const std::exception& e) { + return failed(error, error_len, describe(e)); + } +} + +extern "C" int nimble_lane_read(const char* input, ArrowArrayStream* out, char* error, + size_t error_len) { + try { + auto self = std::make_unique(); + self->file = std::make_unique(input); + self->reader = std::make_unique(self->file.get(), *pools().leaf); + out->get_schema = readSchema; + out->get_next = readNext; + out->get_last_error = readError; + out->release = readRelease; + out->private_data = self.release(); + return 0; + } catch (const std::exception& e) { + return failed(error, error_len, describe(e)); + } +} diff --git a/sidecars/nimble/nimble_lane.h b/sidecars/nimble/nimble_lane.h new file mode 100644 index 0000000..edc2342 --- /dev/null +++ b/sidecars/nimble/nimble_lane.h @@ -0,0 +1,21 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +// +// The nimble@cpp lane: callbacks over upstream Nimble (nimble_lane.cpp) and the sidecar +// contract that runs them (raincloud_nimble_ffi, sidecars/rust/nimble-ffi/src/lib.rs). +#pragma once + +#include + +#include "velox/vector/arrow/Abi.h" + +extern "C" { +int nimble_lane_write(ArrowArrayStream* input, const char* output, char* error, size_t error_len); +int nimble_lane_read(const char* input, ArrowArrayStream* out, char* error, size_t error_len); + +typedef int (*nimble_lane_write_fn)(ArrowArrayStream*, const char*, char*, size_t); +typedef int (*nimble_lane_read_fn)(const char*, ArrowArrayStream*, char*, size_t); +int raincloud_nimble_write_main(int argc, const char* const* argv, nimble_lane_write_fn write, + nimble_lane_read_fn read); +int raincloud_nimble_read_main(int argc, const char* const* argv, nimble_lane_read_fn read); +} diff --git a/sidecars/nimble/read_main.cpp b/sidecars/nimble/read_main.cpp new file mode 100644 index 0000000..d56ed5b --- /dev/null +++ b/sidecars/nimble/read_main.cpp @@ -0,0 +1,6 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +// raincloud-read-nimble-cpp: raincloud's READ sidecar contract, over upstream Nimble. +#include "nimble_lane.h" + +int main(int argc, char** argv) { return raincloud_nimble_read_main(argc, argv, nimble_lane_read); } diff --git a/sidecars/rust/Cargo.lock b/sidecars/rust/Cargo.lock index 080ee21..2d42c2d 100644 --- a/sidecars/rust/Cargo.lock +++ b/sidecars/rust/Cargo.lock @@ -153,7 +153,10 @@ dependencies = [ "arrow-array", "arrow-buffer", "arrow-cast", + "arrow-csv", "arrow-data", + "arrow-ipc", + "arrow-json", "arrow-ord", "arrow-row", "arrow-schema", @@ -186,6 +189,7 @@ dependencies = [ "arrow-data", "arrow-schema", "chrono", + "chrono-tz", "half", "hashbrown 0.17.1", "libc", @@ -194,6 +198,30 @@ dependencies = [ "num-traits", ] +[[package]] +name = "arrow-avro" +version = "59.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9fb45cd6bd2b25c0965793b83200eaca82214273a8030fbbc2d783e4c7c65a61" +dependencies = [ + "arrow-array", + "arrow-buffer", + "arrow-schema", + "bytes", + "bzip2", + "crc", + "flate2", + "indexmap", + "liblzma", + "rand 0.9.5", + "serde", + "serde_json", + "snap", + "strum_macros", + "uuid", + "zstd", +] + [[package]] name = "arrow-buffer" version = "59.3.0" @@ -202,7 +230,7 @@ checksum = "097d193003ce7995d5d087089069ec2a6e0187faf5a6f8c9f38af2645d987182" dependencies = [ "bytes", "half", - "num-bigint", + "num-bigint 0.5.1", "num-traits", ] @@ -221,12 +249,28 @@ dependencies = [ "atoi", "base64", "chrono", + "comfy-table", "half", "lexical-core", "num-traits", "ryu", ] +[[package]] +name = "arrow-csv" +version = "59.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "25011b52b346407d497ef0030e12b45e4f2d0cc279efc09c4f3d09106db30e36" +dependencies = [ + "arrow-array", + "arrow-cast", + "arrow-schema", + "chrono", + "csv", + "csv-core", + "regex", +] + [[package]] name = "arrow-data" version = "59.2.0" @@ -252,9 +296,35 @@ dependencies = [ "arrow-schema", "arrow-select", "flatbuffers", + "lz4_flex 0.14.0", "zstd", ] +[[package]] +name = "arrow-json" +version = "59.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f18b9123ccfec418a663f821c9a034af339711678c11ffe00d3ec07da5ff9f7e" +dependencies = [ + "arrow-array", + "arrow-buffer", + "arrow-cast", + "arrow-ord", + "arrow-schema", + "arrow-select", + "chrono", + "half", + "indexmap", + "itoa", + "lexical-core", + "memchr", + "num-traits", + "ryu", + "serde_core", + "serde_json", + "simdutf8", +] + [[package]] name = "arrow-ord" version = "59.2.0" @@ -617,6 +687,15 @@ version = "1.12.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8ae3f5d315924270530207e2a68396c3cc547f6dca3fbdca317cfb1a51edb593" +[[package]] +name = "bzip2" +version = "0.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f3a53fac24f34a81bc9954b5d6cfce0c21e18ec6959f44f56e8e90e4bb7c346c" +dependencies = [ + "libbz2-rs-sys", +] + [[package]] name = "cc" version = "1.2.66" @@ -668,6 +747,16 @@ dependencies = [ "windows-link", ] +[[package]] +name = "chrono-tz" +version = "0.10.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a6139a8597ed92cf816dfb33f5dd6cf0bb93a6adc938f11039f371bc5bcd26c3" +dependencies = [ + "chrono", + "phf", +] + [[package]] name = "clang-sys" version = "1.8.1" @@ -725,6 +814,16 @@ version = "1.0.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1d07550c9036bf2ae0c684c4297d503f838287c83c53686d05370d0e139ae570" +[[package]] +name = "comfy-table" +version = "7.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "958c5d6ecf1f214b4c2bbbbf6ab9523a864bd136dcf71a7e8904799acfe1ad47" +dependencies = [ + "unicode-segmentation", + "unicode-width", +] + [[package]] name = "concurrent-queue" version = "2.5.0" @@ -775,6 +874,30 @@ dependencies = [ "libc", ] +[[package]] +name = "crc" +version = "3.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5eb8a2a1cd12ab0d987a5d5e825195d372001a4094a0376319d5a0ad71c1ba0d" +dependencies = [ + "crc-catalog", +] + +[[package]] +name = "crc-catalog" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "217698eaf96b4a3f0bc4f3662aaa55bdf913cd54d7204591faa790070c6d0853" + +[[package]] +name = "crc32fast" +version = "1.5.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "01a7799fd6b852db0e61728dde9a204c423b44d689dbd432522543614b490e78" +dependencies = [ + "cfg-if", +] + [[package]] name = "crossbeam-channel" version = "0.5.16" @@ -805,6 +928,27 @@ version = "0.2.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "460fbee9c2c2f33933d720630a6a0bac33ba7053db5344fac858d4b8952d77d5" +[[package]] +name = "csv" +version = "1.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "52cd9d68cf7efc6ddfaaee42e7288d3a99d613d4b50f76ce9827ae0c6e14f938" +dependencies = [ + "csv-core", + "itoa", + "ryu", + "serde_core", +] + +[[package]] +name = "csv-core" +version = "0.1.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "704a3c26996a80471189265814dbc2c257598b96b8a7feae2d31ace646bb9782" +dependencies = [ + "memchr", +] + [[package]] name = "custom-labels" version = "0.4.6" @@ -859,7 +1003,7 @@ version = "1.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "10d60334b3b2e7c9d91ef8150abfb6fa4c1c39ebbcf4a81c2e346aad939fee3e" dependencies = [ - "thiserror", + "thiserror 2.0.18", ] [[package]] @@ -971,6 +1115,12 @@ dependencies = [ "ext-trait", ] +[[package]] +name = "fallible-streaming-iterator" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7360491ce676a36bf9bb3c56c1aa791658183a54d2744120f27285738d90465a" + [[package]] name = "fastlanes" version = "0.6.1" @@ -1023,6 +1173,7 @@ version = "1.1.9" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "843fba2746e448b37e26a819579957415c8cef339bf08564fe8b7ddbd959573c" dependencies = [ + "crc32fast", "miniz_oxide", "zlib-rs", ] @@ -1614,6 +1765,12 @@ dependencies = [ "lexical-util", ] +[[package]] +name = "libbz2-rs-sys" +version = "0.2.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "34b357333733e8260735ba5894eb928c02ecc69c78715f01a8019e7fa7f2db4c" + [[package]] name = "libc" version = "0.2.186" @@ -1630,6 +1787,26 @@ dependencies = [ "windows-link", ] +[[package]] +name = "liblzma" +version = "0.4.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2fe0a34ca854fd4f20c07f696fc8675aec78f87d88d29f5e10257a7490a1b2e1" +dependencies = [ + "liblzma-sys", +] + +[[package]] +name = "liblzma-sys" +version = "0.4.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7c7e3581f367a78d7b7e7ae948d023310556f0cfc13156c2e4e00e25616492b9" +dependencies = [ + "cc", + "libc", + "pkg-config", +] + [[package]] name = "libm" version = "0.2.16" @@ -1663,6 +1840,15 @@ version = "0.4.33" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad" +[[package]] +name = "lz4_flex" +version = "0.11.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "373f5eceeeab7925e0c1098212f2fbc4d416adec9d35051a6ab251e824c1854a" +dependencies = [ + "twox-hash", +] + [[package]] name = "lz4_flex" version = "0.14.0" @@ -1672,6 +1858,16 @@ dependencies = [ "twox-hash", ] +[[package]] +name = "lzokay-native" +version = "0.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "792ba667add2798c6c3e988e630f4eb921b5cbc735044825b7111ef1582c8730" +dependencies = [ + "byteorder", + "thiserror 1.0.69", +] + [[package]] name = "macro_rules_attribute" version = "0.1.3" @@ -1767,6 +1963,30 @@ dependencies = [ "syn 1.0.109", ] +[[package]] +name = "num" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "35bd024e8b2ff75562e5f34e7f4905839deb4b22955ef5e73d2fea1b9813cb23" +dependencies = [ + "num-bigint 0.4.8", + "num-complex", + "num-integer", + "num-iter", + "num-rational", + "num-traits", +] + +[[package]] +name = "num-bigint" +version = "0.4.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c89e69e7e0f03bea5ef08013795c25018e101932225a656383bd384495ecc367" +dependencies = [ + "num-integer", + "num-traits", +] + [[package]] name = "num-bigint" version = "0.5.1" @@ -1795,6 +2015,27 @@ dependencies = [ "num-traits", ] +[[package]] +name = "num-iter" +version = "0.1.46" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c92800bd69a1eac91786bcfe9da64a897eb72911b8dc3095decbd07429e8048b" +dependencies = [ + "num-integer", + "num-traits", +] + +[[package]] +name = "num-rational" +version = "0.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f83d14da390562dca69fc84082e73e548e1ad308d24accdedd2720017cb37824" +dependencies = [ + "num-bigint 0.4.8", + "num-integer", + "num-traits", +] + [[package]] name = "num-traits" version = "0.2.19" @@ -1851,6 +2092,29 @@ dependencies = [ "rand 0.9.5", ] +[[package]] +name = "orc-rust" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e54c163f843c4f0fcfca65b9b73ca6f93c37c8e86736d5a1e578c6e843eed0a0" +dependencies = [ + "arrow", + "bytemuck", + "bytes", + "chrono", + "chrono-tz", + "fallible-streaming-iterator", + "flate2", + "log", + "lz4_flex 0.11.6", + "lzokay-native", + "num", + "prost 0.13.5", + "snafu", + "snap", + "zstd", +] + [[package]] name = "parking" version = "2.2.1" @@ -1900,8 +2164,8 @@ dependencies = [ "flate2", "half", "hashbrown 0.17.1", - "lz4_flex", - "num-bigint", + "lz4_flex 0.14.0", + "num-bigint 0.5.1", "num-integer", "num-traits", "parquet-variant", @@ -1991,6 +2255,24 @@ version = "2.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" +[[package]] +name = "phf" +version = "0.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "913273894cec178f401a31ec4b656318d95473527be05c0752cc41cdc32be8b7" +dependencies = [ + "phf_shared", +] + +[[package]] +name = "phf_shared" +version = "0.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "06005508882fb681fd97892ecff4b7fd0fee13ef1aa569f8695dae7ab9099981" +dependencies = [ + "siphasher", +] + [[package]] name = "pin-project-lite" version = "0.2.17" @@ -2076,6 +2358,16 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "prost" +version = "0.13.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2796faa41db3ec313a31f7624d9286acf277b52de526150b7e69f3debf891ee5" +dependencies = [ + "bytes", + "prost-derive 0.13.5", +] + [[package]] name = "prost" version = "0.14.4" @@ -2083,7 +2375,20 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "528ac67416ff8646872a3c02cad9cc4ee5dc9f9540c9b10771855c95cb2e5ae1" dependencies = [ "bytes", - "prost-derive", + "prost-derive 0.14.4", +] + +[[package]] +name = "prost-derive" +version = "0.13.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8a56d757972c98b346a9b766e3f02746cde6dd1cd1d1d563472929fdd74bec4d" +dependencies = [ + "anyhow", + "itertools 0.14.0", + "proc-macro2", + "quote", + "syn 2.0.118", ] [[package]] @@ -2105,7 +2410,7 @@ version = "0.14.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f94967dc7688f3054c7fac87473ffae4cc4c3904800e2d9f5b857246d8963b0a" dependencies = [ - "prost", + "prost 0.14.4", ] [[package]] @@ -2135,18 +2440,30 @@ version = "0.7.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "dc33ff2d4973d518d823d61aa239014831e521c75da58e3df4840d3f47749d09" +[[package]] +name = "raincloud-nimble-ffi" +version = "0.1.0" +dependencies = [ + "anyhow", + "arrow-array", + "clap", + "raincloud-sidecars", +] + [[package]] name = "raincloud-sidecars" version = "0.1.0" dependencies = [ "anyhow", "arrow-array", + "arrow-avro", "arrow-cast", "arrow-ipc", "arrow-schema", "arrow-select", "clap", "futures", + "orc-rust", "parquet", "serde_json", "tokio", @@ -2397,6 +2714,12 @@ version = "0.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e" +[[package]] +name = "siphasher" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "33f4fe9184a62d842c9ef383018f3306d8ba224fd9d836f56d7288308847c256" + [[package]] name = "sketches-ddsketch" version = "0.4.0" @@ -2432,6 +2755,27 @@ dependencies = [ "futures-lite", ] +[[package]] +name = "snafu" +version = "0.8.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6e84b3f4eacbf3a1ce05eac6763b4d629d60cbc94d632e4092c54ade71f1e1a2" +dependencies = [ + "snafu-derive", +] + +[[package]] +name = "snafu-derive" +version = "0.8.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c1c97747dbf44bb1ca44a561ece23508e99cb592e862f22222dcf42f51d1e451" +dependencies = [ + "heck", + "proc-macro2", + "quote", + "syn 2.0.118", +] + [[package]] name = "snap" version = "1.1.1" @@ -2456,6 +2800,18 @@ version = "0.11.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" +[[package]] +name = "strum_macros" +version = "0.28.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ab85eea0270ee17587ed4156089e10b9e6880ee688791d45a905f5b1ca36f664" +dependencies = [ + "heck", + "proc-macro2", + "quote", + "syn 2.0.118", +] + [[package]] name = "syn" version = "1.0.109" @@ -2518,13 +2874,33 @@ version = "1.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d4d1330fe7f7f872cd05165130b10602d667b205fd85be09be2814b115d4ced9" +[[package]] +name = "thiserror" +version = "1.0.69" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6aaf5339b578ea85b50e080feb250a3e8ae8cfcdff9a461c9ec2904bc923f52" +dependencies = [ + "thiserror-impl 1.0.69", +] + [[package]] name = "thiserror" version = "2.0.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4288b5bcbc7920c07a1149a35cf9590a2aa808e0bc1eafaade0b80947865fbc4" dependencies = [ - "thiserror-impl", + "thiserror-impl 2.0.18", +] + +[[package]] +name = "thiserror-impl" +version = "1.0.69" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4fee6c4efc90059e10f81e6d42c60a18f76588c3d74cb83a0b242a2b6c7504c1" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.118", ] [[package]] @@ -2613,6 +2989,18 @@ version = "1.0.24" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75" +[[package]] +name = "unicode-segmentation" +version = "1.13.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c6f5d3c3b1bf09027a88a6bc961fc00497d651009560b5463668dc81b0fa87a8" + +[[package]] +name = "unicode-width" +version = "0.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b4ac048d71ede7ee76d585517add45da530660ef4390e49b098733c6e897f254" + [[package]] name = "url" version = "2.5.8" @@ -2701,7 +3089,7 @@ dependencies = [ "alp", "itertools 0.14.0", "num-traits", - "prost", + "prost 0.14.4", "vortex-array", "vortex-buffer", "vortex-error", @@ -2736,7 +3124,7 @@ dependencies = [ "parking_lot", "paste", "pin-project-lite", - "prost", + "prost 0.14.4", "rand 0.10.2", "regex", "regex-syntax", @@ -2883,7 +3271,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c1feecd7ce3b66c03370f7a5450a15c759cd0a823c1fcde577c358287a902041" dependencies = [ "num-traits", - "prost", + "prost 0.14.4", "vortex-array", "vortex-buffer", "vortex-error", @@ -2898,7 +3286,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8cdd8ccb809d9bf839c1e3cd1b1adc5fd2a2ed9d3d74bb13764b73a0412fc064" dependencies = [ "num-traits", - "prost", + "prost 0.14.4", "vortex-array", "vortex-buffer", "vortex-error", @@ -2926,7 +3314,7 @@ dependencies = [ "arrow-schema", "flatbuffers", "jiff", - "prost", + "prost 0.14.4", "tokio", ] @@ -2940,7 +3328,7 @@ dependencies = [ "itertools 0.14.0", "lending-iterator", "num-traits", - "prost", + "prost 0.14.4", "vortex-array", "vortex-buffer", "vortex-error", @@ -3014,7 +3402,7 @@ checksum = "612d95a0d18cac5e50318afc59bd41a24a384558c847bfda8797eff274b83a7f" dependencies = [ "fsst-rs", "num-traits", - "prost", + "prost 0.14.4", "vortex-array", "vortex-buffer", "vortex-error", @@ -3078,7 +3466,7 @@ checksum = "3ce0763be583581780f7f2ed8bb799d82b9084618dac19be05e460ca10ceafbb" dependencies = [ "arrow-array", "arrow-schema", - "prost", + "prost 0.14.4", "vortex-array", "vortex-arrow", "vortex-edition", @@ -3108,7 +3496,7 @@ dependencies = [ "parking_lot", "paste", "pin-project-lite", - "prost", + "prost 0.14.4", "rustc-hash", "sketches-ddsketch", "termtree", @@ -3161,7 +3549,7 @@ checksum = "a4f3946d4ce6ebe42253a51ce4d65769e3a0243db3419bf44a60b2a9cb74b26c" dependencies = [ "num-traits", "onpair", - "prost", + "prost 0.14.4", "vortex-array", "vortex-buffer", "vortex-error", @@ -3181,7 +3569,7 @@ dependencies = [ "chrono", "parquet-variant", "parquet-variant-compute", - "prost", + "prost 0.14.4", "vortex-array", "vortex-arrow", "vortex-buffer", @@ -3199,7 +3587,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1113d8543d967d442e5c035392d7c7e65ae5b18267e6fac5f210cc30b7d42fd6" dependencies = [ "pco", - "prost", + "prost 0.14.4", "vortex-array", "vortex-buffer", "vortex-error", @@ -3213,7 +3601,7 @@ version = "0.86.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "95ed2db063cc759e9461fc40a6da95ddf17f2d8af8948a3fe3b5079e1d5f9fae" dependencies = [ - "prost", + "prost 0.14.4", "prost-types", ] @@ -3225,7 +3613,7 @@ checksum = "9614791732d7b4f13036318685808cb3a285b249895b0142fb92f1add57643c2" dependencies = [ "itertools 0.14.0", "num-traits", - "prost", + "prost 0.14.4", "vortex-array", "vortex-buffer", "vortex-error", @@ -3257,7 +3645,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a864402ea44d12e6daae1d63a9d4b65a902ebea7f12436e84066c2bffbdcbd1b" dependencies = [ "num-traits", - "prost", + "prost 0.14.4", "smallvec", "vortex-array", "vortex-buffer", @@ -3288,7 +3676,7 @@ checksum = "a58a99b9033e19bd3dabbbe18fe33cabc99b4ceb4b30329baea317e734e2ddf0" dependencies = [ "itertools 0.14.0", "num-traits", - "prost", + "prost 0.14.4", "vortex-array", "vortex-buffer", "vortex-error", @@ -3328,7 +3716,7 @@ checksum = "c5b49914d9a3b1bcba92a7e79a4cc3e2136f2ced349cbe91ec7eb2d30bc9dec4" dependencies = [ "itertools 0.14.0", "num-traits", - "prost", + "prost 0.14.4", "vortex-array", "vortex-buffer", "vortex-edition", diff --git a/sidecars/rust/Cargo.toml b/sidecars/rust/Cargo.toml index dc2a8f3..28c3a7a 100644 --- a/sidecars/rust/Cargo.toml +++ b/sidecars/rust/Cargo.toml @@ -5,13 +5,17 @@ name = "raincloud-sidecars" version = "0.1.0" edition = "2021" publish = false -description = "Raincloud cross-implementation sidecar binaries: parquet via arrow-rs, vortex via the Vortex Rust core." +description = "Raincloud cross-implementation sidecar binaries: parquet via arrow-rs, vortex via the Vortex Rust core, orc via orc-rust, avro via arrow-avro." # Standalone crate — NOT a member of the Vortex workspace: it lives in the -# raincloud repo and depends on the PUBLISHED Vortex crate. An empty -# [workspace] table makes this its own workspace root so no parent Cargo.toml -# can absorb it and so it never edits a local vortex checkout. +# raincloud repo and depends on the PUBLISHED Vortex crate. This [workspace] +# table makes it its own workspace root, so no parent Cargo.toml can absorb it +# and it never edits a local vortex checkout. Its one other member is +# nimble-ffi, the static library the nimble@cpp binaries (sidecars/nimble) link; +# a plain build builds only this package. [workspace] +members = ["nimble-ffi"] +default-members = ["."] [lib] name = "raincloud_sidecars" @@ -33,6 +37,22 @@ path = "src/bin/vortex_write.rs" name = "vortex-read" path = "src/bin/vortex_read.rs" +[[bin]] +name = "orc-write" +path = "src/bin/orc_write.rs" + +[[bin]] +name = "orc-read" +path = "src/bin/orc_read.rs" + +[[bin]] +name = "avro-write" +path = "src/bin/avro_write.rs" + +[[bin]] +name = "avro-read" +path = "src/bin/avro_read.rs" + [dependencies] # Vortex Rust core, from crates.io. PINNED EXACTLY, and held in LOCKSTEP with # the other two Vortex lanes so a differing conformance verdict is an @@ -61,6 +81,16 @@ arrow-select = "=59.2" arrow-ipc = { version = "=59.2", features = ["zstd"] } parquet = { version = "=59.2", features = ["variant_experimental"] } +# ORC via orc-rust, the Arrow-native Rust implementation, PINNED EXACTLY. Its +# 0.9 line is built on arrow 59, so it shares the =59.2 arrow above. The +# default `async` feature is not needed: the sidecar reads and writes files. +orc-rust = { version = "=0.9.0", default-features = false } + +# Avro object container files via arrow-avro, released with arrow-rs and pinned +# with it (=59.2). Every codec Avro defines, so RAINCLOUD_AVRO_COMPRESSION can +# name any of them (zstd unset). +arrow-avro = { version = "=59.2", default-features = false, features = ["zstd", "deflate", "snappy", "bzip2", "xz"] } + # CLI + JSON report + async runtime + stream adapter + error plumbing. clap = { version = "4.5", features = ["derive"] } serde_json = "1" diff --git a/sidecars/rust/nimble-ffi/Cargo.toml b/sidecars/rust/nimble-ffi/Cargo.toml new file mode 100644 index 0000000..1e976a5 --- /dev/null +++ b/sidecars/rust/nimble-ffi/Cargo.toml @@ -0,0 +1,20 @@ +# SPDX-FileCopyrightText: 2026 Raincloud Maintainers +# SPDX-License-Identifier: Apache-2.0 +[package] +name = "raincloud-nimble-ffi" +version = "0.1.0" +edition = "2021" +publish = false +description = "The nimble@cpp lane's sidecar contract, as a static library the C++ binaries in sidecars/nimble link." + +[lib] +name = "raincloud_nimble_ffi" +path = "src/lib.rs" +crate-type = ["staticlib"] + +[dependencies] +raincloud-sidecars = { path = ".." } +# Pinned with the sidecar crate's arrow (=59.2); `ffi` is the C stream interface. +arrow-array = { version = "=59.2", features = ["ffi"] } +anyhow = "1" +clap = { version = "4.5", features = ["derive"] } diff --git a/sidecars/rust/nimble-ffi/src/lib.rs b/sidecars/rust/nimble-ffi/src/lib.rs new file mode 100644 index 0000000..29786f5 --- /dev/null +++ b/sidecars/rust/nimble-ffi/src/lib.rs @@ -0,0 +1,203 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 + +//! The `nimble@cpp` lane's sidecar contract, for its C++ binaries. +//! +//! Nimble is C++ only, so the lane's binaries (`sidecars/nimble`) are C++: they +//! link this library and call [`raincloud_nimble_write_main`] or +//! [`raincloud_nimble_read_main`] with two callbacks over upstream Nimble's +//! `VeloxWriter` and `VeloxReader`. Everything else is the Rust lanes' own: +//! the CLI, reading the canonical with arrow-rs, the logical comparison and the +//! report. Batches cross in memory through the Arrow C stream interface, with +//! nothing converted on the way. +//! +//! A callback returns 0, or non-zero with a NUL-terminated message in `error` +//! (at most `error_len` bytes). It takes ownership of a stream it is handed and +//! fills a stream it is given with one the caller then owns. + +use std::ffi::{c_char, c_int, CStr, OsString}; +use std::os::unix::ffi::OsStringExt; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; + +use anyhow::{bail, Context, Result}; +use arrow_array::ffi_stream::{ArrowArrayStreamReader, FFI_ArrowArrayStream}; +use arrow_array::RecordBatchReader; +use clap::Parser; +use raincloud_sidecars::{ + canonical_batches, has_variant, logical_eq_stream, open_canonical, run_reader, run_writer, +}; + +/// Write the batches of `stream` to the Nimble file at `output`. +pub type WriteFn = unsafe extern "C" fn( + stream: *mut FFI_ArrowArrayStream, + output: *const c_char, + error: *mut c_char, + error_len: usize, +) -> c_int; + +/// Fill `out` with a stream of the Nimble file at `input`'s batches. +pub type ReadFn = unsafe extern "C" fn( + input: *const c_char, + out: *mut FFI_ArrowArrayStream, + error: *mut c_char, + error_len: usize, +) -> c_int; + +const CELL: &str = "nimble@cpp"; + +#[derive(Parser)] +#[command(about = "Write a Nimble artifact from the canonical Arrow IPC file (upstream Nimble).")] +struct WriteArgs { + /// Canonical Arrow IPC file (`.arrow.zstd`). + #[arg(long)] + input: PathBuf, + /// Destination `.nimble` path. + #[arg(long)] + output: PathBuf, + /// JSON report path. + #[arg(long)] + report: PathBuf, +} + +#[derive(Parser)] +#[command( + about = "Read a Nimble artifact (upstream Nimble) and verdict vs the canonical Arrow IPC file." +)] +struct ReadArgs { + /// `.nimble` artifact to read. + #[arg(long)] + input: PathBuf, + /// Canonical Arrow IPC file (`.arrow.zstd`). + #[arg(long)] + canonical: PathBuf, + /// JSON report path. + #[arg(long)] + report: PathBuf, +} + +/// `argc`/`argv` as clap reads them. +unsafe fn arguments(argc: c_int, argv: *const *const c_char) -> Vec { + (0..argc.max(0) as usize) + .map(|i| OsString::from_vec(CStr::from_ptr(*argv.add(i)).to_bytes().to_vec())) + .collect() +} + +fn c_path(path: &Path) -> Result { + use std::os::unix::ffi::OsStrExt; + std::ffi::CString::new(path.as_os_str().as_bytes()).context("a path with a NUL byte") +} + +/// Run a callback, turning its non-zero status into an error carrying its message. +fn call(what: &str, f: impl FnOnce(*mut c_char, usize) -> c_int) -> Result<()> { + let mut error = vec![0 as c_char; 4096]; + if f(error.as_mut_ptr(), error.len()) == 0 { + return Ok(()); + } + let last = error.len() - 1; + error[last] = 0; + let message = unsafe { CStr::from_ptr(error.as_ptr()) } + .to_string_lossy() + .into_owned(); + bail!( + "{what}: {}", + if message.is_empty() { + "failed" + } else { + &message + } + ) +} + +/// The Nimble file at `path`, read back by upstream Nimble through `read`. +fn open_nimble(read: ReadFn, path: &Path) -> Result { + let input = c_path(path)?; + let mut stream = FFI_ArrowArrayStream::empty(); + call("upstream Nimble read", |error, len| unsafe { + read(input.as_ptr(), &mut stream, error, len) + })?; + ArrowArrayStreamReader::try_new(stream).context("the stream Nimble read back") +} + +fn compare(read: ReadFn, nimble: &Path, canonical: &Path) -> Result<(bool, String)> { + let (schema, expected) = open_canonical(canonical)?; + let got = open_nimble(read, nimble)?; + let got_schema = got.schema(); + logical_eq_stream( + &got_schema, + got.map(|b| b.context("read a batch Nimble read back")), + &schema, + canonical_batches(expected), + ) +} + +/// The `raincloud-export-nimble-cpp` binary: write the canonical with `write`, +/// then self-verify through `read` (raincloud's WRITE CLI contract). +/// +/// # Safety +/// `argv` holds `argc` NUL-terminated strings; the callbacks keep the contract above. +#[no_mangle] +pub unsafe extern "C" fn raincloud_nimble_write_main( + argc: c_int, + argv: *const *const c_char, + write: WriteFn, + read: ReadFn, +) -> c_int { + let args = match WriteArgs::try_parse_from(arguments(argc, argv)) { + Ok(args) => args, + Err(e) => { + let _ = e.print(); + return 2; + } + }; + exit(run_writer(CELL, &args.output, &args.report, || { + let (schema, reader) = open_canonical(&args.input)?; + let variant_faithful = !has_variant(&schema); + let output = c_path(&args.output)?; + let batches: Box = Box::new(reader); + let mut stream = FFI_ArrowArrayStream::new(batches); + call("upstream Nimble write", |error, len| unsafe { + write(&mut stream, output.as_ptr(), error, len) + })?; + let (roundtrip, detail) = compare(read, &args.output, &args.input)?; + let note = if !roundtrip { + format!("{CELL}: self-verify mismatch: {detail}") + } else if variant_faithful { + format!("{CELL}: round-trips to canonical") + } else { + format!("{CELL}: round-trips; VARIANT annotation not preserved (Nimble has no VARIANT type)") + }; + Ok((roundtrip, variant_faithful, note)) + })) +} + +/// The `raincloud-read-nimble-cpp` binary: read a Nimble file with `read` and +/// verdict it against the canonical (raincloud's READ CLI contract). +/// +/// # Safety +/// As for [`raincloud_nimble_write_main`]. +#[no_mangle] +pub unsafe extern "C" fn raincloud_nimble_read_main( + argc: c_int, + argv: *const *const c_char, + read: ReadFn, +) -> c_int { + let args = match ReadArgs::try_parse_from(arguments(argc, argv)) { + Ok(args) => args, + Err(e) => { + let _ = e.print(); + return 2; + } + }; + exit(run_reader(CELL, &args.report, || { + compare(read, &args.input, &args.canonical) + })) +} + +fn exit(code: ExitCode) -> c_int { + if code == ExitCode::SUCCESS { + 0 + } else { + 1 + } +} diff --git a/sidecars/rust/src/bin/avro_read.rs b/sidecars/rust/src/bin/avro_read.rs new file mode 100644 index 0000000..718ffa2 --- /dev/null +++ b/sidecars/rust/src/bin/avro_read.rs @@ -0,0 +1,38 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +//! `avro-read` sidecar — read an Avro artifact via arrow-avro and verdict its +//! round-trip to the canonical. Implements raincloud's READ CLI contract +//! (`raincloud/pipeline/export/readers.py`). + +use std::path::PathBuf; +use std::process::ExitCode; + +use clap::Parser; +use raincloud_sidecars::{ + canonical_batches, logical_eq_stream, open_avro, open_canonical, run_reader, +}; + +#[derive(Parser)] +#[command( + about = "Read an Avro artifact (arrow-avro) and verdict vs the canonical Arrow IPC file." +)] +struct Args { + /// `.avro` artifact to read. + #[arg(long)] + input: PathBuf, + /// Canonical Arrow IPC file (`.arrow.zstd`). + #[arg(long)] + canonical: PathBuf, + /// JSON report path. + #[arg(long)] + report: PathBuf, +} + +fn main() -> ExitCode { + let args = Args::parse(); + run_reader("avro@rs", &args.report, || { + let (schema, canonical) = open_canonical(&args.canonical)?; + let (got_schema, got) = open_avro(&args.input)?; + logical_eq_stream(&got_schema, got, &schema, canonical_batches(canonical)) + }) +} diff --git a/sidecars/rust/src/bin/avro_write.rs b/sidecars/rust/src/bin/avro_write.rs new file mode 100644 index 0000000..f7c080a --- /dev/null +++ b/sidecars/rust/src/bin/avro_write.rs @@ -0,0 +1,50 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +//! `avro-write` sidecar — write an Avro object container file from the canonical via arrow-avro, +//! then self-verify. Implements raincloud's WRITE CLI contract +//! (`raincloud/pipeline/export/sidecar.py`). + +use std::path::PathBuf; +use std::process::ExitCode; + +use clap::Parser; +use raincloud_sidecars::{ + canonical_batches, has_variant, logical_eq_stream, open_avro, open_canonical, run_writer, + write_avro, +}; + +#[derive(Parser)] +#[command(about = "Write an Avro artifact from the canonical Arrow IPC file (arrow-avro).")] +struct Args { + /// Canonical Arrow IPC file (`.arrow.zstd`). + #[arg(long)] + input: PathBuf, + /// Destination `.avro` path. + #[arg(long)] + output: PathBuf, + /// JSON report path. + #[arg(long)] + report: PathBuf, +} + +fn main() -> ExitCode { + let args = Args::parse(); + run_writer("avro@rs", &args.output, &args.report, || { + write_avro(&args.output, &args.input)?; + + let (schema, canonical) = open_canonical(&args.input)?; + let (got_schema, got) = open_avro(&args.output)?; + let (roundtrip, detail) = + logical_eq_stream(&got_schema, got, &schema, canonical_batches(canonical))?; + let variant_faithful = !has_variant(&schema); + let note = if !roundtrip { + format!("avro@rs: self-verify mismatch: {detail}") + } else if variant_faithful { + "avro@rs: round-trips to canonical".to_string() + } else { + "avro@rs: round-trips; VARIANT annotation not preserved (Avro has no VARIANT type)" + .to_string() + }; + Ok((roundtrip, variant_faithful, note)) + }) +} diff --git a/sidecars/rust/src/bin/orc_read.rs b/sidecars/rust/src/bin/orc_read.rs new file mode 100644 index 0000000..a08669c --- /dev/null +++ b/sidecars/rust/src/bin/orc_read.rs @@ -0,0 +1,36 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +//! `orc-read` sidecar — read an `.orc` artifact via orc-rust and verdict its +//! round-trip to the canonical. Implements raincloud's READ CLI contract +//! (`raincloud/pipeline/export/readers.py`). + +use std::path::PathBuf; +use std::process::ExitCode; + +use clap::Parser; +use raincloud_sidecars::{ + canonical_batches, logical_eq_stream, open_canonical, open_orc, run_reader, +}; + +#[derive(Parser)] +#[command(about = "Read an ORC artifact (orc-rust) and verdict vs the canonical Arrow IPC file.")] +struct Args { + /// `.orc` artifact to read. + #[arg(long)] + input: PathBuf, + /// Canonical Arrow IPC file (`.arrow.zstd`). + #[arg(long)] + canonical: PathBuf, + /// JSON report path. + #[arg(long)] + report: PathBuf, +} + +fn main() -> ExitCode { + let args = Args::parse(); + run_reader("orc@rs", &args.report, || { + let (schema, canonical) = open_canonical(&args.canonical)?; + let (got_schema, got) = open_orc(&args.input)?; + logical_eq_stream(&got_schema, got, &schema, canonical_batches(canonical)) + }) +} diff --git a/sidecars/rust/src/bin/orc_write.rs b/sidecars/rust/src/bin/orc_write.rs new file mode 100644 index 0000000..5b0c2c3 --- /dev/null +++ b/sidecars/rust/src/bin/orc_write.rs @@ -0,0 +1,50 @@ +// SPDX-FileCopyrightText: 2026 Raincloud Maintainers +// SPDX-License-Identifier: Apache-2.0 +//! `orc-write` sidecar — write an `.orc` file from the canonical via orc-rust, +//! then self-verify. Implements raincloud's WRITE CLI contract +//! (`raincloud/pipeline/export/sidecar.py`). + +use std::path::PathBuf; +use std::process::ExitCode; + +use clap::Parser; +use raincloud_sidecars::{ + canonical_batches, has_variant, logical_eq_stream, open_canonical, open_orc, run_writer, + write_orc, +}; + +#[derive(Parser)] +#[command(about = "Write an ORC artifact from the canonical Arrow IPC file (orc-rust).")] +struct Args { + /// Canonical Arrow IPC file (`.arrow.zstd`). + #[arg(long)] + input: PathBuf, + /// Destination `.orc` path. + #[arg(long)] + output: PathBuf, + /// JSON report path. + #[arg(long)] + report: PathBuf, +} + +fn main() -> ExitCode { + let args = Args::parse(); + run_writer("orc@rs", &args.output, &args.report, || { + write_orc(&args.output, &args.input)?; + + let (schema, canonical) = open_canonical(&args.input)?; + let (got_schema, got) = open_orc(&args.output)?; + let (roundtrip, detail) = + logical_eq_stream(&got_schema, got, &schema, canonical_batches(canonical))?; + let variant_faithful = !has_variant(&schema); + let note = if !roundtrip { + format!("orc@rs: self-verify mismatch: {detail}") + } else if variant_faithful { + "orc@rs: round-trips to canonical".to_string() + } else { + "orc@rs: round-trips; VARIANT annotation not preserved (ORC has no VARIANT type)" + .to_string() + }; + Ok((roundtrip, variant_faithful, note)) + }) +} diff --git a/sidecars/rust/src/bin/parquet_write.rs b/sidecars/rust/src/bin/parquet_write.rs index 32d705e..32968bf 100644 --- a/sidecars/rust/src/bin/parquet_write.rs +++ b/sidecars/rust/src/bin/parquet_write.rs @@ -10,7 +10,7 @@ use std::process::ExitCode; use clap::Parser; use raincloud_sidecars::{ canonical_batches, logical_eq_stream, open_canonical, open_parquet, parquet_variant_loss, - run_writer, variant_columns, write_parquet, RowGroupLimits, + run_writer, variant_columns, write_parquet_with, ParquetOptions, RowGroupLimits, }; #[derive(Parser)] @@ -33,10 +33,11 @@ fn main() -> ExitCode { // Knobs first, so a malformed one fails before this run writes any // output. (Any error removes `--output`, whoever wrote it.) let limits = RowGroupLimits::from_env()?; + let options = ParquetOptions::from_env()?; // Streamed end to end: the canonical is re-read rather than held, since a // large one (SF100 lineitem, ~100 GB decoded) does not fit in memory. let (schema, _) = open_canonical(&args.input)?; - write_parquet(&args.output, schema.clone(), limits, || { + write_parquet_with(&args.output, schema.clone(), limits, options, || { Ok(canonical_batches(open_canonical(&args.input)?.1)) })?; diff --git a/sidecars/rust/src/lib.rs b/sidecars/rust/src/lib.rs index 9575379..f570591 100644 --- a/sidecars/rust/src/lib.rs +++ b/sidecars/rust/src/lib.rs @@ -3,15 +3,19 @@ //! Shared machinery for raincloud's Rust sidecar binaries. //! -//! Four `[[bin]]` targets implement raincloud's fixed sidecar CLI contracts +//! The `[[bin]]` targets implement raincloud's fixed sidecar CLI contracts //! (see `raincloud/pipeline/export/sidecar.py` for WRITE and //! `raincloud/pipeline/export/readers.py` for READ): //! -//! * `parquet-write` / `vortex-write` — read the canonical Arrow IPC file, write -//! the target format, then SELF-VERIFY (re-read, compare) and emit -//! `{"roundtrip", "variant_faithful", "note"}`. -//! * `parquet-read` / `vortex-read` — read an artifact, compare to the canonical -//! with LOGICAL equality, emit `{"status", "note", "detail"}`. +//! * `parquet-write` / `vortex-write` / `orc-write` / `avro-write` — read the +//! canonical Arrow IPC file, write the target format, then SELF-VERIFY +//! (re-read, compare) and emit `{"roundtrip", "variant_faithful", "note"}`. +//! * `parquet-read` / `vortex-read` / `orc-read` / `avro-read` — read an +//! artifact, compare to the canonical with LOGICAL equality, emit +//! `{"status", "note", "detail"}`. +//! +//! The `nimble-ffi` member exposes the same contract to the nimble@cpp +//! binaries (`sidecars/nimble`), which are C++ and link it. //! //! Every lane streams: both sides are read batch by batch and compared window //! by window, so memory does not grow with the table. Any failure, a panic @@ -38,6 +42,10 @@ use std::sync::Arc; use anyhow::{bail, Context, Result}; use arrow_array::{Array, RecordBatch, RecordBatchReader}; +use arrow_avro::compression::CompressionCodec; +use arrow_avro::reader::ReaderBuilder as AvroReaderBuilder; +use arrow_avro::writer::format::AvroOcfFormat; +use arrow_avro::writer::WriterBuilder as AvroWriterBuilder; use arrow_ipc::reader::FileReader; use arrow_schema::{DataType, Field, Fields, Schema, SchemaRef}; use parquet::arrow::arrow_reader::{ @@ -58,6 +66,18 @@ use vortex::io::session::RuntimeSessionExt; use vortex::session::VortexSession; use vortex::VortexSessionDefault; +/// The write settings of the other formats (`spec.FORMAT_SETTINGS`). +const ORC_COMPRESSION: &str = "RAINCLOUD_ORC_COMPRESSION"; +const ORC_COMPRESSION_STRATEGY: &str = "RAINCLOUD_ORC_COMPRESSION_STRATEGY"; +const ORC_STRIPE_BYTES: &str = "RAINCLOUD_ORC_STRIPE_BYTES"; +const ORC_COMPRESSION_BLOCK_BYTES: &str = "RAINCLOUD_ORC_COMPRESSION_BLOCK_BYTES"; +const AVRO_COMPRESSION: &str = "RAINCLOUD_AVRO_COMPRESSION"; +const AVRO_COMPRESSION_LEVEL: &str = "RAINCLOUD_AVRO_COMPRESSION_LEVEL"; +const AVRO_BLOCK_BYTES: &str = "RAINCLOUD_AVRO_BLOCK_BYTES"; +const VORTEX_COMPACT: &str = "RAINCLOUD_VORTEX_COMPACT"; +const VORTEX_ROW_BLOCK_ROWS: &str = "RAINCLOUD_VORTEX_ROW_BLOCK_ROWS"; +const VORTEX_DATA_BLOCK_BYTES: &str = "RAINCLOUD_VORTEX_DATA_BLOCK_BYTES"; + /// raincloud's top-level VARIANT marker (see `discovery._is_variant_field`). const VARIANT_MARKER: &str = "__variant_type"; @@ -238,6 +258,276 @@ impl RowGroupLimits { } } +/// An on/off knob as every lane reads it (`spec._env_switch`): unset or empty +/// -> `None`; otherwise 1/true/yes/on or 0/false/no/off, in any case. +fn switch(var: &str, raw: Option<&OsStr>) -> Result> { + let Some(raw) = raw else { + return Ok(None); + }; + let Some(text) = raw.to_str() else { + bail!("{var}={raw:?} is not valid UTF-8; give 1 or 0"); + }; + match text + .trim_matches(is_knob_space) + .to_ascii_lowercase() + .as_str() + { + "" => Ok(None), + "1" | "true" | "yes" | "on" => Ok(Some(true)), + "0" | "false" | "no" | "off" => Ok(Some(false)), + _ => bail!("{var}='{text}' is not a switch; give 1 or 0 (true/false, yes/no, on/off)"), + } +} + +/// A setting that names one of `choices` (`spec._env_choice`): unset or empty +/// -> `None`; otherwise one of them, in any case. +fn choice(var: &str, choices: &[&'static str]) -> Result> { + let Some(raw) = std::env::var_os(var) else { + return Ok(None); + }; + let text = raw + .to_str() + .map(|s| s.trim_matches(is_knob_space).to_ascii_lowercase()); + match text.as_deref() { + Some("") => Ok(None), + Some(value) => match choices.iter().find(|c| **c == value) { + Some(found) => Ok(Some(found)), + None => bail!("{var}={raw:?} is not one of {}", choices.join(", ")), + }, + None => bail!("{var}={raw:?} is not valid UTF-8"), + } +} + +/// A size or count setting from the environment, as [`ParquetOptions`] reads +/// its own: unset -> `None` (the library's default); empty or 0 -> no limit. +fn setting_count(var: &str) -> Result> { + std::env::var_os(var) + .map(|raw| knob(var, Some(&raw), 0, (1 << 31) - 1)) + .transpose() +} + +/// Refuse a set setting this lane's library has no way to honour. +fn refuse_set( + lane: &str, + var: &str, + value: Option, + why: &str, +) -> Result<()> { + match value { + Some(value) => bail!("{lane} cannot honour {var}={value}: {why}"), + None => Ok(()), + } +} + +/// The Parquet write options every Parquet lane is given the same way +/// (`raincloud/pipeline/spec.py::ParquetOptions`, which documents each). The +/// sidecar never sees the recipe: `SidecarExporter` passes its +/// `write.compression` and `write.statistics` as `RAINCLOUD_PARQUET_COMPRESSION` +/// and `RAINCLOUD_PARQUET_STATISTICS`, and each install setting only when it is +/// set. Unset leaves arrow-rs's own default: a page index, 1 MiB pages, 20,000 +/// rows a page, dictionaries on, no page checksums. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ParquetOptions { + /// The codec, at `RAINCLOUD_PARQUET_COMPRESSION_LEVEL` when that is set. + pub compression: Compression, + pub statistics: bool, + /// Statistics only for the first N leaf columns. + pub statistics_columns: Option, + /// A ColumnIndex and OffsetIndex for every column chunk, or neither. + pub page_index: Option, + /// Page statistics only for the first N leaf columns; every column keeps + /// its chunk statistics. + pub page_index_columns: Option, + /// Data page size target, in bytes. + pub page_bytes: Option, + /// Data page row limit. + pub page_rows: Option, + pub dictionary: Option, + /// Dictionary page size limit, in bytes. + pub dictionary_page_bytes: Option, +} + +impl Default for ParquetOptions { + fn default() -> Self { + Self { + compression: Compression::ZSTD(ZstdLevel::default()), + statistics: true, + statistics_columns: None, + page_index: None, + page_index_columns: None, + page_bytes: None, + page_rows: None, + dictionary: None, + dictionary_page_bytes: None, + } + } +} + +impl ParquetOptions { + pub fn from_env() -> Result { + Self::from_vars(|name| std::env::var_os(name)) + } + + /// [`Self::from_env`] over `var`, so a test need not touch the process + /// environment. + pub fn from_vars(var: impl Fn(&str) -> Option) -> Result { + const COMPRESSION: &str = "RAINCLOUD_PARQUET_COMPRESSION"; + const LEVEL: &str = "RAINCLOUD_PARQUET_COMPRESSION_LEVEL"; + const STATISTICS: &str = "RAINCLOUD_PARQUET_STATISTICS"; + const STATISTICS_COLUMNS: &str = "RAINCLOUD_PARQUET_STATISTICS_COLUMNS"; + const PAGE_INDEX: &str = "RAINCLOUD_PARQUET_PAGE_INDEX"; + const PAGE_INDEX_COLUMNS: &str = "RAINCLOUD_PARQUET_PAGE_INDEX_COLUMNS"; + const PAGE_BYTES: &str = "RAINCLOUD_PARQUET_PAGE_BYTES"; + const PAGE_ROWS: &str = "RAINCLOUD_PARQUET_PAGE_ROWS"; + const DICTIONARY: &str = "RAINCLOUD_PARQUET_DICTIONARY"; + const DICTIONARY_PAGE_BYTES: &str = "RAINCLOUD_PARQUET_DICTIONARY_PAGE_BYTES"; + const PAGE_CHECKSUMS: &str = "RAINCLOUD_PARQUET_PAGE_CHECKSUMS"; + let flag = |name: &str| switch(name, var(name).as_deref()); + // Unset is arrow-rs's default; empty or 0 is no limit, as in parquet@py. + let count = |name: &str| -> Result> { + var(name) + .map(|raw| knob(name, Some(&raw), 0, (1 << 31) - 1)) + .transpose() + }; + let level = match var(LEVEL) { + None => None, + Some(raw) => match raw.to_str().map(|s| s.trim_matches(is_knob_space)) { + Some("") => None, + Some(text) if text.bytes().all(|b| b.is_ascii_digit()) => Some( + text.parse::() + .with_context(|| format!("{LEVEL}='{text}' is too large"))?, + ), + _ => bail!( + "{LEVEL}={raw:?} is not a compression level; give a whole number such as 3" + ), + }, + }; + let codec = match var(COMPRESSION) { + None => "zstd".to_string(), + Some(raw) => raw + .to_str() + .map(|s| s.trim_matches(is_knob_space).to_string()) + .unwrap_or_default(), + }; + let leveled = |name: &str, result: parquet::errors::Result| { + result.map_err(|e| anyhow::anyhow!("{LEVEL}={} for {name}: {e}", level.unwrap())) + }; + use parquet::basic::{BrotliLevel, GzipLevel}; + let compression = match (codec.as_str(), level) { + ("zstd", None) => Compression::ZSTD(ZstdLevel::default()), + ("zstd", Some(l)) => leveled( + "zstd", + i32::try_from(l) + .map_err(|_| parquet::errors::ParquetError::General("too large".into())) + .and_then(ZstdLevel::try_new) + .map(Compression::ZSTD), + )?, + ("gzip", None) => Compression::GZIP(Default::default()), + ("gzip", Some(l)) => leveled("gzip", GzipLevel::try_new(l).map(Compression::GZIP))?, + ("brotli", None) => Compression::BROTLI(Default::default()), + ("brotli", Some(l)) => { + leveled("brotli", BrotliLevel::try_new(l).map(Compression::BROTLI))? + } + ("snappy" | "lz4" | "none", Some(l)) => { + bail!("{LEVEL}={l}: {codec} takes no compression level") + } + ("snappy", None) => Compression::SNAPPY, + // pyarrow's "lz4" and Hardwood's LZ4_RAW: the framing the format + // recommends, not the deprecated Hadoop LZ4. + ("lz4", None) => Compression::LZ4_RAW, + ("none", None) => Compression::UNCOMPRESSED, + _ => { + bail!("{COMPRESSION}='{codec}' is not one of zstd, snappy, gzip, lz4, brotli, none") + } + }; + let statistics = flag(STATISTICS)?.unwrap_or(true); + let options = Self { + compression, + statistics, + statistics_columns: count(STATISTICS_COLUMNS)?, + page_index: flag(PAGE_INDEX)?, + page_index_columns: count(PAGE_INDEX_COLUMNS)?, + page_bytes: count(PAGE_BYTES)?, + page_rows: count(PAGE_ROWS)?, + dictionary: flag(DICTIONARY)?, + dictionary_page_bytes: count(DICTIONARY_PAGE_BYTES)?, + }; + if !statistics { + for (name, asked) in [ + (PAGE_INDEX, options.page_index == Some(true)), + (STATISTICS_COLUMNS, options.statistics_columns.is_some()), + (PAGE_INDEX_COLUMNS, options.page_index_columns.is_some()), + ] { + if asked { + bail!("{name} asks for statistics, but {STATISTICS} is 0"); + } + } + } + if options.page_index == Some(false) && options.page_index_columns.is_some() { + bail!("{PAGE_INDEX_COLUMNS} asks for a page index, but {PAGE_INDEX} is 0"); + } + if flag(PAGE_CHECKSUMS)? == Some(true) { + bail!("parquet@rs cannot honour {PAGE_CHECKSUMS}=1: arrow-rs writes no page checksums"); + } + Ok(options) + } + + /// `builder` with these options for a file of `columns`, except + /// compression, which the caller sets. + fn apply( + &self, + mut builder: parquet::file::properties::WriterPropertiesBuilder, + columns: &parquet::schema::types::SchemaDescriptor, + ) -> parquet::file::properties::WriterPropertiesBuilder { + use parquet::file::properties::EnabledStatistics; + if !self.statistics { + builder = builder.set_statistics_enabled(EnabledStatistics::None); + } + match self.page_index { + Some(true) => builder = builder.set_statistics_enabled(EnabledStatistics::Page), + // Neither index, as pyarrow writes without one: Chunk statistics alone + // still write an OffsetIndex. + Some(false) => { + if self.statistics { + builder = builder.set_statistics_enabled(EnabledStatistics::Chunk); + } + builder = builder.set_offset_index_disabled(true); + } + None => {} + } + if self.statistics { + for (i, column) in columns.columns().iter().enumerate() { + let path = column.path().clone(); + let level = if self.statistics_columns.is_some_and(|n| i >= n) { + EnabledStatistics::None + } else if let Some(n) = self.page_index_columns { + if i < n { + EnabledStatistics::Page + } else { + EnabledStatistics::Chunk + } + } else { + continue; + }; + builder = builder.set_column_statistics_enabled(path, level); + } + } + if let Some(bytes) = self.page_bytes { + builder = builder.set_data_page_size_limit(bytes); + } + if let Some(rows) = self.page_rows { + builder = builder.set_data_page_row_count_limit(rows); + } + if let Some(enabled) = self.dictionary { + builder = builder.set_dictionary_enabled(enabled); + } + if let Some(bytes) = self.dictionary_page_bytes { + builder = builder.set_dictionary_page_size_limit(bytes); + } + builder + } +} + /// Rows handed to arrow-rs per `write` call. /// /// arrow-rs applies `max_row_group_bytes` by measuring the rows it has already @@ -302,7 +592,22 @@ where Ok(()) } -/// Stream record batches into a Parquet file (zstd compression, arrow-rs). +/// Stream record batches into a Parquet file (zstd compression, arrow-rs), with +/// the default [`ParquetOptions`]; see [`write_parquet_with`]. +pub fn write_parquet( + output: &Path, + schema: SchemaRef, + limits: RowGroupLimits, + open: F, +) -> Result<()> +where + F: Fn() -> Result, + I: IntoIterator>, +{ + write_parquet_with(output, schema, limits, ParquetOptions::default(), open) +} + +/// Stream record batches into a Parquet file with `options` (arrow-rs). /// /// `open` is called twice, since the batches are read twice rather than held: /// SF100 lineitem is ~100 GB decoded. Each call must replay the same rows in @@ -318,11 +623,14 @@ where /// put TPC-H customer at 377 MiB encoded per group for a 128 MiB target. /// So the first pass encodes uncompressed into a sink, where arrow-rs's measure /// is the encoded size, and records where each group ends; the second writes -/// the file with zstd and closes groups at those rows. -pub fn write_parquet( +/// the file with `options.compression` and closes groups at those rows. Both +/// passes write the same pages and statistics, so the planned encoded size is +/// the file's. +pub fn write_parquet_with( output: &Path, schema: SchemaRef, limits: RowGroupLimits, + options: ParquetOptions, open: F, ) -> Result<()> where @@ -343,7 +651,11 @@ where // smaller limit", and its 1Mi-row default would cap TPC-H lineitem at // ~63 MiB encoded, exactly as parquet-java ships its own row limit // effectively off so that bytes decide. The backstop stays. - let plan_props = WriterProperties::builder() + let columns = parquet::arrow::ArrowSchemaConverter::new() + .convert(&schema) + .context("derive the Parquet schema")?; + let plan_props = options + .apply(WriterProperties::builder(), &columns) .set_compression(Compression::UNCOMPRESSED) .set_max_row_group_row_count(Some(limits.max_rows)) .set_max_row_group_bytes(Some(limits.target_encoded_bytes)) @@ -365,8 +677,9 @@ where // a group. let file = File::create(output).with_context(|| format!("create parquet {}", output.display()))?; - let props = WriterProperties::builder() - .set_compression(Compression::ZSTD(ZstdLevel::default())) + let props = options + .apply(WriterProperties::builder(), &columns) + .set_compression(options.compression) .set_max_row_group_row_count(None) .set_max_row_group_bytes(None) .build(); @@ -586,7 +899,9 @@ fn vortex_storage_schema(schema: &Schema) -> SchemaRef { } /// Stream the canonical into a `.vortex` file with Vortex's default write -/// strategy — chunking, statistics and compression — and the schema +/// strategy — chunking, statistics and compression — unless a Vortex write +/// setting asks otherwise (`RAINCLOUD_VORTEX_COMPACT`, `_ROW_BLOCK_ROWS`, +/// `_DATA_BLOCK_BYTES`), and the schema /// [`vortex_storage_schema`] gives, as `vortex@py` (`vortex.io.write`) does, /// so the artifact is the one a build would publish, not merely one that /// round-trips. @@ -609,8 +924,34 @@ pub fn write_vortex(output: &Path, canonical: &Path) -> Result<()> { let mut file = tokio::fs::File::create(output) .await .with_context(|| format!("create vortex {}", output.display()))?; - session - .write_options() + // The Vortex write settings (`spec.FORMAT_SETTINGS["vortex"]`). Unset is the + // default strategy; set, the strategy vortex@py builds, BtrBlocks limited to + // the encodings the session's editions allow, compact on request. + let compact = switch(VORTEX_COMPACT, std::env::var_os(VORTEX_COMPACT).as_deref())?; + let row_block_rows = setting_count(VORTEX_ROW_BLOCK_ROWS)?; + let data_block_bytes = setting_count(VORTEX_DATA_BLOCK_BYTES)?; + let mut options = session.write_options(); + if compact.is_some() || row_block_rows.is_some() || data_block_bytes.is_some() { + use vortex::editions::{ComponentKind, EditionSessionExt}; + let allowed = session + .enabled_component_ids(ComponentKind::Array) + .into_iter() + .collect(); + let mut compressor = vortex::compressor::BtrBlocksCompressorBuilder::default(); + if compact == Some(true) { + compressor = compressor.with_compact(); + } + let mut strategy = vortex::file::WriteStrategyBuilder::default() + .with_btrblocks_builder(compressor.retain_allowed_encodings(&allowed)); + if let Some(rows) = row_block_rows { + strategy = strategy.with_row_block_size(rows); + } + if let Some(bytes) = data_block_bytes { + strategy = strategy.with_data_block_target_bytes(Some(bytes as u64)); + } + options = options.with_strategy(strategy.build()); + } + options .write(&mut file, stream) .await .context("write vortex file")?; @@ -661,6 +1002,290 @@ pub fn open_vortex(input: &Path) -> Result<(SchemaRef, VortexBatches)> { // Logical comparison + variant detection // --------------------------------------------------------------------------- +// --------------------------------------------------------------------------- +// ORC (orc-rust) +// --------------------------------------------------------------------------- + +/// `data_type` as the ORC lane stores it: ORC has no unsigned integers or view +/// types, so the lane deliberately expands them, losslessly, before any writer +/// sees them: UInt8/16/32 to the next wider signed integer, UInt64 to +/// Decimal128(20, 0), and views to their plain equivalents, at any depth. +fn orc_storage_type(data_type: &DataType) -> DataType { + let field = |f: &Arc| { + Arc::new( + f.as_ref() + .clone() + .with_data_type(orc_storage_type(f.data_type())), + ) + }; + match data_type { + DataType::UInt8 => DataType::Int16, + DataType::UInt16 => DataType::Int32, + DataType::UInt32 => DataType::Int64, + DataType::UInt64 => DataType::Decimal128(20, 0), + DataType::Utf8View => DataType::Utf8, + DataType::BinaryView => DataType::Binary, + DataType::List(f) => DataType::List(field(f)), + DataType::LargeList(f) => DataType::LargeList(field(f)), + DataType::FixedSizeList(f, n) => DataType::FixedSizeList(field(f), *n), + DataType::Map(f, sorted) => DataType::Map(field(f), *sorted), + DataType::Struct(fields) => DataType::Struct(fields.iter().map(field).collect()), + other => other.clone(), + } +} + +/// Write `output` from the canonical with orc-rust's `ArrowWriter` and the ORC +/// write settings: unset, zstd, since the API makes the caller pick a codec (its +/// default is none), and its own default stripe and compression block sizes. Columns are first expanded to +/// [`orc_storage_type`]. orc-rust panics on a type it does not write +/// (`unimplemented!("unsupported datatype")`) -- 0.9.0 writes no decimal, so a +/// uint64 column fails here (measured); [`run_writer`] reports that panic as +/// the write's failure. +pub fn write_orc(output: &Path, canonical: &Path) -> Result<()> { + let (canonical_schema, reader) = open_canonical(canonical)?; + let schema = Arc::new(Schema::new_with_metadata( + canonical_schema + .fields() + .iter() + .map(|f| { + f.as_ref() + .clone() + .with_data_type(orc_storage_type(f.data_type())) + }) + .collect::>(), + canonical_schema.metadata().clone(), + )); + // The ORC write settings (`spec.FORMAT_SETTINGS["orc"]`); unset, zstd. + use orc_rust::compression::CompressionType; + refuse_set( + "orc@rs", + ORC_COMPRESSION_STRATEGY, + choice(ORC_COMPRESSION_STRATEGY, &["speed", "compression"])?, + "orc-rust has no compression strategy", + )?; + let mut builder = orc_rust::ArrowWriterBuilder::new( + File::create(output).with_context(|| format!("create {}", output.display()))?, + schema.clone(), + ); + let codec = choice(ORC_COMPRESSION, &["zstd", "snappy", "zlib", "lz4", "none"])?; + builder = match codec.unwrap_or("zstd") { + "zstd" => builder.with_compression(CompressionType::Zstd), + "snappy" => builder.with_compression(CompressionType::Snappy), + "zlib" => builder.with_compression(CompressionType::Zlib), + "lz4" => builder.with_compression(CompressionType::Lz4), + _ => builder, + }; + if let Some(bytes) = setting_count(ORC_STRIPE_BYTES)? { + builder = builder.with_stripe_byte_size(bytes); + } + if let Some(bytes) = setting_count(ORC_COMPRESSION_BLOCK_BYTES)? { + builder = builder.with_compression_block_size(bytes); + } + let mut writer = builder + .try_build() + .context("orc-rust: start the ORC file")?; + for batch in canonical_batches(reader) { + let batch = batch?; + let columns = batch + .columns() + .iter() + .zip(schema.fields()) + .map(|(column, field)| arrow_cast::cast(column, field.data_type())) + .collect::, _>>() + .context("expand the canonical to ORC's types")?; + writer + .write(&RecordBatch::try_new(schema.clone(), columns)?) + .context("orc-rust: write a record batch")?; + } + writer.close().context("orc-rust: finish the ORC file") +} + +/// Open an ORC file with orc-rust's `ArrowReader`, in the batches it yields. +pub fn open_orc(input: &Path) -> Result<(SchemaRef, impl Iterator>)> { + let file = File::open(input).with_context(|| format!("open {}", input.display()))?; + let reader = orc_rust::ArrowReaderBuilder::try_new(file) + .with_context(|| format!("orc-rust: read the ORC footer of {}", input.display()))? + .build(); + let schema = reader.schema(); + Ok(( + schema, + reader.map(|b| b.context("orc-rust: read a record batch")), + )) +} + +// --------------------------------------------------------------------------- +// Avro (arrow-avro) +// --------------------------------------------------------------------------- + +/// The sync marker of every Avro file raincloud writes. An object container +/// file separates its blocks with a 16-byte marker the writer chooses, and +/// arrow-avro draws it at random, so the same canonical would give a file with +/// a different sha256 on every build. The Java lane writes the same marker. +pub const AVRO_SYNC_MARKER: &[u8; 16] = b"raincloud-avro01"; + +/// Replace the random sync marker `drawn` with [`AVRO_SYNC_MARKER`] in the +/// object container file at `path`, in place: after the header and after each +/// block, each checked to be `drawn` first. Nothing else in the file changes. +/// +/// arrow-avro offers no way to choose the marker: `AvroOcfFormat` draws it, +/// and the `AvroFormat` trait a caller could implement instead must write the +/// header itself, whose schema JSON arrow-avro builds with a crate-private +/// option. Rewriting the marker keeps every other byte arrow-avro's. +fn fix_sync_marker(path: &Path, drawn: &[u8; 16]) -> Result<()> { + use std::io::{BufReader, Read, Seek, SeekFrom, Write}; + + struct Counted { + inner: R, + at: u64, + } + impl Counted { + fn bytes(&mut self, n: u64) -> Result> { + let mut buf = vec![0; n as usize]; + self.inner.read_exact(&mut buf)?; + self.at += n; + Ok(buf) + } + /// An Avro `long` (zig-zag varint), or None at a clean end of file. + fn long(&mut self) -> Result> { + let (mut n, mut shift) = (0u64, 0); + loop { + let mut byte = [0u8]; + if self.inner.read(&mut byte)? == 0 { + if shift == 0 { + return Ok(None); + } + bail!("truncated Avro long"); + } + self.at += 1; + n |= u64::from(byte[0] & 0x7f) << shift; + if byte[0] & 0x80 == 0 { + return Ok(Some((n >> 1) as i64 ^ -((n & 1) as i64))); + } + shift += 7; + } + } + fn need(&mut self) -> Result { + self.long()?.context("unexpected end of the Avro file") + } + fn marker(&mut self, drawn: &[u8; 16], markers: &mut Vec) -> Result<()> { + let at = self.at; + if self.bytes(16)? != drawn { + bail!("no sync marker at byte {at}"); + } + markers.push(at); + Ok(()) + } + } + + let mut file = Counted { + inner: BufReader::new(File::open(path)?), + at: 0, + }; + let mut markers = Vec::new(); + if file.bytes(4)? != b"Obj\x01" { + bail!("not an Avro object container file"); + } + // The header's metadata map: blocks of key/value pairs, ending with 0. + loop { + let mut count = file.need()?; + if count == 0 { + break; + } + if count < 0 { + count = -count; + file.need()?; // the block's byte size + } + for _ in 0..2 * count { + let len = file.need()?; + file.bytes(len as u64)?; + } + } + file.marker(drawn, &mut markers)?; + // Data blocks: a row count, a byte size, the bytes, the marker. + while file.long()?.is_some() { + let size = file.need()?; + file.bytes(size as u64)?; + file.marker(drawn, &mut markers)?; + } + let mut out = std::fs::OpenOptions::new().write(true).open(path)?; + for at in markers { + out.seek(SeekFrom::Start(at))?; + out.write_all(AVRO_SYNC_MARKER)?; + } + out.sync_all()?; + Ok(()) +} + +/// Write `output` from the canonical with arrow-avro's `AvroWriter`: the Avro +/// write settings' codec (unset, zstd), one block per canonical batch, then [`AVRO_SYNC_MARKER`] in place of its random +/// sync marker, for a reproducible file. A type arrow-avro does not write is +/// its error; nothing here converts a column. +pub fn write_avro(output: &Path, canonical: &Path) -> Result<()> { + let (schema, reader) = open_canonical(canonical)?; + // The Avro write settings (`spec.FORMAT_SETTINGS["avro"]`); unset, zstd. + let level = std::env::var_os(AVRO_COMPRESSION_LEVEL) + .and_then(|raw| { + raw.to_str() + .map(|s| s.trim_matches(is_knob_space).to_string()) + }) + .filter(|s| !s.is_empty()); + refuse_set( + "avro@rs", + AVRO_COMPRESSION_LEVEL, + level, + "arrow-avro has no compression level", + )?; + refuse_set( + "avro@rs", + AVRO_BLOCK_BYTES, + setting_count(AVRO_BLOCK_BYTES)?, + "arrow-avro writes one block per batch, with no block size", + )?; + let codec = match choice( + AVRO_COMPRESSION, + &["zstd", "deflate", "snappy", "bzip2", "xz", "none"], + )? + .unwrap_or("zstd") + { + "zstd" => Some(CompressionCodec::ZStandard), + "deflate" => Some(CompressionCodec::Deflate), + "snappy" => Some(CompressionCodec::Snappy), + "bzip2" => Some(CompressionCodec::Bzip2), + "xz" => Some(CompressionCodec::Xz), + _ => None, + }; + let file = File::create(output).with_context(|| format!("create {}", output.display()))?; + let mut writer = AvroWriterBuilder::new(schema.as_ref().clone()) + .with_compression(codec) + .build::<_, AvroOcfFormat>(std::io::BufWriter::new(file)) + .context("arrow-avro: start the Avro file")?; + let drawn = *writer.sync_marker().context("arrow-avro: no sync marker")?; + for batch in canonical_batches(reader) { + writer + .write(&batch?) + .context("arrow-avro: write a record batch")?; + } + writer + .finish() + .context("arrow-avro: finish the Avro file")?; + drop(writer); + fix_sync_marker(output, &drawn).context("set the Avro sync marker") +} + +/// Open an Avro object container file with arrow-avro's `Reader`, in the +/// batches it yields. +pub fn open_avro(input: &Path) -> Result<(SchemaRef, impl Iterator>)> { + let file = File::open(input).with_context(|| format!("open {}", input.display()))?; + let reader = AvroReaderBuilder::new() + .build(std::io::BufReader::new(file)) + .with_context(|| format!("arrow-avro: read the Avro header of {}", input.display()))?; + let schema = reader.schema(); + Ok(( + schema, + reader.map(|b| b.context("arrow-avro: read a record batch")), + )) +} + /// True if any top-level field carries raincloud's VARIANT marker. pub fn has_variant(schema: &Schema) -> bool { schema @@ -733,7 +1358,9 @@ pub fn parquet_variant_loss( // Check schema before values: reversible casts of empty/null arrays can erase // incompatible types and struct children. Representation widths, dictionary -// indices, list element names and metadata are intentionally not identities. +// indices, list element names and metadata are intentionally not identities; +// nor are a null|T union's spelling of a nullable T, a timestamp's zone label, +// or a scale-0 decimal's spelling of an integer (sidecars/compare_cases). fn compatible_fields(got: &Fields, expected: &Fields) -> bool { got.len() == expected.len() && got @@ -742,8 +1369,118 @@ fn compatible_fields(got: &Fields, expected: &Fields) -> bool { .all(|(g, e)| g.name() == e.name() && compatible(g.data_type(), e.data_type())) } +/// The type id and field of `T` in a union of exactly `null` and `T`, else +/// None: such a union is how some readers spell a nullable `T` (Avro's +/// `["null", T]`). +fn nullable_member(data_type: &DataType) -> Option<(i8, &Arc)> { + let DataType::Union(fields, _) = data_type else { + return None; + }; + let mut members = fields + .iter() + .filter(|(_, f)| f.data_type() != &DataType::Null); + match (fields.len(), members.next(), members.next()) { + (2, Some(member), None) => Some(member), + _ => None, + } +} + +fn has_nullable_union(data_type: &DataType) -> bool { + use DataType::*; + nullable_member(data_type).is_some() + || match data_type { + Struct(fields) => fields.iter().any(|f| has_nullable_union(f.data_type())), + List(f) + | LargeList(f) + | ListView(f) + | LargeListView(f) + | FixedSizeList(f, _) + | Map(f, _) => has_nullable_union(f.data_type()), + _ => false, + } +} + +/// `array` with every union of `null` and `T` replaced by the nullable `T` it +/// spells, at any depth of struct, list and map: a row selecting the null +/// member is null, any other is the `T` member's value (a sparse member at the +/// row, a dense one at its offset). +fn without_nullable_unions( + array: &arrow_array::ArrayRef, +) -> Result { + use DataType::*; + if !has_nullable_union(array.data_type()) { + return Ok(Arc::clone(array)); + } + if let Some((code, _)) = nullable_member(array.data_type()) { + let union = array + .as_any() + .downcast_ref::() + .expect("a union type is a UnionArray"); + let picks: arrow_array::UInt32Array = (0..union.len()) + .map(|i| (union.type_id(i) == code).then(|| union.value_offset(i) as u32)) + .collect(); + let picked = arrow_select::take::take(union.child(code).as_ref(), &picks, None)?; + return without_nullable_unions(&picked); + } + let data = array.to_data(); + let children = data + .child_data() + .iter() + .map(|c| without_nullable_unions(&arrow_array::make_array(c.clone())).map(|a| a.to_data())) + .collect::, _>>()?; + // A field that held the union now holds nulls for its null member. + let field = |f: &Arc, child: &DataType| { + let nullable = f.is_nullable() || nullable_member(f.data_type()).is_some(); + Arc::new( + f.as_ref() + .clone() + .with_data_type(child.clone()) + .with_nullable(nullable), + ) + }; + let child = |i: usize| children[i].data_type(); + let dtype = match array.data_type() { + Struct(fields) => Struct( + fields + .iter() + .enumerate() + .map(|(i, f)| field(f, child(i))) + .collect(), + ), + List(f) => List(field(f, child(0))), + LargeList(f) => LargeList(field(f, child(0))), + ListView(f) => ListView(field(f, child(0))), + LargeListView(f) => LargeListView(field(f, child(0))), + FixedSizeList(f, n) => FixedSizeList(field(f, child(0)), *n), + Map(f, sorted) => Map(field(f, child(0)), *sorted), + _ => unreachable!("has_nullable_union covers these containers only"), + }; + Ok(arrow_array::make_array( + data.into_builder() + .data_type(dtype) + .child_data(children) + .build()?, + )) +} + +/// A decimal holding integers: scale 0. +fn is_integral_decimal(data_type: &DataType) -> bool { + use DataType::*; + matches!( + data_type, + Decimal32(_, 0) | Decimal64(_, 0) | Decimal128(_, 0) | Decimal256(_, 0) + ) +} + fn compatible(got: &DataType, expected: &DataType) -> bool { use DataType::*; + // A null|T union is a nullable T, on either side. + if let Some((_, member)) = nullable_member(got) { + return compatible(member.data_type(), expected); + } + if let Some((_, member)) = nullable_member(expected) { + return compatible(got, member.data_type()); + } match (got, expected) { (Dictionary(_, value), other) | (other, Dictionary(_, value)) => compatible(value, other), (Struct(g), Struct(e)) => compatible_fields(g, e), @@ -761,7 +1498,13 @@ fn compatible(got: &DataType, expected: &DataType) -> bool { }) } (RunEndEncoded(_, g), RunEndEncoded(_, e)) => compatible(g.data_type(), e.data_type()), - (Timestamp(_, g), Timestamp(_, e)) => g == e, + // A zone labels UTC instants; it is not data. Naive and zoned differ in kind. + (Timestamp(_, g), Timestamp(_, e)) => g.is_some() == e.is_some(), + _ if (got.is_integer() && is_integral_decimal(expected)) + || (is_integral_decimal(got) && expected.is_integer()) => + { + true + } (FixedSizeBinary(g), FixedSizeBinary(e)) if g != e => false, (Utf8 | LargeUtf8 | Utf8View, Utf8 | LargeUtf8 | Utf8View) | ( @@ -875,8 +1618,9 @@ fn normalize( /// /// Returns `(matches, detail)` — `detail` is a human-readable mismatch reason /// when `!matches`, else empty. Row count, names and recursive logical schema -/// compatibility first, then each column is cast losslessly to the canonical type -/// (identity when types already match) and compared at the `ArrayData` level. `ArrayData` equality ignores +/// compatibility first, then each column, with null|T unions read as nullable T, +/// is cast losslessly to the canonical type (identity when types already match) +/// and compared at the `ArrayData` level. `ArrayData` equality ignores /// field metadata, so a dropped VARIANT annotation does not fail the round-trip. fn logical_eq(got: &RecordBatch, expected: &RecordBatch) -> (bool, String) { if got.num_rows() != expected.num_rows() { @@ -942,6 +1686,19 @@ fn logical_eq(got: &RecordBatch, expected: &RecordBatch) -> (bool, String) { ), ); } + let (g, e) = match (without_nullable_unions(g), without_nullable_unions(e)) { + (Ok(g), Ok(e)) => (g, e), + (Err(err), _) | (_, Err(err)) => { + return ( + false, + format!( + "column {:?}: cannot read a null|T union as T: {err}", + exp_names[i] + ), + ); + } + }; + let (g, e) = (&g, &e); let g_cast = if g.data_type() == e.data_type() { Arc::clone(g) } else { @@ -1410,6 +2167,33 @@ mod tests { assert!(!compatible(&fields("a"), &fields("b"))); } + #[test] + fn comparisons_follow_the_shared_cases() { + // The same pairs pytest and JUnit read; Rust takes no `gap`. + let table: serde_json::Value = + serde_json::from_str(include_str!("../../compare_cases/cases.json")).unwrap(); + let dir = Path::new(env!("CARGO_MANIFEST_DIR")).join("../compare_cases"); + for case in table["cases"].as_array().unwrap() { + let name = case["name"].as_str().unwrap(); + let (got_schema, got) = open_canonical(&dir.join(format!("{name}.got.arrow"))).unwrap(); + let (expected_schema, expected) = + open_canonical(&dir.join(format!("{name}.expected.arrow"))).unwrap(); + let (equal, detail) = logical_eq_stream( + &got_schema, + canonical_batches(got), + &expected_schema, + canonical_batches(expected), + ) + .unwrap(); + let verdict = if equal { "equal" } else { "differ" }; + assert_eq!( + verdict, + case["verdict"].as_str().unwrap(), + "{name}: {detail}" + ); + } + } + #[test] fn normalize_bridges_fixed_size_forms() { let opts = arrow_cast::CastOptions { @@ -1600,6 +2384,146 @@ mod tests { assert!(decoded_bytes(group) >= 100_000, "{}", decoded_bytes(group)); } + fn options_from(pairs: &[(&str, &str)]) -> Result { + let vars: std::collections::HashMap = pairs + .iter() + .map(|(k, v)| (format!("RAINCLOUD_PARQUET_{k}"), (*v).into())) + .collect(); + ParquetOptions::from_vars(|name| vars.get(name).cloned()) + } + + #[test] + fn parquet_options_read_as_the_python_lane_writes_them() { + assert_eq!(options_from(&[]).unwrap(), ParquetOptions::default()); + let set = options_from(&[ + ("COMPRESSION", "lz4"), + ("STATISTICS", "1"), + ("PAGE_INDEX", " On "), + ("PAGE_BYTES", "4096"), + ("PAGE_ROWS", "0"), + ]) + .unwrap(); + assert_eq!(set.compression, Compression::LZ4_RAW); + assert_eq!(set.page_index, Some(true)); + assert_eq!(set.page_bytes, Some(4096)); + // 0 is no limit, as in parquet@py. + assert_eq!(set.page_rows, Some((1 << 31) - 1)); + assert_eq!( + options_from(&[("PAGE_INDEX", "")]).unwrap().page_index, + None + ); + let leveled = options_from(&[ + ("COMPRESSION_LEVEL", "9"), + ("STATISTICS_COLUMNS", "100"), + ("PAGE_INDEX_COLUMNS", "10"), + ("DICTIONARY", "off"), + ("DICTIONARY_PAGE_BYTES", "65536"), + ("PAGE_CHECKSUMS", "0"), + ]) + .unwrap(); + assert_eq!( + leveled.compression, + Compression::ZSTD(ZstdLevel::try_new(9).unwrap()) + ); + assert_eq!( + (leveled.statistics_columns, leveled.page_index_columns), + (Some(100), Some(10)) + ); + assert_eq!( + (leveled.dictionary, leveled.dictionary_page_bytes), + (Some(false), Some(65_536)) + ); + let gzip = options_from(&[("COMPRESSION", "gzip"), ("COMPRESSION_LEVEL", "0")]).unwrap(); + assert_eq!( + gzip.compression, + Compression::GZIP(parquet::basic::GzipLevel::try_new(0).unwrap()) + ); + for (pairs, error) in [ + (&[("PAGE_INDEX", "maybe")][..], "is not a switch"), + (&[("COMPRESSION", "lzo")][..], "is not one of"), + (&[("PAGE_BYTES", "1MiB")][..], "is not a number"), + ( + &[("PAGE_INDEX", "1"), ("STATISTICS", "0")][..], + "asks for statistics", + ), + ( + &[("STATISTICS_COLUMNS", "5"), ("STATISTICS", "0")][..], + "asks for statistics", + ), + ( + &[("PAGE_INDEX", "0"), ("PAGE_INDEX_COLUMNS", "5")][..], + "asks for a page index", + ), + ( + &[("COMPRESSION_LEVEL", "-1")][..], + "is not a compression level", + ), + (&[("COMPRESSION_LEVEL", "23")][..], "for zstd"), + ( + &[("COMPRESSION", "snappy"), ("COMPRESSION_LEVEL", "1")][..], + "takes no compression level", + ), + ( + &[("PAGE_CHECKSUMS", "1")][..], + "parquet@rs cannot honour RAINCLOUD_PARQUET_PAGE_CHECKSUMS=1", + ), + ] { + let message = options_from(pairs).unwrap_err().to_string(); + assert!(message.contains(error), "{pairs:?}: {message}"); + } + } + + #[test] + fn parquet_options_decide_the_page_index_and_the_pages() { + let schema = Arc::new(Schema::new(vec![Field::new("x", DataType::Int64, false)])); + let b = RecordBatch::try_new( + schema.clone(), + vec![Arc::new(arrow_array::Int64Array::from_iter_values( + 0..50_000, + ))], + ) + .unwrap(); + let limits = RowGroupLimits { + target_encoded_bytes: 128 << 20, + max_rows: 10_000_000, + }; + let scratch = Scratch::new("parquet-options"); + let written = |name: &str, options: ParquetOptions| { + let output = scratch.path(name); + write_parquet_with(&output, schema.clone(), limits, options, || { + Ok(stream(vec![b.clone()])) + }) + .unwrap(); + let file = File::open(&output).unwrap(); + let options = parquet::arrow::arrow_reader::ArrowReaderOptions::new() + .with_page_index_policy(parquet::file::metadata::PageIndexPolicy::Optional); + let reader = + ParquetRecordBatchReaderBuilder::try_new_with_options(file, options).unwrap(); + let metadata = reader.metadata().clone(); + let pages = metadata + .offset_index() + .map(|index| index[0][0].page_locations().len()); + (metadata.column_index().is_some(), pages) + }; + // arrow-rs's default: a page index, 20,000 rows a page. + assert_eq!( + written("default.parquet", ParquetOptions::default()), + (true, Some(3)) + ); + let rows = ParquetOptions { + page_rows: Some(1_000), + ..Default::default() + }; + // arrow-rs checks the row limit once per 1,024-value write batch, so a + // 1,000-row limit gives 1,024-row pages. + assert_eq!(written("rows.parquet", rows), (true, Some(49))); + let without = ParquetOptions { + page_index: Some(false), + ..Default::default() + }; + assert_eq!(written("without.parquet", without), (false, None)); + } + #[test] fn seconds_times_and_timestamps_are_written_as_milliseconds() { use arrow_array::{Time32SecondArray, TimestampSecondArray}; @@ -1718,6 +2642,149 @@ mod tests { vortex_round_trip(&scratch, &schema, &[RecordBatch::new_empty(schema.clone())]); } + #[test] + fn orc_writes_and_reads_back_across_batches() { + let schema = mixed_schema(); + let batch = |xs: Vec| { + let s: Vec> = xs + .iter() + .map(|x| (x % 3 != 0).then(|| format!("v{x}"))) + .collect(); + RecordBatch::try_new( + schema.clone(), + vec![ + Arc::new(Int64Array::from(xs)), + Arc::new(StringArray::from(s)), + ], + ) + .unwrap() + }; + let scratch = Scratch::new("orc"); + let source = scratch.canonical( + "source.arrow", + &schema, + &[batch((0..1000).collect()), batch((1000..5000).collect())], + ); + let output = scratch.path("out.orc"); + write_orc(&output, &source).unwrap(); + let (got_schema, got) = open_orc(&output).unwrap(); + let (_, expected) = open_canonical(&source).unwrap(); + assert_eq!( + logical_eq_stream(&got_schema, got, &schema, canonical_batches(expected)).unwrap(), + (true, String::new()) + ); + } + + #[test] + fn orc_expands_unsigned_columns_to_wider_signed_ones() { + let schema = Arc::new(Schema::new(vec![ + Field::new("a", DataType::UInt8, true), + Field::new("b", DataType::UInt32, true), + ])); + let b = RecordBatch::try_new( + schema.clone(), + vec![ + Arc::new(arrow_array::UInt8Array::from(vec![ + Some(0), + None, + Some(255), + ])), + Arc::new(arrow_array::UInt32Array::from(vec![ + Some(0), + Some(u32::MAX), + None, + ])), + ], + ) + .unwrap(); + let scratch = Scratch::new("orc-unsigned"); + let source = scratch.canonical("source.arrow", &schema, &[b]); + let output = scratch.path("out.orc"); + write_orc(&output, &source).unwrap(); + let (got_schema, got) = open_orc(&output).unwrap(); + let types: Vec<_> = got_schema.fields().iter().map(|f| f.data_type()).collect(); + assert_eq!(types, [&DataType::Int16, &DataType::Int64]); + let (_, expected) = open_canonical(&source).unwrap(); + assert_eq!( + logical_eq_stream(&got_schema, got, &schema, canonical_batches(expected)).unwrap(), + (true, String::new()) + ); + } + + #[test] + fn orc_rust_panicking_on_a_uint64_column_is_an_error() { + // Expanded to Decimal128(20, 0), which orc-rust 0.9.0 does not write. + let schema = Arc::new(Schema::new(vec![Field::new("u", DataType::UInt64, true)])); + let b = RecordBatch::try_new( + schema.clone(), + vec![Arc::new(arrow_array::UInt64Array::from(vec![1, u64::MAX]))], + ) + .unwrap(); + let scratch = Scratch::new("orc-uint64"); + let source = scratch.canonical("source.arrow", &schema, &[b]); + let output = scratch.path("out.orc"); + let err = unwound(|| write_orc(&output, &source)).unwrap_err(); + assert!( + format!("{err:#}").contains("unsupported datatype"), + "{err:#}" + ); + } + + #[test] + fn avro_changes_only_the_sync_marker() { + let schema = mixed_schema(); + let batch = |xs: Vec| { + let s: Vec> = xs + .iter() + .map(|x| (x % 3 != 0).then(|| format!("v{x}"))) + .collect(); + RecordBatch::try_new( + schema.clone(), + vec![ + Arc::new(Int64Array::from(xs)), + Arc::new(StringArray::from(s)), + ], + ) + .unwrap() + }; + let scratch = Scratch::new("avro"); + let batches = [batch((0..1000).collect()), batch((1000..3000).collect())]; + let source = scratch.canonical("source.arrow", &schema, &batches); + let output = scratch.path("out.avro"); + write_avro(&output, &source).unwrap(); + let ours = std::fs::read(&output).unwrap(); + + // What arrow-avro writes on its own, with its marker swapped for ours. + let mut w = AvroWriterBuilder::new(schema.as_ref().clone()) + .with_compression(Some(CompressionCodec::ZStandard)) + .build::<_, AvroOcfFormat>(Vec::new()) + .unwrap(); + let drawn = *w.sync_marker().unwrap(); + for b in &batches { + w.write(b).unwrap(); + } + w.finish().unwrap(); + let mut theirs = w.into_inner(); + let mut at = 0; + while let Some(i) = theirs[at..].windows(16).position(|w| w == drawn) { + theirs[at + i..at + i + 16].copy_from_slice(AVRO_SYNC_MARKER); + at += i + 16; + } + assert_eq!(ours, theirs); + // After the header and after each of the two blocks. + assert_eq!( + ours.windows(16).filter(|w| w == AVRO_SYNC_MARKER).count(), + 3 + ); + let (got_schema, got) = open_avro(&output).unwrap(); + let (_, expected) = open_canonical(&source).unwrap(); + assert!( + logical_eq_stream(&got_schema, got, &schema, canonical_batches(expected)) + .unwrap() + .0 + ); + } + #[test] fn vortex_is_handed_a_variant_column_as_its_storage_struct() { let storage = DataType::Struct( diff --git a/sources.json b/sources.json index 6cd7ff0..80a076e 100644 --- a/sources.json +++ b/sources.json @@ -1561,12 +1561,6 @@ "schema_hash": null, "notes": "100M upstream lines; the dumps wrap 16 records longer than 65,535 bytes onto two lines (files 5-7, 68, 82, 88, 94, 96, 97), which the handler rejoins: 99,999,984 records.", "row_stability": "static" - }, - "export": { - "formats": [ - "parquet", - "vortex" - ] } }, { @@ -10374,12 +10368,6 @@ "notes": "Wikimedia refreshes the Kaggle dataset in place; the catalog records the count each build produced.", "row_stability": "mutable" }, - "export": { - "formats": [ - "parquet", - "vortex" - ] - }, "tags": [ "identifiers", "nested-json", @@ -10495,12 +10483,6 @@ "notes": null, "row_stability": "static" }, - "export": { - "formats": [ - "parquet", - "vortex" - ] - }, "tags": [ "code-strings" ] @@ -13986,12 +13968,6 @@ "notes": null, "row_stability": "static" }, - "export": { - "formats": [ - "parquet", - "vortex" - ] - }, "references": [ { "kind": "paper", diff --git a/sources.schema.json b/sources.schema.json index 2951b1d..8f96921 100644 --- a/sources.schema.json +++ b/sources.schema.json @@ -589,7 +589,7 @@ "description": "schema_version 1 only (rejected in v2): never read by any writer." } }, - "description": "Parquet writer settings. parquet@py honours all three of compression, row_group_size_rows and statistics. The sidecar writers (parquet@rs, parquet@java, parquet@hardwood) receive row_group_size_rows as RAINCLOUD_ROW_GROUP_MAX_ROWS in their environment, so the recipe's row cap wins in every lane; they choose their own compression and statistics." + "description": "Parquet writer settings, honoured by every Parquet writer. parquet@py reads them here; the sidecar writers (parquet@rs, parquet@java, parquet@hardwood) never see the recipe, so the build passes them in their environment: row_group_size_rows as RAINCLOUD_ROW_GROUP_MAX_ROWS (the recipe's cap wins in every lane), compression and statistics as RAINCLOUD_PARQUET_COMPRESSION and RAINCLOUD_PARQUET_STATISTICS. The compression level, page index, page layout, dictionaries and checksums are the install's RAINCLOUD_PARQUET_* settings, not the recipe's. A writer whose library cannot do what is asked (parquet@java has no lz4 or brotli; Hardwood cannot turn statistics off) records Parquet unavailable rather than writing something else." }, "Expect": { "type": "object", @@ -702,7 +702,7 @@ }, "Export": { "type": "object", - "description": "Per-slug export policy (v2): which formats to materialise from the canonical Arrow spine beyond the always-produced `arrow/.arrow.zstd`, and which writer makes them. Each format is one file, `/.`, whichever writer made it; the catalog records the writer beside the file's sha256. In a v2 manifest this is the only declaration of a dataset's formats; omitted, it exports parquet and vortex.", + "description": "Per-slug export policy (v2): which formats to materialise from the canonical Arrow spine beyond the always-produced `arrow/.arrow.zstd`, and which writer makes them. Each format is one file, `/.`, whichever writer made it; the catalog records the writer beside the file's sha256. A dataset offers every exported format unless `formats` narrows the list; which of them a build writes is the install's `formats` setting (only vortex by default).", "additionalProperties": false, "properties": { "formats": { @@ -710,11 +710,14 @@ "items": { "enum": [ "parquet", - "vortex" + "vortex", + "orc", + "avro", + "nimble" ] }, "uniqueItems": true, - "description": "Formats the dataset wants. Replaces the default [\"parquet\", \"vortex\"]; [] produces only canonical Arrow. Never `arrow` (the canonical every exporter reads) and never a writer-qualified name — the writer comes from `priority`. A format the planned writer cannot produce for this dataset is not declared here: the build measures the failure and records it, and the catalog shows the format as unavailable with that measurement." + "description": "Formats the dataset offers, narrowing the default of every exported format; [] offers only canonical Arrow. Which offered formats a build writes is the install's choice (its `formats` setting). Never `arrow` (the canonical every exporter reads) and never a writer-qualified name — the writer comes from `priority`. A format the planned writer cannot produce for this dataset is not declared here: the build measures the failure and records it, and the catalog shows the format as unavailable with that measurement." }, "priority": { "$ref": "#/$defs/WriterPriority", @@ -786,7 +789,10 @@ "propertyNames": { "enum": [ "parquet", - "vortex" + "vortex", + "orc", + "avro", + "nimble" ] }, "additionalProperties": { diff --git a/sources.schema.md b/sources.schema.md index f2f91cc..92148a4 100644 --- a/sources.schema.md +++ b/sources.schema.md @@ -78,13 +78,15 @@ An optional top-level `export_priority` sets a catalog-wide writer order, in the }, /* Parquet writer settings, read when a writer derives .parquet from the - canonical Arrow spine (raincloud/pipeline/canonical.py). parquet@py - (raincloud/pipeline/export/) honours all three fields. The sidecar writers - (parquet@rs, parquet@java, parquet@hardwood) receive row_group_size_rows as - RAINCLOUD_ROW_GROUP_MAX_ROWS in their environment, so the recipe's cap wins - in every lane; they choose their own compression and statistics. v1 - manifests also carried `output` and `page_index`; no writer read either, - and v2 rejects them. */ + canonical Arrow spine (raincloud/pipeline/canonical.py). Every Parquet + writer honours all three fields: parquet@py reads them, and the sidecar + writers (parquet@rs, parquet@java, parquet@hardwood) receive them in their + environment (RAINCLOUD_ROW_GROUP_MAX_ROWS, RAINCLOUD_PARQUET_COMPRESSION, + RAINCLOUD_PARQUET_STATISTICS), so the recipe wins in every lane. A writer + whose library cannot do what is asked records Parquet unavailable. The + compression level, page index, page layout, dictionaries and checksums are + the install's RAINCLOUD_PARQUET_* settings, not the recipe's. v1 manifests also carried `output` and `page_index`; no writer + read either, and v2 rejects them. */ "write": { "compression": "zstd", "row_group_size_rows": 10000000, // a backstop row cap, not a target: groups are sized by encoded @@ -149,11 +151,11 @@ The v1 Vortex opt-in: `{"vortex": true, "vortex_skip_reason": null}` emits a sib Says which formats to materialise from the canonical Arrow spine (`outputs/v{n}//arrow/.arrow.zstd`), which is always produced under v2 and is *not* itself an export target, and which writer makes them. Each format is **one file**, `/.`, whichever writer made it; the catalog records the writer (`parquet_writer`, `vortex_writer`) beside the file's sha256. The writer is provenance, not part of the file's address. -- `formats` *(array of `"parquet"` / `"vortex"`, optional)* — in v2 the **only** declaration of the formats a dataset *wants*; omitted, it is `["parquet", "vortex"]`. `[]` produces only canonical Arrow. A format the planned writer cannot produce for the dataset (it raises, dies, reports a failed round-trip, or exceeds `RAINCLOUD_EXPORT_TIMEOUT`) stays listed: the build records the failure (writer cell, error, toolchain versions, recipe, time) in the install's build record and carries on with the formats that worked, and regenerating the catalog carries that measurement into the snapshot as `_unavailable`, where `raincloud describe` and the loader report it. A later build or export does not repeat it: while the measurement names the same writer cell, toolchain versions and canonical as the writer that would run, the format is skipped (`[skip]`) unless `--retry-errors` is passed. A later successful export replaces it, and `compliance` announces a `[stale opt-out]` when a writer round-trips a format the catalog records as unavailable. The schema and `validate_manifest` reject a `convert` block in a v2 manifest being authored. v2 catalogs released before that rule may still carry `convert.vortex: false`, and they keep reading: such a spec with no `export.formats` exports Parquet only. +- `formats` *(array of exported format names, optional)* — the formats a dataset *offers*; omitted, it offers every exported format. `[]` offers only canonical Arrow. Which offered formats a build actually writes is the install's choice: its `formats` setting (only Vortex by default; `all` for every one), or `raincloud build --format`. A format the planned writer cannot produce for the dataset (it raises, dies, reports a failed round-trip, or exceeds `RAINCLOUD_EXPORT_TIMEOUT`) stays listed: the build records the failure (writer cell, error, toolchain versions, recipe, time) in the install's build record and carries on with the formats that worked, and regenerating the catalog carries that measurement into the snapshot as `_unavailable`, where `raincloud describe` and the loader report it. A later build or export does not repeat it: while the measurement names the same writer cell, toolchain versions and canonical as the writer that would run, the format is skipped (`[skip]`) unless `--retry-errors` is passed. A later successful export replaces it, and `compliance` announces a `[stale opt-out]` when a writer round-trips a format the catalog records as unavailable. The schema and `validate_manifest` reject a `convert` block in a v2 manifest being authored. v2 catalogs released before that rule may still carry `convert.vortex: false`, and they keep reading: such a spec with no `export.formats` exports Parquet only. - `priority` *(optional)* — this dataset's writer preference, either a list of writer names that serves every format (`["rs", "py"]`) or a map from format to such a list (`{"parquet": ["rs", "py"]}`) that serves only the formats it names. For each format, the first writer in the list that exists for it and is installed writes it, so a machine without a preferred sidecar falls through to the next writer. A list does not fall through to the next level, so it must name a writer for every format the dataset exports: `["hardwood"]` (a Parquet-only writer) on a dataset that also exports Vortex leaves Vortex with no writer, and its export fails. `validate_manifest` rejects a list that leaves an exported format with no writer; use a map to prefer a writer for one format only. - `notes` *(string | null, optional)* — free-form annotation, such as why a deliberate policy leaves a format out. Never a writer's technical limitation: that is measured, never hand-written, so it cannot go stale. `list_datasets --no-vortex` and the TUI show it for a spec whose `formats` leaves out `vortex`. -**Writer precedence, per format** (stated here once): the spec's `export.priority`, then the catalog's top-level `export_priority` (same list-or-map shape), then the machine's `RAINCLOUD_EXPORT_PRIORITY` (a comma-separated list), then the built-in `py, rs, java, canonical`. A level that is a map without the format falls through to the next. `canonical` is the Arrow spine's writer; it has no Parquet or Vortex cell, so for exported formats the built-in order is effectively `py, rs, java`. +**Writer precedence, per format** (stated here once): the spec's `export.priority`, then the catalog's top-level `export_priority` (same list-or-map shape), then the machine's `RAINCLOUD_EXPORT_PRIORITY` (a comma-separated list), then the built-in `py, rs, java, cpp, canonical`. A level that is a map without the format falls through to the next. `canonical` is the Arrow spine's writer; it has no cell for an exported format, so for those the built-in order is effectively `py, rs, java, cpp` (`cpp` writes only Nimble). Unknown writer names are treated differently by audience: `validate_manifest` rejects a manifest name (spec or catalog) that is no export writer for its format, because in a manifest a typo would silently fall through to the next writer; at run time an unknown name is skipped, so a machine's `RAINCLOUD_EXPORT_PRIORITY` may name a writer this release does not ship. @@ -192,7 +194,7 @@ The full list is `HANDLERS` in `raincloud/_registry.py` (and `docs/v2/handlers.m 4. **transform** dispatches to `handler` with the parsed tables as input. Output is `(output_slug, arrow_table)` tuples. Streaming handlers write the canonical Arrow spine directly (via `open_canonical_writer`) and return `[]`. 5. **write_canonical** ([`raincloud/pipeline/canonical.py`](raincloud/pipeline/canonical.py)) writes the canonical Arrow IPC spine — zstd-compressed — to `outputs/v{schema_version}//arrow/.arrow.zstd`. This is the source of truth every format derives from; the former `write.py` parquet stage was removed. (For a streaming handler the spine was already written in transform.) 6. **validate** reads the canonical spine and compares row count + schema hash to `expect.*`. Warnings by default; `--strict` promotes mismatches to errors. -7. **export** ([`raincloud/pipeline/export/`](raincloud/pipeline/export/)) re-encodes canonical Arrow through the configured exporter cells: the formats in `export.formats` (default Parquet and Vortex), each written once, to `parquet/` or `vortex/`, by the first installed writer in that format's priority (see [`export`](#export-object-optional-v2-only) for the precedence). Use `python -m raincloud.pipeline.compliance` to measure the writer/reader matrix. +7. **export** ([`raincloud/pipeline/export/`](raincloud/pipeline/export/)) re-encodes canonical Arrow through the configured exporter cells: the formats the install builds (its `formats` setting, only Vortex by default) among those the dataset offers, each written once, to `parquet/` or `vortex/`, by the first installed writer in that format's priority (see [`export`](#export-object-optional-v2-only) for the precedence). Use `python -m raincloud.pipeline.compliance` to measure the writer/reader matrix. ## Multi-output datasets diff --git a/tests/conftest.py b/tests/conftest.py index 0e33e46..0708308 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -72,6 +72,7 @@ def _isolate_loader_cache(tmp_path, monkeypatch): Points the loader cache at a per-test tmp dir so no test can read or write the developer's real ~/.cache/raincloud (the loader's cache_root() default), + pins the build settings most tests assume (below), and clears the catalog lru_cache around each test so a snapshot/manifest set by one test never leaks into the next. Tests that need a specific cache location just set RAINCLOUD_CACHE again — a later monkeypatch.setenv wins. @@ -79,6 +80,12 @@ def _isolate_loader_cache(tmp_path, monkeypatch): monkeypatch.setenv("RAINCLOUD_CATALOG_DIR", str(tmp_path / "_catalogs")) monkeypatch.setenv("RAINCLOUD_NO_CONFIG", "1") monkeypatch.setenv("RAINCLOUD_CACHE", str(tmp_path / "_loader_cache")) + # Most tests exercise the pipeline over a dataset's Parquet and Vortex with + # everything kept; the opt-in defaults (Vortex only, nothing kept) have + # tests of their own, which clear these. + monkeypatch.setenv("RAINCLOUD_FORMATS", "parquet,vortex") + monkeypatch.setenv("RAINCLOUD_KEEP_RAW", "1") + monkeypatch.setenv("RAINCLOUD_KEEP_CANONICAL", "1") def _clear(): try: diff --git a/tests/installed_base_probe.py b/tests/installed_base_probe.py index 83cdb4e..ba253d9 100644 --- a/tests/installed_base_probe.py +++ b/tests/installed_base_probe.py @@ -53,7 +53,9 @@ assert settings.scratch_dir == root / "scratch" # Metadata stays lazy even though the catalog advertises an unavailable reader. handle = raincloud.load("tiny", config=config) -assert handle.format == "parquet" +# auto picks among the formats the install builds (only Vortex by default), then the +# canonical Arrow: a base install has no Vortex reader, so it is the Arrow file. +assert handle.format == "arrow", handle.format assert not (root / "hdd").exists() try: raincloud.load("tiny", format="vortex", config=config) diff --git a/tests/test_avro.py b/tests/test_avro.py new file mode 100644 index 0000000..29fa238 --- /dev/null +++ b/tests/test_avro.py @@ -0,0 +1,39 @@ +# SPDX-FileCopyrightText: 2026 Raincloud Maintainers +# SPDX-License-Identifier: Apache-2.0 +"""Avro: written by the sidecar lanes on request, and served by path -- pyarrow +reads no Avro, so the loader has no in-process reader for it.""" +from __future__ import annotations + +import os + +import pytest + +import raincloud +from raincloud._readers import reader_capabilities +from raincloud.catalogs import operation +from raincloud.exceptions import MissingDependency +from raincloud.pipeline import build +from raincloud.pipeline.spec import prepared_artifact +from tests.test_pipeline_contracts import _catalog +from tests.test_pipeline_contracts import stages as stages # noqa: F401 (fixture) + +SPEC = {"slug": "tiny"} + + +def test_avro_has_no_in_process_reader(): + assert reader_capabilities()["avro"] == {"available": False, "implementation": None} + + +@pytest.mark.skipif(not os.environ.get("RAINCLOUD_SIDECAR_AVRO_RS"), + reason="needs the avro@rs sidecar (RAINCLOUD_SIDECAR_AVRO_RS)") +def test_an_avro_file_is_written_on_request_and_served_by_path(tmp_path, stages): + cfg = _catalog(tmp_path, "avro", [SPEC]) + with operation(cfg): + assert build.run_one(SPEC, strict=False, formats=["avro"]) + written = prepared_artifact("tiny", "avro") + assert written.read_bytes()[:4] == b"Obj\x01" + assert written.read_bytes()[-16:] == b"raincloud-avro01" + ds = raincloud.load("tiny", format="avro", config=cfg, offline=True) + assert ds.path() == written + with pytest.raises(MissingDependency, match=r"Dataset.path\(\)"): + ds.to_arrow() diff --git a/tests/test_build_formats.py b/tests/test_build_formats.py new file mode 100644 index 0000000..a4a3988 --- /dev/null +++ b/tests/test_build_formats.py @@ -0,0 +1,144 @@ +# SPDX-FileCopyrightText: 2026 Raincloud Maintainers +# SPDX-License-Identifier: Apache-2.0 +"""Opt-in formats: a build writes the install's formats (only Vortex by +default) or the ones asked for, and removes what it was made from unless the +install keeps it -- except a canonical that is the dataset's only file.""" +from __future__ import annotations + +from dataclasses import replace + +import pytest + +import raincloud +from raincloud.catalogs import operation +from raincloud.pipeline import build, status +from raincloud.pipeline.spec import prepared_arrow, prepared_parquet, prepared_vortex, raw_slug_dir +from tests.test_export_unavailable import _broken_vortex +from tests.test_pipeline_contracts import _catalog +from tests.test_pipeline_contracts import stages as stages # noqa: F401 (fixture) + +pytest.importorskip("vortex") +# No `export.formats`: the dataset offers every format. +SPEC = {"slug": "tiny"} + + +@pytest.fixture +def defaults(monkeypatch): + """The settings an install has when it sets none (conftest pins others).""" + for name in ("RAINCLOUD_FORMATS", "RAINCLOUD_KEEP_RAW", "RAINCLOUD_KEEP_CANONICAL"): + monkeypatch.delenv(name, raising=False) + + +def _store(tmp_path, datasets=(SPEC,), **settings): + cfg = replace(_catalog(tmp_path, "formats", list(datasets)), **settings) + with operation(cfg): + raw = raw_slug_dir("tiny") + raw.mkdir(parents=True) + (raw / "upstream.csv").write_text("x\n1\n") + return cfg, raw + + +def test_a_default_build_writes_only_vortex_and_keeps_nothing(tmp_path, stages, defaults): + cfg, raw = _store(tmp_path) + assert cfg.formats == ("vortex",) and not cfg.keep_raw and not cfg.keep_canonical + with operation(cfg): + assert build.run_one(SPEC, strict=False) + assert prepared_vortex("tiny").is_file() + assert not prepared_parquet("tiny").exists() + assert not prepared_arrow("tiny").exists() + assert not raw.exists() + assert raincloud.load("tiny", config=cfg, offline=True).format == "vortex" + + +def test_the_canonical_stays_when_no_format_was_written(tmp_path, stages, defaults, monkeypatch): + cfg, _ = _store(tmp_path) + _broken_vortex(monkeypatch) + with operation(cfg): + assert build.run_one(SPEC, strict=False) + assert prepared_arrow("tiny").is_file() and not prepared_vortex("tiny").exists() + # Vortex is measured unavailable, so the canonical is what `auto` serves. + assert raincloud.load("tiny", config=cfg, offline=True).format == "arrow" + + +def test_keep_settings_keep_the_raw_download_and_the_canonical(tmp_path, stages, defaults): + cfg, raw = _store(tmp_path, keep_raw=True, keep_canonical=True) + with operation(cfg): + assert build.run_one(SPEC, strict=False) + assert prepared_arrow("tiny").is_file() and prepared_vortex("tiny").is_file() + assert (raw / "upstream.csv").is_file() + + +def test_format_flag_writes_only_what_is_asked(tmp_path, stages, defaults): + cfg, _ = _store(tmp_path) + with operation(cfg): + assert build._main(["tiny", "--format", "parquet"]) == 0 + assert prepared_parquet("tiny").is_file() + assert not prepared_vortex("tiny").exists() and not prepared_arrow("tiny").exists() + # Asking for arrow keeps the canonical; nothing else is written. + assert build._main(["tiny", "--format", "arrow"]) == 0 + assert prepared_arrow("tiny").is_file() and not prepared_vortex("tiny").exists() + + +def test_formats_all_writes_every_offered_format(tmp_path, stages, defaults): + cfg, _ = _store(tmp_path, formats=("all",)) + with operation(cfg): + assert build.run_one(SPEC, strict=False) + assert prepared_parquet("tiny").is_file() and prepared_vortex("tiny").is_file() + + +def test_formats_all_leaves_out_a_format_with_no_installed_writer(tmp_path, stages, defaults, monkeypatch, + capsys): + from raincloud._formats import WRITERS + from raincloud.pipeline.export import get_exporter + for writer in WRITERS["orc"]: + monkeypatch.setattr(get_exporter(f"orc@{writer}"), "unavailable", lambda: "not on this machine") + cfg, _ = _store(tmp_path, formats=("all",)) + with operation(cfg): + assert build.run_one(SPEC, strict=False) + assert prepared_vortex("tiny").is_file() + # Named outright, it fails the build instead. + assert not build.run_one(SPEC, strict=False, formats=["orc"]) + assert "[skip] orc: tiny: no installed writer for 'orc'" in capsys.readouterr().out + + +def test_a_format_the_dataset_does_not_offer_fails_the_build(tmp_path, stages, defaults, capsys): + narrow = {"slug": "tiny", "export": {"formats": ["parquet"]}} + cfg, _ = _store(tmp_path, datasets=(narrow,)) + with operation(cfg): + assert not build.run_one(narrow, strict=False, formats=["vortex"]) + assert "tiny does not offer vortex (it offers parquet)" in capsys.readouterr().out + + +def test_unknown_format_names_are_refused_with_a_suggestion(tmp_path, stages, defaults, capsys): + with pytest.raises(ValueError, match="parquet"): + raincloud.resolve_config(no_config=True, formats="parqet") + with pytest.raises(ValueError, match="keep_canonical"): + raincloud.resolve_config(no_config=True, formats="arrow") + cfg, _ = _store(tmp_path) + with operation(cfg), pytest.raises(SystemExit): + build._main(["tiny", "--format", "vortx"]) + assert "vortex" in capsys.readouterr().err + + +def test_status_counts_a_dataset_complete_without_what_it_does_not_keep(tmp_path, stages, defaults): + cfg, _ = _store(tmp_path) + with operation(cfg): + assert build.run_one(SPEC, strict=False) + manifest = {"schema_version": 2, "datasets": [SPEC]} + row = status.gather(SPEC, manifest, fast=True) + assert not row["raw"].get("present") and not row["arrow"].get("present") + assert row["parquet"] == {"expected": False} + assert not status._is_incomplete(row) + + +def test_a_v1_catalog_loads_and_builds_as_before(tmp_path, defaults): + """Install formats arrived in 0.3.1; a v1 catalog keeps 0.3.0's behaviour so + its users do not break: `auto` tries vortex, parquet, arrow, and a build + writes what the recipe lists.""" + from raincloud._catalog import Entry, FormatInfo + from raincloud._formats import build_formats + cfg = raincloud.resolve_config(no_config=True) + parquet_only = Entry("old", 1, formats={"parquet": FormatInfo(None, 10)}, version=1) + assert raincloud._choose_format(parquet_only, "auto", False, config=cfg) == "parquet" + v1 = {"slug": "old", "convert": {"vortex": True}} + assert build_formats(v1, 1, cfg) == ["parquet", "vortex"] diff --git a/tests/test_catalog_provenance.py b/tests/test_catalog_provenance.py index d00b255..3ec332c 100644 --- a/tests/test_catalog_provenance.py +++ b/tests/test_catalog_provenance.py @@ -91,7 +91,7 @@ def test_streaming_template_builds_canonical_and_exports(tmp_path, monkeypatch): manifest.write_text(json.dumps({"schema_version": 2, "datasets": [spec]})) cfg = raincloud.resolve_config(manifest=manifest, data_dir=tmp_path / "hdd", scratch_dir=tmp_path / "scratch") with operation(cfg): - assert run_one(spec, strict=True) + assert run_one(spec, strict=True, formats=["parquet", "arrow"]) ds = raincloud.load("template-probe", format="arrow", config=cfg, offline=True) assert ds.to_arrow().to_pydict() == {"x": [1, 2], "s": ["a", "b"]} assert raincloud.load("template-probe", format="parquet", config=cfg, offline=True).to_arrow().equals(ds.to_arrow()) diff --git a/tests/test_compare_cases.py b/tests/test_compare_cases.py new file mode 100644 index 0000000..a6d41f9 --- /dev/null +++ b/tests/test_compare_cases.py @@ -0,0 +1,28 @@ +# SPDX-FileCopyrightText: 2026 Raincloud Maintainers +# SPDX-License-Identifier: Apache-2.0 +"""The Python comparator gives every shared comparison case its verdict +(sidecars/compare_cases: the Rust and JVM comparators read the same files).""" +from __future__ import annotations + +import json + +import pyarrow as pa +import pytest + +from raincloud.pipeline.export.compare import values_equal +from raincloud.pipeline.spec import REPO_ROOT + +CASES = REPO_ROOT / "sidecars" / "compare_cases" + + +def _read(path): + with pa.memory_map(str(path), "r") as source: + return pa.ipc.open_file(source).read_all() + + +@pytest.mark.parametrize("case", json.loads((CASES / "cases.json").read_text())["cases"], ids=lambda c: c["name"]) +def test_shared_comparison_case(case): + got = _read(CASES / f"{case['name']}.got.arrow") + expected = _read(CASES / f"{case['name']}.expected.arrow") + equal, detail = values_equal(got, expected) + assert equal == (case["verdict"] == "equal"), detail diff --git a/tests/test_compliance.py b/tests/test_compliance.py index cc7c99e..ebd43e0 100644 --- a/tests/test_compliance.py +++ b/tests/test_compliance.py @@ -24,6 +24,7 @@ import pyarrow as pa import pyarrow.parquet as pq +from raincloud._registry import PY_EXPORTERS, SIDECAR_EXPORTERS from raincloud.pipeline import compliance from raincloud.pipeline.export import run_reader from raincloud.pipeline.export.exporters import ParquetExporter, VortexExporter @@ -344,14 +345,15 @@ def test_run_compliance_matrix_over_synthetic_slug(tmp_path, monkeypatch, capsys # In-process cells produced artifacts; the absent sidecars were skipped (no # RAINCLOUD_SIDECAR_* env override + the binary names are not on PATH here). - assert set(sc.artifact_cells()) == {"parquet@py", "vortex@py"} + assert set(sc.artifact_cells()) == set(PY_EXPORTERS) skipped = {c for c, _ in sc.skipped_cells} - assert skipped == {"parquet@rs", "parquet@java", "parquet@hardwood", "vortex@rs", "vortex@jni"} + assert skipped == set(SIDECAR_EXPORTERS) verdict_by = {(rr.artifact_cell, rr.reader_id): rr.verdict.status for rr in sc.read_results} # In-process readers over their own format -> pass. assert verdict_by[("parquet@py", "parquet@py")] == "pass" assert verdict_by[("vortex@py", "vortex@py")] == "pass" + assert verdict_by[("orc@py", "orc@py")] == "pass" # Cross-format -> na. assert verdict_by[("parquet@py", "vortex@py")] == "na" assert verdict_by[("vortex@py", "parquet@py")] == "na" @@ -360,7 +362,7 @@ def test_run_compliance_matrix_over_synthetic_slug(tmp_path, monkeypatch, capsys assert verdict_by[("vortex@py", "vortex@jni")] == "skip" counts = sc.counts() - assert counts["pass"] == 2 + assert counts["pass"] == len(PY_EXPORTERS) assert counts["fail"] == 0 assert counts["skip"] >= 2 and counts["na"] >= 2 diff --git a/tests/test_dataset_readers.py b/tests/test_dataset_readers.py index f86bd34..d473b35 100644 --- a/tests/test_dataset_readers.py +++ b/tests/test_dataset_readers.py @@ -7,6 +7,7 @@ import pytest import raincloud +from raincloud._formats import ALL_FORMATS from raincloud.exceptions import ArtifactNotFound, FormatUnavailable from tests.reader_fixture import create @@ -30,7 +31,7 @@ def test_streaming_projection_and_early_close(prepared, fmt): with ds.batches(batch_size=1) as batches: assert next(batches).num_rows == 1 assert ds.schema.names == table.schema.names - assert len(ds.artifacts) == 3 + assert len(ds.artifacts) == len(ALL_FORMATS) def test_explicit_format_never_falls_back(prepared): diff --git a/tests/test_docs.py b/tests/test_docs.py index 0c161a6..68ae6e5 100644 --- a/tests/test_docs.py +++ b/tests/test_docs.py @@ -27,6 +27,11 @@ } +def _artifacts(tmp_path, paths): + """A `prepared_artifact` stand-in: `paths[fmt]`, else a file that never exists.""" + return lambda slug, fmt, manifest=None: paths.get(fmt, tmp_path / f"missing.{fmt}") + + @pytest.fixture def patched_docs(tmp_path, monkeypatch): """Redirect docs.py I/O to tmp_path and stub the manifest + path helpers.""" @@ -42,7 +47,7 @@ def patched_docs(tmp_path, monkeypatch): # → forces the code through the missing-on-disk branch. monkeypatch.setattr(docs, "prepared_parquet", lambda slug: tmp_path / "missing.parquet") monkeypatch.setattr(docs, "prepared_vortex", lambda slug: tmp_path / "missing.vortex") - monkeypatch.setattr(docs, "prepared_arrow", lambda slug: tmp_path / "missing.arrow.zstd") + monkeypatch.setattr(docs, "prepared_artifact", _artifacts(tmp_path, {"parquet": tmp_path / "missing.parquet", "vortex": tmp_path / "missing.vortex", "arrow": tmp_path / "missing.arrow.zstd"})) return docs, tmp_path @@ -59,7 +64,7 @@ def test_snapshot_preserves_absent_formats_individually(patched_docs, monkeypatc docs.SNAPSHOT_JSON.write_text(json.dumps({"schema_version": 1, "slugs": {"fake-slug": prior}})) table = pa.table({"new": [1, 2]}) path = root / f"local.{present}" - monkeypatch.setattr(docs, f"prepared_{present}", lambda slug: path) + monkeypatch.setattr(docs, "prepared_artifact", _artifacts(root, {present: path})) if present == "parquet": pq.write_table(table, path) else: @@ -180,6 +185,7 @@ def test_datasets_md_falls_back_to_tracked_v1_snapshot(tmp_path, monkeypatch): ) monkeypatch.setattr(docs, "prepared_parquet", lambda slug: tmp_path / "missing.parquet") monkeypatch.setattr(docs, "prepared_vortex", lambda slug: tmp_path / "missing.vortex") + monkeypatch.setattr(docs, "prepared_artifact", _artifacts(tmp_path, {"parquet": tmp_path / "missing.parquet", "vortex": tmp_path / "missing.vortex"})) docs.generate_datasets_md() row = _fake_slug_row((tmp_path / "datasets.md").read_text()) @@ -233,6 +239,7 @@ def test_datasets_md_v2_does_not_use_v1_snapshot(tmp_path, monkeypatch): lambda: {"schema_version": 2, "datasets": [dict(_FAKE_SPEC)]}) monkeypatch.setattr(docs, "prepared_parquet", lambda slug: tmp_path / "missing.parquet") monkeypatch.setattr(docs, "prepared_vortex", lambda slug: tmp_path / "missing.vortex") + monkeypatch.setattr(docs, "prepared_artifact", _artifacts(tmp_path, {"parquet": tmp_path / "missing.parquet", "vortex": tmp_path / "missing.vortex"})) docs.generate_datasets_md() row = _fake_slug_row((tmp_path / "datasets.md").read_text()) @@ -266,6 +273,7 @@ def test_snapshot_captures_row_groups_for_built_slugs(tmp_path, monkeypatch): ) monkeypatch.setattr(docs, "prepared_parquet", lambda slug: fake_pq) monkeypatch.setattr(docs, "prepared_vortex", lambda slug: tmp_path / "missing.vortex") + monkeypatch.setattr(docs, "prepared_artifact", _artifacts(tmp_path, {"parquet": fake_pq, "vortex": tmp_path / "missing.vortex"})) docs.generate_snapshot(overwrite_missing=True) @@ -297,7 +305,7 @@ def test_snapshot_captures_arrow_when_present(tmp_path, monkeypatch): ) monkeypatch.setattr(docs, "prepared_parquet", lambda slug: tmp_path / "missing.parquet") monkeypatch.setattr(docs, "prepared_vortex", lambda slug: tmp_path / "missing.vortex") - monkeypatch.setattr(docs, "prepared_arrow", lambda slug: arrow) + monkeypatch.setattr(docs, "prepared_artifact", _artifacts(tmp_path, {"parquet": tmp_path / "missing.parquet", "vortex": tmp_path / "missing.vortex", "arrow": arrow})) docs.generate_snapshot(overwrite_missing=True) @@ -319,7 +327,7 @@ def test_snapshot_omits_arrow_keys_when_absent(tmp_path, monkeypatch): ) monkeypatch.setattr(docs, "prepared_parquet", lambda slug: tmp_path / "missing.parquet") monkeypatch.setattr(docs, "prepared_vortex", lambda slug: tmp_path / "missing.vortex") - monkeypatch.setattr(docs, "prepared_arrow", lambda slug: tmp_path / "missing.arrow.zstd") + monkeypatch.setattr(docs, "prepared_artifact", _artifacts(tmp_path, {"parquet": tmp_path / "missing.parquet", "vortex": tmp_path / "missing.vortex", "arrow": tmp_path / "missing.arrow.zstd"})) docs.generate_snapshot(overwrite_missing=True) @@ -354,7 +362,7 @@ def test_snapshot_arrow_only_slug_written_on_default_regen(tmp_path, monkeypatch ) monkeypatch.setattr(docs, "prepared_parquet", lambda slug: tmp_path / "missing.parquet") monkeypatch.setattr(docs, "prepared_vortex", lambda slug: tmp_path / "missing.vortex") - monkeypatch.setattr(docs, "prepared_arrow", lambda slug: arrow) + monkeypatch.setattr(docs, "prepared_artifact", _artifacts(tmp_path, {"parquet": tmp_path / "missing.parquet", "vortex": tmp_path / "missing.vortex", "arrow": arrow})) docs.generate_snapshot(overwrite_missing=False) diff --git a/tests/test_docs_contracts.py b/tests/test_docs_contracts.py index 3ac9222..f9ff3a9 100644 --- a/tests/test_docs_contracts.py +++ b/tests/test_docs_contracts.py @@ -363,6 +363,21 @@ def test_agents_row_group_defaults_match_the_code(monkeypatch): assert row.rstrip(" |").endswith(spelled), row +def test_agents_parquet_settings_default_to_each_writers_own(monkeypatch): + from raincloud.pipeline import spec + table = (ROOT / "AGENTS.md").read_text() + for var in spec.PARQUET_SETTING_VARS: + monkeypatch.delenv(var, raising=False) + row = next(line for line in table.splitlines() if line.startswith(f"| `{var}`")) + assert "unset (each writer's own" in row, row + assert spec.parquet_page_options().chosen() == {} + for settings in spec.FORMAT_SETTINGS.values(): + for _, var, _ in settings: + monkeypatch.delenv(var, raising=False) + row = next(line for line in table.splitlines() if line.startswith(f"| `{var}`")) + assert "unset (" in row, row + + def test_hydrate_bypass_disables_every_layer_as_documented(): """HYDRATING.md: the scheme allowlist and blocked_hosts_extra apply unless the two-flag bypass is active, and the bypass disables every layer.""" diff --git a/tests/test_export_readback.py b/tests/test_export_readback.py index de3612a..d9c1845 100644 --- a/tests/test_export_readback.py +++ b/tests/test_export_readback.py @@ -32,7 +32,7 @@ def _entry(cfg, fmt): def _wrong_parquet(monkeypatch): """parquet@py writes a valid file holding other values than the canonical's.""" - def write(canonical, dest, row_group, byte_target, *, compression, stats): + def write(canonical, dest, row_group, byte_target, *, options): with pa.ipc.open_file(str(canonical)) as reader: table = reader.read_all() pq.write_table(table.set_column(0, "x", pa.array([7] * table.num_rows)), dest) diff --git a/tests/test_exporters.py b/tests/test_exporters.py index 10c5c52..dbe39fe 100644 --- a/tests/test_exporters.py +++ b/tests/test_exporters.py @@ -37,7 +37,7 @@ def test_run_exporters_default_produces_parquet_and_vortex(tmp_path, monkeypatch results = run_exporters({"slug": slug}, canonical_path) - # Both default cells ran, tagged with their qualified ledger cell-ids. + # The install's formats (conftest: parquet, vortex), tagged with their qualified ledger cell-ids. assert {r.format_id for r in results} == {"parquet@py", "vortex@py"} for r in results: assert r.nbytes > 0 and len(r.sha256) == 64 diff --git a/tests/test_extract_identity.py b/tests/test_extract_identity.py index e471a4a..10c86fe 100644 --- a/tests/test_extract_identity.py +++ b/tests/test_extract_identity.py @@ -47,8 +47,9 @@ def test_extract_cli_catalog_collision_repeat_and_build_cleanup(tmp_path): manifests = [manifest(tmp_path, [item], f"catalog{i}") for i, item in enumerate((first, second))] config = tmp_path / "config.toml" + # Raw bytes are kept: the test counts them across recipes. config.write_text('[raincloud]\ndata_dir = "data"\nscratch_dir = "scratch"\n' - 'catalog_dir = "catalogs"\n') + 'catalog_dir = "catalogs"\nkeep_raw = true\n') env = {k: v for k, v in os.environ.items() if not k.startswith("RAINCLOUD_")} env["RAINCLOUD_CONFIG"] = str(config) diff --git a/tests/test_generated_groups.py b/tests/test_generated_groups.py new file mode 100644 index 0000000..145afc4 --- /dev/null +++ b/tests/test_generated_groups.py @@ -0,0 +1,75 @@ +# SPDX-FileCopyrightText: 2026 Raincloud Maintainers +# SPDX-License-Identifier: Apache-2.0 +"""A generated table is one of a group its generator writes at once: building +one builds the group, its generator output goes once the group has built +(unless keep_raw keeps it), and `--only` builds just the tables named.""" +from __future__ import annotations + +from dataclasses import replace + +import pytest + +import raincloud +from raincloud import _bundle, catalogs +from raincloud.catalogs import operation +from raincloud.pipeline import build, generate +from raincloud.pipeline.spec import prepared_artifact +from tests.test_generated import FixtureGenerator + + +def _spec(name): + return {"slug": f"fixture-{name}", "fetch": {"type": "generated", "generator": "fixture", "version": "1", + "parameters": {"size": 2}, "output": name}, + "extract": {"type": "passthrough"}, "parse": {"reader": "parquet"}, + "transform": {"handler": "identity"}} + + +@pytest.fixture +def group(tmp_path, monkeypatch): + monkeypatch.delenv("RAINCLOUD_KEEP_RAW") # conftest pins it; these tests need the default + producer = FixtureGenerator() + monkeypatch.setitem(generate.REGISTRY, "fixture", producer) + caps = _bundle.capabilities() + caps["builders"].append("generator:fixture") + monkeypatch.setattr(catalogs, "capabilities", lambda: caps) + specs = [_spec("left"), _spec("right")] + bundle = _bundle.make_bundle(_bundle.encode({"schema_version": 2, "datasets": specs}), + _bundle.encode({"schema_version": 2, "slugs": {}}), "groups") + directory = tmp_path / "catalog" + directory.mkdir() + for filename, raw in bundle.files().items(): + (directory / filename).write_bytes(raw) + cfg = raincloud.resolve_config(no_config=True, catalog=str(directory), data_dir=tmp_path / "data", + raw_dir=tmp_path / "raw", scratch_dir=tmp_path / "scratch", + catalog_dir=tmp_path / "catalogs", formats="parquet") + return cfg, producer, specs + + +def _built(cfg, name): + with operation(cfg): + return prepared_artifact(f"fixture-{name}", "parquet").is_file() + + +def test_building_one_table_builds_its_group_and_then_cleans_the_generator_output(group): + cfg, producer, specs = group + with operation(cfg): + assert build._main(["fixture-left"]) == 0 + root = generate.group_root(specs[0]["fetch"]) + assert _built(cfg, "left") and _built(cfg, "right") + assert producer.calls == 1 and not root.exists() + + +def test_keep_raw_keeps_the_generator_output(group): + cfg, _, specs = group + cfg = replace(cfg, keep_raw=True) + with operation(cfg): + assert build._main(["fixture-left"]) == 0 + assert generate.group_root(specs[0]["fetch"]).exists() + + +def test_only_builds_just_the_named_table(group): + cfg, _, specs = group + with operation(cfg): + assert build._main(["fixture-left", "--only"]) == 0 + assert not generate.group_root(specs[0]["fetch"]).exists() + assert _built(cfg, "left") and not _built(cfg, "right") diff --git a/tests/test_loader_arrow.py b/tests/test_loader_arrow.py index 66f4093..2fb3557 100644 --- a/tests/test_loader_arrow.py +++ b/tests/test_loader_arrow.py @@ -15,6 +15,8 @@ import pyarrow as pa import pytest +from raincloud._formats import ALL_FORMATS + @pytest.mark.parametrize("extra,formats", [ ({"export": {"formats": []}}, {"arrow"}), @@ -22,7 +24,7 @@ ({"export": {"formats": ["vortex"]}}, {"arrow", "vortex"}), ({"export": {"formats": ["parquet"], "priority": ["rs"]}}, {"arrow", "parquet"}), ({"export": {"formats": ["parquet"], "notes": "x"}}, {"arrow", "parquet"}), - ({}, {"arrow", "parquet", "vortex"}), + ({}, set(ALL_FORMATS)), ]) def test_unbuilt_v2_catalog_uses_export_policy(extra, formats): from raincloud._catalog import Catalog @@ -104,7 +106,7 @@ def test_catalog_recognizes_arrow_and_carries_version(tmp_path, monkeypatch): _catalog = _write_catalog(tmp_path, monkeypatch, snapshot, manifest) try: e = _catalog.load_catalog().entry("tiny") - assert set(e.formats) == {"parquet", "vortex", "arrow"} + assert set(e.formats) == set(ALL_FORMATS) assert e.formats["arrow"].sha256 == "cc" * 32 assert e.formats["arrow"].nbytes == 90 assert e.version == 2 diff --git a/tests/test_loader_contracts.py b/tests/test_loader_contracts.py index a4e02cf..c7008f3 100644 --- a/tests/test_loader_contracts.py +++ b/tests/test_loader_contracts.py @@ -66,7 +66,8 @@ def cli(capsys, options, *args): @pytest.mark.parametrize("empty", [False, True]) def test_subprocess_env_round_trips_every_setting(tmp_path, monkeypatch, key, empty): values = {"export_priority": "rs,py", "offline": True, "retry_errors": True, "mirror": "s3://bucket/prefix", - "catalog": "checkout", "catalog_url": "https://example.com/catalogs"} + "catalog": "checkout", "catalog_url": "https://example.com/catalogs", "formats": "parquet,vortex", + "keep_raw": True, "keep_canonical": True} value = values.get(key, str(tmp_path / key)) parent = resolve_config(no_config=True, **({} if empty else {key: value})) for name, setting in parent.subprocess_env().items(): @@ -255,10 +256,17 @@ def test_cli_resolution_does_not_need_a_python_reader(fixture, no_vortex, capsys assert reply["path"].endswith("tiny.vortex") and reply["catalog_revision"] code, out, _ = cli(capsys, options, "--json", "describe", "tiny") assert json.loads(out)["format"] == "vortex" + # Parquet is opt-in: an install that does not build it gets the canonical + # Arrow when it cannot read Vortex, even with a Parquet file present... code, out, _ = cli(capsys, options, "--json", "describe", "tiny", "--readers", "arrow,parquet") + assert json.loads(out)["format"] == "arrow" + # ...and Parquet once it opts in. + code, out, _ = cli(capsys, {**options, "formats": "vortex,parquet"}, "--json", "describe", "tiny", + "--readers", "arrow,parquet") assert json.loads(out)["format"] == "parquet" # Python reads still need their reader. - assert raincloud.load("tiny", config=cfg).format == "parquet" + assert raincloud.load("tiny", config=cfg).format == "arrow" + assert raincloud.load("tiny", config=replace(cfg, formats=("vortex", "parquet"))).format == "parquet" with pytest.raises(MissingDependency): raincloud.load("tiny", format="vortex", config=cfg) diff --git a/tests/test_manifest.py b/tests/test_manifest.py index 23aeab7..38bb5e8 100644 --- a/tests/test_manifest.py +++ b/tests/test_manifest.py @@ -358,15 +358,13 @@ def test_export_cross_check_names_formats_and_writers(manifest): def test_live_manifest_export_policy_is_writer_priority_only(manifest, schema): - """The live sources.json is v2. Its export blocks either list both formats - (`formats: ["parquet", "vortex"]`, the specs whose Vortex writer once failed: - the build now measures that) or, for the SF100 TPC specs, prefer arrow-rs for - Parquet only (a per-format priority map); schema and cross-checks pass.""" + """The live sources.json is v2. No dataset narrows its formats (every one + offers all; an install picks what it builds), and the export blocks there + are, for the SF100 TPC specs, a preference for arrow-rs for Parquet only (a + per-format priority map); schema and cross-checks pass.""" assert manifest["schema_version"] == 2 exports = {d["slug"]: d["export"] for d in manifest["datasets"] if "export" in d} - assert all(set(e) <= {"formats", "priority", "notes"} for e in exports.values()), exports - formats = {slug: e["formats"] for slug, e in exports.items() if "formats" in e} - assert formats and all(f == ["parquet", "vortex"] for f in formats.values()), formats + assert all(set(e) <= {"priority", "notes"} for e in exports.values()), exports priorities = {slug: e["priority"] for slug, e in exports.items() if "priority" in e} assert all(p == {"parquet": ["rs", "py"]} for p in priorities.values()), priorities sf100 = {d["slug"] for d in manifest["datasets"] if "-sf100-" in d["slug"]} diff --git a/tests/test_manifest_policy.py b/tests/test_manifest_policy.py index 580061d..7b20eb0 100644 --- a/tests/test_manifest_policy.py +++ b/tests/test_manifest_policy.py @@ -14,6 +14,7 @@ from raincloud._bundle import build_requirements, validate_documents from raincloud._formats import ( DEFAULT_EXPORT_PRIORITY, + EXPORTED_FORMATS, WRITERS, buildable_formats, export_cells, @@ -86,9 +87,10 @@ def test_priority_rejects_a_bad_shape(): def test_export_cells_per_format(): s = {"export": {"priority": {"parquet": ["rs", "py"]}}} - assert export_cells(s) == ["parquet@rs", "vortex@py"] - assert export_cells({"export": {"priority": ["java", "py"]}}) == ["parquet@java", "vortex@py"] - assert export_cells({}, {"schema_version": 2, "export_priority": {"vortex": ["rs"]}}) == ["parquet@py", "vortex@rs"] + assert export_cells(s) == ["parquet@rs", "vortex@py", "orc@py", "avro@rs", "nimble@cpp"] + assert export_cells({"export": {"priority": ["java", "py"]}}) == ["parquet@java", "vortex@py", "orc@py", "avro@java", "nimble@cpp"] + assert export_cells({}, {"schema_version": 2, "export_priority": {"vortex": ["rs"]}}) == \ + ["parquet@py", "vortex@rs", "orc@py", "avro@rs", "nimble@cpp"] def test_resolve_export_cell_skips_uninstalled_and_unknown(): @@ -110,13 +112,13 @@ def test_sf100_specs_prefer_rs_for_parquet_only(): assert len(sf100) == 32 for d in sf100: assert d["export"]["priority"] == {"parquet": ["rs", "py"]}, d["slug"] - assert export_cells(d, m) == ["parquet@rs", "vortex@py"] + assert export_cells(d, m) == ["parquet@rs", "vortex@py", "orc@py", "avro@rs", "nimble@cpp"] # ---------- export.formats is the one v2 declaration ---------- def test_v2_reads_export_formats_only(): - assert export_formats({}) == ["parquet", "vortex"] + assert export_formats({}) == list(EXPORTED_FORMATS) assert export_formats({"export": {"formats": ["parquet"]}}) == ["parquet"] assert vortex_cells({"export": {"formats": ["parquet"]}}, 2) == [] assert buildable_formats({"export": {"formats": []}}, 2) == {"arrow"} @@ -363,7 +365,19 @@ def test_released_v2_catalog_with_convert_still_reads(): {"slug": "old-default", "convert": {"vortex": True}}]} make_bundle(encode(manifest), encode({"schema_version": 2, "slugs": {}}), "released") assert export_formats(manifest["datasets"][0], 2) == ["parquet"] - assert export_formats(manifest["datasets"][1], 2) == ["parquet", "vortex"] + assert export_formats(manifest["datasets"][1], 2) == list(EXPORTED_FORMATS) + + +def test_offered_formats_are_not_part_of_the_recipe(): + """Since formats are the install's choice, `export.formats` decides no + file's bytes; an `export` emptied without it fingerprints like none.""" + from raincloud._bundle import recipe_hash + base = {"slug": "x", "fetch": {"type": "http"}} + narrowed = {**base, "export": {"formats": ["parquet"]}} + assert recipe_hash(narrowed, 2, specs=None) == recipe_hash(base, 2, specs=None) + prioritised = {**base, "export": {"priority": ["rs", "py"]}} + assert (recipe_hash({**base, "export": {"formats": ["vortex"], "priority": ["rs", "py"]}}, 2, specs=None) + == recipe_hash(prioritised, 2, specs=None) != recipe_hash(base, 2, specs=None)) def test_recipe_keys_only_grow(): @@ -398,10 +412,40 @@ class Unset: def test_a_priority_naming_no_writer_for_a_format_falls_back_to_the_default(): - assert export_cells({"export": {"priority": ["java"]}}) == ["parquet@java", "vortex@py"] + assert export_cells({"export": {"priority": ["java"]}}) == ["parquet@java", "vortex@py", "orc@py", "avro@java", "nimble@cpp"] def test_format_sets_derive_from_the_writers(): from raincloud._formats import ALL_FORMATS, EXPORTED_FORMATS - assert set(EXPORTED_FORMATS) == {base for base in WRITERS if base != "arrow"} == {"parquet", "vortex"} + assert set(EXPORTED_FORMATS) == {base for base in WRITERS if base != "arrow"} == {"parquet", "vortex", "orc", "avro", "nimble"} assert set(ALL_FORMATS) == set(WRITERS) + + +def test_the_built_in_order_names_a_writer_for_every_format(): + """A dataset with no priority still exports every format it offers.""" + from raincloud._formats import EXPORTED_FORMATS + for fmt in EXPORTED_FORMATS: + assert set(WRITERS[fmt]) & set(DEFAULT_EXPORT_PRIORITY), fmt + + +def test_every_artifact_format_is_declared_once(): + """A format is declared in `_registry.FORMATS`; the extension map, reader + capabilities and `auto` order are derived from it, never restated.""" + from raincloud._cache import EXT + from raincloud._formats import ALL_FORMATS, AUTO_FORMATS + from raincloud._readers import reader_capabilities + from raincloud._registry import FORMATS + assert set(ALL_FORMATS) <= set(FORMATS) + assert set(EXT) == set(reader_capabilities()) == set(FORMATS) + assert AUTO_FORMATS == ("vortex", "parquet", "arrow") + + +def test_the_schema_names_exactly_the_exported_formats(): + """sources.schema.json cannot import the registry, so this is its gate: a + format added to (or removed from) the registry must be added to the schema's + `export.formats` and `export.priority` enums too.""" + from raincloud._formats import EXPORTED_FORMATS + defs = SCHEMA["$defs"] + assert defs["Export"]["properties"]["formats"]["items"]["enum"] == list(EXPORTED_FORMATS) + by_format = next(option for option in defs["WriterPriority"]["oneOf"] if option.get("type") == "object") + assert by_format["propertyNames"]["enum"] == list(EXPORTED_FORMATS) diff --git a/tests/test_nimble.py b/tests/test_nimble.py new file mode 100644 index 0000000..30d227f --- /dev/null +++ b/tests/test_nimble.py @@ -0,0 +1,36 @@ +# SPDX-FileCopyrightText: 2026 Raincloud Maintainers +# SPDX-License-Identifier: Apache-2.0 +"""Nimble: written by `nimble@cpp` (upstream Nimble's C++, built by +sidecars/nimble/build.sh) on request, served by path, absent where it is not built.""" +from __future__ import annotations + +import os + +import pytest + +import raincloud +from raincloud.catalogs import operation +from raincloud.pipeline import build +from raincloud.pipeline.export import cell_available +from raincloud.pipeline.spec import prepared_artifact +from tests.test_pipeline_contracts import _catalog +from tests.test_pipeline_contracts import stages as stages # noqa: F401 (fixture) + +SPEC = {"slug": "tiny"} +BUILT = bool(os.environ.get("RAINCLOUD_SIDECAR_NIMBLE_CPP")) + + +def test_the_lane_is_absent_without_its_binary(monkeypatch, tmp_path): + monkeypatch.delenv("RAINCLOUD_SIDECAR_NIMBLE_CPP", raising=False) + monkeypatch.setenv("PATH", str(tmp_path)) + assert not cell_available("nimble@cpp") + + +@pytest.mark.skipif(not BUILT, reason="needs the nimble@cpp binaries (sidecars/nimble/build.sh)") +def test_a_nimble_file_is_written_on_request_and_served_by_path(tmp_path, stages): + cfg = _catalog(tmp_path, "nimble", [SPEC]) + with operation(cfg): + assert build.run_one(SPEC, strict=False, formats=["nimble"]) + written = prepared_artifact("tiny", "nimble") + assert written.is_file() + assert raincloud.load("tiny", format="nimble", config=cfg, offline=True).path() == written diff --git a/tests/test_orc.py b/tests/test_orc.py new file mode 100644 index 0000000..f2dcd58 --- /dev/null +++ b/tests/test_orc.py @@ -0,0 +1,60 @@ +# SPDX-FileCopyrightText: 2026 Raincloud Maintainers +# SPDX-License-Identifier: Apache-2.0 +"""ORC: written by `orc@py` (pyarrow, the Apache ORC C++ library) on request, +read back by the loader, and a type the library does not write is measured, +never converted for it.""" +from __future__ import annotations + +import pyarrow as pa +import pytest + +import raincloud +from raincloud import _builds +from raincloud._resolve import artifact_key +from raincloud.catalogs import operation +from raincloud.pipeline import build +from raincloud.pipeline.spec import prepared_artifact +from tests.test_pipeline_contracts import TABLE, _catalog +from tests.test_pipeline_contracts import stages as stages # noqa: F401 (fixture) + +pytest.importorskip("pyarrow._orc") +SPEC = {"slug": "tiny"} + + +def test_an_orc_file_is_written_on_request_and_loads(tmp_path, stages): + cfg = _catalog(tmp_path, "orc", [SPEC]) + with operation(cfg): + assert build.run_one(SPEC, strict=False, formats=["orc"]) + assert prepared_artifact("tiny", "orc").is_file() + ds = raincloud.load("tiny", format="orc", config=cfg, offline=True) + assert ds.path().name == "tiny.orc" + assert ds.to_arrow().equals(TABLE) + with ds.batches(batch_size=1, columns=["x"]) as batches: + assert [b.to_pydict() for b in batches] == [{"x": [1]}, {"x": [2]}] + assert ds.dataset().count_rows() == 2 + # ORC is opt-in and never what `auto` picks. + assert raincloud.load("tiny", config=cfg, offline=True).format != "orc" + + +def test_unsigned_and_view_columns_are_widened_and_read_back_equal(tmp_path, stages, monkeypatch): + table = pa.table({"u8": pa.array([1, None], pa.uint8()), "u64": pa.array([2**64 - 1, 0], pa.uint64()), + "v": pa.array(["a", None], pa.string_view()), + "nested": pa.array([{"u": 1}, None], pa.struct([("u", pa.uint32())]))}) + monkeypatch.setattr(build, "transform", lambda spec, tables: [(spec["slug"], table)]) + cfg = _catalog(tmp_path, "orc-widen", [SPEC]) + with operation(cfg): + assert build.run_one(SPEC, strict=False, formats=["orc@py"]) + stored = raincloud.load("tiny", format="orc", config=cfg, offline=True).schema + assert [stored.field(n).type for n in ("u8", "u64", "v")] == [pa.int16(), pa.decimal128(20, 0), pa.string()] + assert stored.field("nested").type == pa.struct([("u", pa.int64())]) + + +def test_a_type_orc_cannot_write_is_measured_unavailable(tmp_path, stages, monkeypatch): + table = pa.table({"t": pa.array([1, 2], pa.time64("us"))}) + monkeypatch.setattr(build, "transform", lambda spec, tables: [(spec["slug"], table)]) + cfg = _catalog(tmp_path, "orc-uint", [SPEC]) + with operation(cfg): + assert build.run_one(SPEC, strict=False, formats=["orc"]) + assert not prepared_artifact("tiny", "orc").exists() + measured = _builds.read(cfg.data_dir)[artifact_key("tiny", "orc", 2)]["unavailable"] + assert measured["cell"] == "orc@py" and "time64" in measured["error"] diff --git a/tests/test_parquet_options.py b/tests/test_parquet_options.py new file mode 100644 index 0000000..ae5f921 --- /dev/null +++ b/tests/test_parquet_options.py @@ -0,0 +1,328 @@ +# SPDX-FileCopyrightText: 2026 Raincloud Maintainers +# SPDX-License-Identifier: Apache-2.0 +"""The Parquet write options: one set (`spec.ParquetOptions`) given the same way +to every Parquet writer, each page knob leaving the writer's own default when +unset, and a writer that cannot do what a set knob asks failing rather than +writing something else.""" +from __future__ import annotations + +import pyarrow as pa +import pyarrow.parquet as pq +import pytest + +from raincloud.pipeline.export import writer_toolchain +from raincloud.pipeline.export.exporters import ParquetExporter, UnsupportedOption +from raincloud.pipeline.export.sidecar import SidecarExporter +from raincloud.pipeline.spec import PARQUET_SETTING_VARS, ParquetOptions, parquet_options +from tests._helpers import find_sidecar, write_ipc + +KNOBS = PARQUET_SETTING_VARS +SPEC = {"slug": "pages", "write": {"compression": "zstd", "statistics": True}} +TABLE = pa.table({"x": pa.array(range(50_000), pa.int64()), "s": [f"v{i % 97}" for i in range(50_000)]}) + + +@pytest.fixture(autouse=True) +def _unset(monkeypatch): + for var in KNOBS: + monkeypatch.delenv(var, raising=False) + + +def _data_pages(path, column: int = 0) -> list[dict]: + """The data page headers of one column chunk of the first row group, read + from the file: {thrift field id: value}, as parquet.thrift numbers them.""" + meta = pq.ParquetFile(path).metadata + chunk = meta.row_group(0).column(column) + data = path.read_bytes() + start = chunk.dictionary_page_offset if chunk.has_dictionary_page else chunk.data_page_offset + pos, end, pages = start, start + chunk.total_compressed_size, [] + while pos < end: + header, pos = _thrift_struct(data, pos) + pos += header[3] # compressed_page_size + if header[1] in (0, 3): # DATA_PAGE, DATA_PAGE_V2 + pages.append(header) + return pages + + +def _thrift_struct(data: bytes, pos: int) -> tuple[dict, int]: + """A thrift compact-protocol struct at `pos`: ({field id: value}, end).""" + def varint(): + nonlocal pos + shift = value = 0 + while True: + byte = data[pos] + pos += 1 + value |= (byte & 0x7F) << shift + shift += 7 + if not byte & 0x80: + return value + + def zigzag(): + value = varint() + return (value >> 1) ^ -(value & 1) + + def read(kind, element=False): + nonlocal pos + if kind in (1, 2): + if element: + pos += 1 + return data[pos - 1] == 1 + return kind == 1 + if kind == 3: + pos += 1 + return data[pos - 1] + if kind in (4, 5, 6): + return zigzag() + if kind == 7: + pos += 8 + return None + if kind == 8: + size = varint() + pos += size + return data[pos - size:pos] + if kind in (9, 10): + head = data[pos] + pos += 1 + count = head >> 4 if head >> 4 != 15 else varint() + return [read(head & 0xF, True) for _ in range(count)] + if kind == 12: + value, pos = _thrift_struct(data, pos) + return value + raise ValueError(f"thrift type {kind} at byte {pos}") + + fields, last = {}, 0 + while data[pos]: + head = data[pos] + pos += 1 + last = last + (head >> 4) if head >> 4 else zigzag() + fields[last] = read(head & 0xF) + return fields, pos + 1 + + +def _layout(path) -> tuple[bool, int]: + """(every column chunk has a page index, data pages in the first chunk).""" + meta = pq.ParquetFile(path).metadata + chunks = [meta.row_group(g).column(c) for g in range(meta.num_row_groups) for c in range(meta.num_columns)] + indexed = {chunk.has_column_index and chunk.has_offset_index for chunk in chunks} + assert len(indexed) == 1, "some column chunks have a page index and some do not" + return indexed.pop(), len(_data_pages(path)) + + +# ---- the options ----------------------------------------------------------------------- + + +def test_unset_page_knobs_leave_each_writers_default(monkeypatch): + assert parquet_options(SPEC) == ParquetOptions() + assert parquet_options(SPEC).chosen() == {} + assert parquet_options(SPEC).env() == {"RAINCLOUD_PARQUET_COMPRESSION": "zstd", + "RAINCLOUD_PARQUET_STATISTICS": "1"} + monkeypatch.setenv("RAINCLOUD_PARQUET_PAGE_INDEX", " ") + assert parquet_options(SPEC).page_index is None + + +def test_set_knobs_are_read_once_and_passed_in_one_form(monkeypatch): + monkeypatch.setenv("RAINCLOUD_PARQUET_PAGE_INDEX", " Yes ") + monkeypatch.setenv("RAINCLOUD_PARQUET_PAGE_BYTES", "64e3") + monkeypatch.setenv("RAINCLOUD_PARQUET_PAGE_ROWS", "0") + options = parquet_options({"write": {"compression": "gzip", "statistics": True}}) + assert options == ParquetOptions("gzip", True, page_index=True, page_bytes=64_000, page_rows=(1 << 31) - 1) + assert options.env() == {"RAINCLOUD_PARQUET_COMPRESSION": "gzip", "RAINCLOUD_PARQUET_STATISTICS": "1", + "RAINCLOUD_PARQUET_PAGE_INDEX": "1", "RAINCLOUD_PARQUET_PAGE_BYTES": "64000", + "RAINCLOUD_PARQUET_PAGE_ROWS": str((1 << 31) - 1)} + assert options.chosen() == {"parquet_page_index": "1", "parquet_page_bytes": "64000", + "parquet_page_rows": str((1 << 31) - 1)} + + +@pytest.mark.parametrize("env, spec, error", [ + ({"RAINCLOUD_PARQUET_PAGE_INDEX": "maybe"}, SPEC, "is not a switch"), + ({"RAINCLOUD_PARQUET_PAGE_BYTES": "1MiB"}, SPEC, "is not a number"), + ({"RAINCLOUD_PARQUET_PAGE_ROWS": "-1"}, SPEC, "is not a number"), + ({"RAINCLOUD_PARQUET_PAGE_INDEX": "1"}, {"write": {"statistics": False}}, "asks for statistics"), + ({"RAINCLOUD_PARQUET_STATISTICS_COLUMNS": "5"}, {"write": {"statistics": False}}, "asks for statistics"), + ({"RAINCLOUD_PARQUET_PAGE_INDEX": "0", "RAINCLOUD_PARQUET_PAGE_INDEX_COLUMNS": "5"}, SPEC, + "asks for a page index"), + ({"RAINCLOUD_PARQUET_COMPRESSION_LEVEL": "-1"}, SPEC, "is not a compression level"), + ({"RAINCLOUD_PARQUET_COMPRESSION_LEVEL": "23"}, SPEC, "outside zstd's levels 1..22"), + ({"RAINCLOUD_PARQUET_COMPRESSION_LEVEL": "1"}, {"write": {"compression": "snappy"}}, + "snappy takes no compression level"), + ({}, {"write": {"compression": "lzo"}}, "is not one of"), +]) +def test_a_malformed_option_is_refused_naming_it(monkeypatch, env, spec, error): + for var, value in env.items(): + monkeypatch.setenv(var, value) + with pytest.raises(ValueError, match=error): + parquet_options(spec) + + +def test_set_page_knobs_are_part_of_every_parquet_writers_toolchain(monkeypatch): + rs = SidecarExporter("parquet@rs", "parquet", "raincloud-export-parquet-rs") + vortex = SidecarExporter("vortex@rs", "vortex", "raincloud-export-vortex-rs") + before = writer_toolchain(ParquetExporter()), writer_toolchain(rs), writer_toolchain(vortex) + monkeypatch.setenv("RAINCLOUD_PARQUET_PAGE_INDEX", "1") + assert writer_toolchain(ParquetExporter()) == {**before[0], "parquet_page_index": "1"} + assert writer_toolchain(rs) == {**before[1], "parquet_page_index": "1"} + assert writer_toolchain(vortex) == before[2] + + +def test_a_sidecar_gets_the_options_in_place_of_the_raw_environment(monkeypatch): + monkeypatch.setenv("RAINCLOUD_PARQUET_PAGE_INDEX", " on ") + monkeypatch.setenv("RAINCLOUD_PARQUET_COMPRESSION", "snappy") # the recipe's wins + env = SidecarExporter("parquet@java", "parquet", "raincloud-export-parquet-java")._child_env(SPEC) + assert (env["RAINCLOUD_PARQUET_PAGE_INDEX"], env["RAINCLOUD_PARQUET_COMPRESSION"]) == ("1", "zstd") + assert "RAINCLOUD_PARQUET_PAGE_ROWS" not in env + other = SidecarExporter("orc@rs", "orc", "raincloud-export-orc-rs")._child_env(SPEC) + assert other["RAINCLOUD_PARQUET_PAGE_INDEX"] == " on " # passed through untouched, unread + + +# ---- every writer -------------------------------------------------------------------------- + + +def _write(tmp_path, cell: str, spec: dict = SPEC): + """Write TABLE with `cell`: (round-trips, note, the file), or None when not installed.""" + canonical = tmp_path / "pages.arrow.zstd" + if not canonical.exists(): + write_ipc(canonical, TABLE, max_chunksize=8192) + dest = tmp_path / f"{cell.replace('@', '-')}.parquet" + if cell == "parquet@py": + try: + result = ParquetExporter().export(spec, canonical, dest=dest) + except UnsupportedOption as refused: # recorded unavailable when a build runs it + return False, str(refused), dest + else: + impl = cell.partition("@")[2] + if not find_sidecar(cell): + return None + result = SidecarExporter(cell, "parquet", f"raincloud-export-parquet-{impl}").export(spec, canonical, dest=dest) + return result.compliance.roundtrip, result.compliance.note, dest + + +CELLS = ("parquet@py", "parquet@rs", "parquet@java", "parquet@hardwood") + + +@pytest.mark.parametrize("cell", CELLS) +def test_a_page_index_is_written_on_request_or_refused(tmp_path, monkeypatch, cell): + monkeypatch.setenv("RAINCLOUD_PARQUET_PAGE_INDEX", "1") + written = _write(tmp_path, cell) + if written is None: + pytest.skip(f"{cell} not installed") + roundtrip, note, dest = written + if cell == "parquet@hardwood": + assert roundtrip is False and "cannot honour RAINCLOUD_PARQUET_PAGE_INDEX=1" in note, note + assert not dest.exists() + return + assert roundtrip is True, note + assert _layout(dest)[0] is True + + +@pytest.mark.parametrize("cell", CELLS) +def test_no_page_index_is_written_on_request_or_refused(tmp_path, monkeypatch, cell): + monkeypatch.setenv("RAINCLOUD_PARQUET_PAGE_INDEX", "0") + written = _write(tmp_path, cell) + if written is None: + pytest.skip(f"{cell} not installed") + roundtrip, note, dest = written + if cell == "parquet@java": + assert roundtrip is False and "cannot honour RAINCLOUD_PARQUET_PAGE_INDEX=0" in note, note + return + assert roundtrip is True, note + assert _layout(dest)[0] is False + + +@pytest.mark.parametrize("cell", CELLS) +def test_the_page_size_knob_reaches_every_writer(tmp_path, monkeypatch, cell): + unset = _write(tmp_path, cell) + if unset is None: + pytest.skip(f"{cell} not installed") + monkeypatch.setenv("RAINCLOUD_PARQUET_PAGE_BYTES", "4096") + (tmp_path / "small").mkdir() + roundtrip, note, small = _write(tmp_path / "small", cell) + assert roundtrip is True, note + # More pages, not a figure: each library measures a page its own way (arrow-rs + # sizes a dictionary-encoded page by its estimate, checked every 1,024 values). + assert _layout(small)[1] > _layout(unset[2])[1], (cell, _layout(small), _layout(unset[2])) + + +def _statistics(path) -> list[tuple[bool, bool]]: + """(chunk statistics, page index) per column of the first row group.""" + group = pq.ParquetFile(path).metadata.row_group(0) + return [(group.column(c).is_stats_set, group.column(c).has_column_index) for c in range(group.num_columns)] + + +def _codec(path) -> str: + return pq.ParquetFile(path).metadata.row_group(0).column(0).compression + + +REFUSED = "refused" +# Each setting, and what every writer does with it: REFUSED, or a check of the file. +SETTINGS = { + "statistics for the first column": ( + {"RAINCLOUD_PARQUET_STATISTICS_COLUMNS": "1"}, + {"parquet@py": lambda f: [s for s, _ in _statistics(f)] == [True, False], + "parquet@rs": lambda f: _statistics(f) == [(True, True), (False, False)], + "parquet@java": lambda f: _statistics(f) == [(True, True), (False, False)], + "parquet@hardwood": REFUSED}), + "page index for the first column": ( + {"RAINCLOUD_PARQUET_PAGE_INDEX_COLUMNS": "1"}, + {"parquet@rs": lambda f: _statistics(f) == [(True, True), (True, False)], + "parquet@py": REFUSED, "parquet@java": REFUSED, "parquet@hardwood": REFUSED}), + "no dictionaries": ( + {"RAINCLOUD_PARQUET_DICTIONARY": "0"}, + dict.fromkeys(CELLS, lambda f: not any(pq.ParquetFile(f).metadata.row_group(0).column(c).has_dictionary_page + for c in range(2)))), + "a small dictionary page": ( + # 50,000 distinct int64s do not fit 1 KiB: the column falls back to PLAIN pages. + {"RAINCLOUD_PARQUET_DICTIONARY_PAGE_BYTES": "1024"}, + {**dict.fromkeys(("parquet@py", "parquet@rs", "parquet@java"), + lambda f: any(page[5][2] == 0 for page in _data_pages(f) if 5 in page)), + "parquet@hardwood": REFUSED}), + "a zstd level": ( + {"RAINCLOUD_PARQUET_COMPRESSION_LEVEL": "19"}, + {**dict.fromkeys(("parquet@py", "parquet@rs", "parquet@java"), lambda f: _codec(f) == "ZSTD"), + "parquet@hardwood": REFUSED}), + + "page checksums": ( + {"RAINCLOUD_PARQUET_PAGE_CHECKSUMS": "1"}, + {**dict.fromkeys(("parquet@py", "parquet@java", "parquet@hardwood"), + lambda f: all(4 in page for page in _data_pages(f))), + "parquet@rs": REFUSED}), + "no page checksums": ( + {"RAINCLOUD_PARQUET_PAGE_CHECKSUMS": "0"}, + {**dict.fromkeys(("parquet@py", "parquet@rs", "parquet@java"), + lambda f: not any(4 in page for page in _data_pages(f))), + "parquet@hardwood": REFUSED}), +} + + +@pytest.mark.parametrize("cell", CELLS) +@pytest.mark.parametrize("setting", SETTINGS) +def test_each_setting_is_honoured_or_refused_by_every_writer(tmp_path, monkeypatch, setting, cell): + env, expected = SETTINGS[setting] + for var, value in env.items(): + monkeypatch.setenv(var, value) + written = _write(tmp_path, cell) + if written is None: + pytest.skip(f"{cell} not installed") + roundtrip, note, dest = written + if expected[cell] is REFUSED: + assert roundtrip is False and f"{cell} cannot honour {next(iter(env))}=" in note, note + assert not dest.exists() + return + assert roundtrip is True, note + assert expected[cell](dest), (setting, cell) + + +@pytest.mark.parametrize("cell", CELLS) +# pyarrow names the LZ4_RAW codec "LZ4" (it reads no Hadoop-framed LZ4 either). +@pytest.mark.parametrize("codec, written", [("lz4", "LZ4"), ("gzip", "GZIP"), ("snappy", "SNAPPY"), + ("none", "UNCOMPRESSED"), ("brotli", "BROTLI")]) +def test_the_recipes_codec_reaches_every_writer(tmp_path, cell, codec, written): + roundtrip_note_dest = _write(tmp_path, cell, {"slug": "pages", "write": {"compression": codec, "statistics": True}}) + if roundtrip_note_dest is None: + pytest.skip(f"{cell} not installed") + roundtrip, note, dest = roundtrip_note_dest + if (cell, codec) == ("parquet@java", "brotli"): + assert roundtrip is False and "parquet@java cannot honour RAINCLOUD_PARQUET_COMPRESSION=brotli" in note, note + return + assert roundtrip is True, note + assert _codec(dest) == written + # Another library decodes it: pyarrow reads every writer's file in every codec. + assert pq.read_table(dest).select(["x", "s"]).to_pydict() == TABLE.to_pydict() diff --git a/tests/test_parquet_py_row_group_cuts.py b/tests/test_parquet_py_row_group_cuts.py index 08a0327..ad1fda2 100644 --- a/tests/test_parquet_py_row_group_cuts.py +++ b/tests/test_parquet_py_row_group_cuts.py @@ -6,6 +6,7 @@ import pyarrow.parquet as pq from raincloud.pipeline.export import exporters as exporters_mod +from raincloud.pipeline.spec import ParquetOptions from tests._helpers import write_ipc @@ -26,7 +27,7 @@ def _cuts(canonical, row_group, byte_target): def test_parquet_py_cuts_groups_at_exactly_the_planned_row_across_batches(tmp_path): canonical = _canonical(tmp_path, [_ints(i, min(i + 700, 4500)) for i in range(0, 4500, 700)]) out = tmp_path / "out.parquet" - assert exporters_mod._write_parquet(canonical, out, 1000, 1 << 40, compression="zstd", stats=True) is False + assert exporters_mod._write_parquet(canonical, out, 1000, 1 << 40, options=ParquetOptions()) is False meta = pq.ParquetFile(out).metadata assert [meta.row_group(i).num_rows for i in range(meta.num_row_groups)] == [1000] * 4 + [500] assert pq.read_table(out).column("x").to_pylist() == list(range(4500)) @@ -56,6 +57,6 @@ def test_empty_batches_never_make_an_empty_group_beside_real_ones(tmp_path): def test_the_probe_encodes_exactly_the_first_planned_group(tmp_path): canonical = _canonical(tmp_path, [_ints(i, i + 700) for i in range(0, 2800, 700)]) rows, encoded = exporters_mod._probe_encoded(canonical, 1000, tmp_path / "probe.parquet", - compression="zstd", stats=True) + options=ParquetOptions()) assert rows == 1000 and encoded > 0 assert not (tmp_path / "probe.parquet").exists() diff --git a/tests/test_pipeline_contracts.py b/tests/test_pipeline_contracts.py index 8f61263..cb276f6 100644 --- a/tests/test_pipeline_contracts.py +++ b/tests/test_pipeline_contracts.py @@ -24,7 +24,7 @@ from raincloud.pipeline.export import exporters as exporters_mod from raincloud.pipeline.export.__main__ import main as export_main from raincloud.pipeline.lifecycle import BuildOutputs, build_outputs, operation_lock -from raincloud.pipeline.spec import prepared_arrow, prepared_parquet, prepared_vortex +from raincloud.pipeline.spec import ParquetOptions, prepared_arrow, prepared_parquet, prepared_vortex from raincloud.pipeline.validate import validate TINY = {"slug": "tiny", "export": {"formats": ["parquet", "vortex"]}} @@ -33,6 +33,11 @@ TABLE = pa.table({"x": [1, 2], "url": ["https://example.test/a", None]}) +def _artifacts(tmp_path, paths): + """A `prepared_artifact` stand-in: `paths[fmt]`, else a file that never exists.""" + return lambda slug, fmt, manifest=None: paths.get(fmt, tmp_path / f"missing.{fmt}") + + def _catalog(tmp_path, name, datasets, slugs=None): """A config selecting a catalog of `datasets`, over one shared store.""" bundle = make_bundle(encode({"schema_version": 2, "datasets": datasets}), @@ -60,7 +65,9 @@ def stages(monkeypatch): @pytest.fixture def store(tmp_path, stages): - cfg = _catalog(tmp_path, "contracts", [TINY, HYDRATED, {"slug": "other", "export": {"formats": ["parquet"]}}]) + # `other` has a recipe of its own (offered formats are not part of one). + other = {"slug": "other", "export": {"formats": ["parquet"]}, "expect": {"rows": 7}} + cfg = _catalog(tmp_path, "contracts", [TINY, HYDRATED, other]) with operation(cfg): assert build.run_one(TINY, strict=False) yield cfg @@ -464,7 +471,7 @@ def _docs_env(tmp_path, monkeypatch, version): monkeypatch.setattr(docs, "load_manifest", lambda: {"schema_version": version, "datasets": [{"slug": "kept"}]}) monkeypatch.setattr(docs, "prepared_parquet", lambda slug: tmp_path / "missing.parquet") monkeypatch.setattr(docs, "prepared_vortex", lambda slug: tmp_path / "missing.vortex") - monkeypatch.setattr(docs, "prepared_arrow", lambda slug: tmp_path / "missing.arrow.zstd") + monkeypatch.setattr(docs, "prepared_artifact", _artifacts(tmp_path, {"parquet": tmp_path / "missing.parquet", "vortex": tmp_path / "missing.vortex", "arrow": tmp_path / "missing.arrow.zstd"})) monkeypatch.setenv("RAINCLOUD_HOME", str(tmp_path / "home")) (tmp_path / "docs" / f"v{version}").mkdir(parents=True) @@ -507,14 +514,14 @@ def test_row_group_second_pass_rules(tmp_path): canonical_path = _canonical_file(tmp_path, table, 100) out = tmp_path / "out.parquet" # Rows closed every group: not byte-bound. - assert exporters_mod._write_parquet(canonical_path, out, 1000, 1 << 40, compression="zstd", stats=True) is False + assert exporters_mod._write_parquet(canonical_path, out, 1000, 1 << 40, options=ParquetOptions()) is False assert pq.ParquetFile(out).metadata.num_row_groups == 4 # The decoded-byte ceiling closed the groups first. - assert exporters_mod._write_parquet(canonical_path, out, 4000, 1000, compression="zstd", stats=True) is True + assert exporters_mod._write_parquet(canonical_path, out, 4000, 1000, options=ParquetOptions()) is True one = tmp_path / "one.parquet" pq.write_table(table, one) assert exporters_mod._corrected_rows(one, 4000, 10, 1 << 30) is None # < 2 groups - exporters_mod._write_parquet(canonical_path, out, 1000, 1 << 40, compression="zstd", stats=True) + exporters_mod._write_parquet(canonical_path, out, 1000, 1 << 40, options=ParquetOptions()) median = sorted(pq.ParquetFile(out).metadata.row_group(i).total_byte_size for i in range(4))[2] assert exporters_mod._corrected_rows(out, 1000, median, 1 << 30) is None # on target assert exporters_mod._corrected_rows(out, 1000, median * 2, 1 << 30) == pytest.approx(2000, rel=0.02) @@ -526,10 +533,10 @@ def test_probe_converges_on_a_uniform_table(tmp_path, monkeypatch): canonical_path = _canonical_file(tmp_path, table, 10_000) probe = tmp_path / "probe.parquet" monkeypatch.setenv("RAINCLOUD_ROW_GROUP_PROBE_ROWS", "10000") - rows, encoded = exporters_mod._probe_encoded(canonical_path, 200_000, probe, compression="zstd", stats=True) + rows, encoded = exporters_mod._probe_encoded(canonical_path, 200_000, probe, options=ParquetOptions()) monkeypatch.setenv("RAINCLOUD_ROW_GROUP_TARGET_ENCODED_BYTES", str(encoded // 4)) - want = exporters_mod._rows_for_encoded_target(canonical_path, probe, compression="zstd", - stats=True, row_cap=1 << 30) + want = exporters_mod._rows_for_encoded_target(canonical_path, probe, options=ParquetOptions(), + row_cap=1 << 30) assert 0.5 * rows / 4 < want < 2 * rows / 4 assert not probe.exists() @@ -795,8 +802,8 @@ def test_a_byte_heavy_region_does_not_cancel_the_second_pass(tmp_path): canonical_path = _canonical_file(tmp_path, table, 100) out = tmp_path / "out.parquet" # The ceiling closes the heavy region's groups; the light ones close on rows. - assert exporters_mod._write_parquet(canonical_path, out, 1000, 100_000, compression="zstd", stats=True) is False - assert exporters_mod._write_parquet(canonical_path, out, 1000, 1000, compression="zstd", stats=True) is True + assert exporters_mod._write_parquet(canonical_path, out, 1000, 100_000, options=ParquetOptions()) is False + assert exporters_mod._write_parquet(canonical_path, out, 1000, 1000, options=ParquetOptions()) is True def test_a_nested_dictionary_larger_than_the_target_does_not_split_to_single_rows(): @@ -827,7 +834,7 @@ def test_snapshot_regen_describes_a_multi_batch_ipc_canonical(tmp_path, monkeypa with pa.ipc.new_file(arrow, table.schema, options=pa.ipc.IpcWriteOptions(compression="zstd")) as writer: for batch in table.to_batches(3): writer.write_batch(batch) - monkeypatch.setattr(docs, "prepared_arrow", lambda slug: arrow) + monkeypatch.setattr(docs, "prepared_artifact", _artifacts(tmp_path, {"arrow": arrow})) dest = tmp_path / "out.json" docs.generate_snapshot(destination=dest) entry = json.loads(dest.read_text())["slugs"]["kept"] diff --git a/tests/test_pipeline_real.py b/tests/test_pipeline_real.py index 0bb8cd5..4095680 100644 --- a/tests/test_pipeline_real.py +++ b/tests/test_pipeline_real.py @@ -32,6 +32,7 @@ def test_real_build_via_load(tmp_path, slug, expected_rows): # Assert the actual public read resolves the artifact produced by the child # builder inside this store; ambient machine configuration cannot redirect it. assert ds.path().is_relative_to(cfg.data_dir) - parquet = next(a["key"] for a in ds.artifacts if a["format"] == "parquet" and a["writer"] == "py") - assert (cfg.data_dir / parquet).is_file() + # The build wrote the format the load asked for: auto, on an install that builds + # only Vortex (the default), with every extra installed. + assert ds.format == "vortex" and ds.path().is_file() assert not cfg.cache_dir.exists() diff --git a/tests/test_write_settings.py b/tests/test_write_settings.py new file mode 100644 index 0000000..7452d05 --- /dev/null +++ b/tests/test_write_settings.py @@ -0,0 +1,174 @@ +# SPDX-FileCopyrightText: 2026 Raincloud Maintainers +# SPDX-License-Identifier: Apache-2.0 +"""The ORC, Avro and Vortex write settings (`spec.FORMAT_SETTINGS`): given the +same way to every writer of the format, unset leaving what raincloud has always +written, and a writer whose library cannot honour a set one refusing it.""" +from __future__ import annotations + +import numpy as np +import pyarrow as pa +import pytest + +from raincloud.pipeline.export import writer_toolchain +from raincloud.pipeline.export.exporters import OrcExporter, UnsupportedOption, VortexExporter +from raincloud.pipeline.export.sidecar import SidecarExporter +from raincloud.pipeline.spec import FORMAT_SETTINGS, chosen_settings, sidecar_settings, write_settings +from tests._helpers import find_sidecar, write_ipc + +# `f` does not compress: ORC C++ measures a stripe once encoded and compressed. +TABLE = pa.table({"x": pa.array(range(40_000), pa.int64()), "s": [f"v{i % 97}" for i in range(40_000)], + "f": np.random.default_rng(0).random(40_000)}) +SPEC = {"slug": "settings"} + + +@pytest.fixture(autouse=True) +def _unset(monkeypatch): + for settings in FORMAT_SETTINGS.values(): + for _, var, _ in settings: + monkeypatch.delenv(var, raising=False) + + +# ---- the settings ------------------------------------------------------------------------ + + +def test_unset_is_what_raincloud_has_always_written(): + for fmt in FORMAT_SETTINGS: + assert set(write_settings(fmt).values()) == {None} + assert chosen_settings(fmt) == {} and sidecar_settings(fmt, SPEC) == {} + + +def test_set_settings_are_passed_in_one_form(monkeypatch): + monkeypatch.setenv("RAINCLOUD_AVRO_COMPRESSION", " Deflate ") + monkeypatch.setenv("RAINCLOUD_AVRO_COMPRESSION_LEVEL", "9") + monkeypatch.setenv("RAINCLOUD_VORTEX_COMPACT", "yes") + monkeypatch.setenv("RAINCLOUD_ORC_STRIPE_BYTES", "64e3") + assert sidecar_settings("avro", SPEC) == {"RAINCLOUD_AVRO_COMPRESSION": "deflate", + "RAINCLOUD_AVRO_COMPRESSION_LEVEL": "9"} + assert sidecar_settings("vortex", SPEC) == {"RAINCLOUD_VORTEX_COMPACT": "1"} + assert chosen_settings("orc") == {"orc_stripe_bytes": "64000"} + rs = SidecarExporter("avro@rs", "avro", "raincloud-export-avro-rs") + assert writer_toolchain(rs)["avro_compression"] == "deflate" + assert writer_toolchain(VortexExporter())["vortex_compact"] == "1" + assert rs._child_env(SPEC)["RAINCLOUD_AVRO_COMPRESSION"] == "deflate" + + +@pytest.mark.parametrize("var, value, error", [ + ("RAINCLOUD_ORC_COMPRESSION", "brotli", "is not one of zstd, snappy, zlib, lz4, none"), + ("RAINCLOUD_ORC_COMPRESSION_STRATEGY", "fast", "is not one of speed, compression"), + ("RAINCLOUD_AVRO_COMPRESSION_LEVEL", "23", "outside zstd's levels 1..22"), + ("RAINCLOUD_VORTEX_COMPACT", "maybe", "is not a switch"), + ("RAINCLOUD_VORTEX_ROW_BLOCK_ROWS", "8k", "is not a number"), +]) +def test_a_malformed_setting_is_refused_naming_it(monkeypatch, var, value, error): + monkeypatch.setenv(var, value) + with pytest.raises(ValueError, match=error): + write_settings(var.split("_")[1].lower()) + + +def test_a_level_for_a_codec_without_one_is_refused(monkeypatch): + monkeypatch.setenv("RAINCLOUD_AVRO_COMPRESSION", "snappy") + monkeypatch.setenv("RAINCLOUD_AVRO_COMPRESSION_LEVEL", "3") + with pytest.raises(ValueError, match="snappy takes no compression level"): + write_settings("avro") + + +# ---- every writer -------------------------------------------------------------------------- + +IN_PROCESS = {"orc@py": OrcExporter, "vortex@py": VortexExporter} + + +def _write(tmp_path, cell: str): + """Write TABLE with `cell`: (round-trips, note, the file), or None when not installed.""" + fmt, _, impl = cell.partition("@") + canonical = tmp_path / "settings.arrow.zstd" + if not canonical.exists(): + write_ipc(canonical, TABLE, max_chunksize=8192) + dest = tmp_path / f"{fmt}-{impl}.{fmt}" + if cell in IN_PROCESS: + try: + result = IN_PROCESS[cell]().export(SPEC, canonical, dest=dest) + except UnsupportedOption as refused: # recorded unavailable when a build runs it + return False, str(refused), dest + else: + if not find_sidecar(cell): + return None + result = SidecarExporter(cell, fmt, f"raincloud-export-{fmt}-{impl}").export(SPEC, canonical, dest=dest) + return result.compliance.roundtrip, result.compliance.note, dest + + +def _orc(path): + import pyarrow.orc as orc + return orc.ORCFile(str(path)) + + +def _avro_codec(path) -> str: + """The `avro.codec` an Avro object container file's header names.""" + data = path.read_bytes() + at = data.index(b"avro.codec") + len(b"avro.codec") + size = data[at] >> 1 # a short zigzag varint length + return data[at + 1:at + 1 + size].decode() + + +def _avro_blocks(path) -> int: + """Data blocks in an Avro file: each ends with the sync marker both lanes + write, which the header carries once too.""" + return path.read_bytes().count(b"raincloud-avro01") - 1 + + +REFUSED = "refused" +SETTINGS = { + "an ORC codec": ( + {"RAINCLOUD_ORC_COMPRESSION": "snappy"}, + dict.fromkeys(("orc@py", "orc@rs"), lambda f: _orc(f).compression == "SNAPPY")), + "no ORC compression": ( + {"RAINCLOUD_ORC_COMPRESSION": "none"}, + dict.fromkeys(("orc@py", "orc@rs"), lambda f: _orc(f).compression == "UNCOMPRESSED")), + "an ORC compression strategy": ( + {"RAINCLOUD_ORC_COMPRESSION_STRATEGY": "compression"}, + {"orc@py": lambda f: _orc(f).compression == "ZSTD", "orc@rs": REFUSED}), + "small ORC stripes": ( + {"RAINCLOUD_ORC_STRIPE_BYTES": "65536"}, + dict.fromkeys(("orc@py", "orc@rs"), lambda f: _orc(f).nstripes > 1)), + "an ORC compression block size": ( + # ORC C++ takes only a multiple of its 64 KiB memory block; another size fails there. + {"RAINCLOUD_ORC_COMPRESSION_BLOCK_BYTES": "131072"}, + dict.fromkeys(("orc@py", "orc@rs"), lambda f: _orc(f).compression_size == 131072)), + **{f"Avro {codec}": ( + {"RAINCLOUD_AVRO_COMPRESSION": codec}, + dict.fromkeys(("avro@rs", "avro@java"), + lambda f, name={"zstd": "zstandard", "none": "null"}.get(codec, codec): _avro_codec(f) == name)) + for codec in ("zstd", "deflate", "snappy", "bzip2", "xz", "none")}, + "an Avro level": ( + {"RAINCLOUD_AVRO_COMPRESSION": "deflate", "RAINCLOUD_AVRO_COMPRESSION_LEVEL": "9"}, + {"avro@java": lambda f: _avro_codec(f) == "deflate", "avro@rs": REFUSED}), + "small Avro blocks": ( + {"RAINCLOUD_AVRO_BLOCK_BYTES": "4096"}, + {"avro@java": lambda f: _avro_blocks(f) > 10, "avro@rs": REFUSED}), + "compact Vortex": ( + {"RAINCLOUD_VORTEX_COMPACT": "1"}, + {"vortex@py": lambda f: True, "vortex@rs": lambda f: True, "vortex@jni": REFUSED}), + "a Vortex row block size": ( + {"RAINCLOUD_VORTEX_ROW_BLOCK_ROWS": "1024"}, + {"vortex@rs": lambda f: True, "vortex@py": REFUSED, "vortex@jni": REFUSED}), + "a Vortex data block size": ( + {"RAINCLOUD_VORTEX_DATA_BLOCK_BYTES": "65536"}, + {"vortex@rs": lambda f: True, "vortex@py": REFUSED, "vortex@jni": REFUSED}), +} +CASES = [(setting, cell) for setting, (_, cells) in SETTINGS.items() for cell in cells] + + +@pytest.mark.parametrize("setting, cell", CASES) +def test_each_setting_is_honoured_or_refused_by_every_writer(tmp_path, monkeypatch, setting, cell): + env, expected = SETTINGS[setting] + for var, value in env.items(): + monkeypatch.setenv(var, value) + written = _write(tmp_path, cell) + if written is None: + pytest.skip(f"{cell} not installed") + roundtrip, note, dest = written + if expected[cell] is REFUSED: + assert roundtrip is False and f"{cell} cannot honour RAINCLOUD_" in note, note + assert not dest.exists() + return + assert roundtrip is True, note + assert expected[cell](dest), (setting, cell)