diff --git a/.github/workflows/deploy-cloudflare-pages.yml b/.github/workflows/deploy-cloudflare-pages.yml new file mode 100644 index 0000000..e75bb57 --- /dev/null +++ b/.github/workflows/deploy-cloudflare-pages.yml @@ -0,0 +1,90 @@ +name: Deploy to Cloudflare Pages + +# Single workflow handles both Cloudflare Pages environments: +# - push to `main` → wrangler --branch main → production (petrodb.ocortez.com) +# - push to `stage` → wrangler --branch stage → preview deployment +# The Cloudflare project is the same for both; the `--branch` arg matches the +# GitHub ref so the two GH branches map onto the project's two deployments. +# +# Trigger: any push to main/stage that touches deployable content — committed +# datasets under `parquet/`, the 3W pipeline sources (whose outputs are +# gitignored and rebuilt in CI), or this workflow file. `workflow_dispatch` +# allows manual redeploys without an empty commit. +# +# 3W pipeline re-run is gated by a cache keyed on the 3W source hashes plus +# the resolved per-branch BASE_URL (so the prod/stage catalogs never share a +# cache entry): if neither has changed, the previously-generated 3W tree is +# restored and the deploy goes out without re-running the pipeline. + +on: + push: + branches: [main, stage] + paths: + - 'parquet/**' + - 'scripts/transform/petrobras_3w/**' + - 'scripts/export/petrobras_3w/**' + - 'scripts/run_pipeline/petrobras_3w/**' + - '.github/workflows/deploy-cloudflare-pages.yml' + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: deploy-cloudflare-pages-${{ github.ref_name }} + cancel-in-progress: false + +jobs: + deploy: + if: github.ref_name == 'main' || github.ref_name == 'stage' + runs-on: ubuntu-latest + timeout-minutes: 30 + env: + # Baked into `instances.parquet#source_url` so consumers can round-trip + # the catalog to Observations URLs. Per-branch because main and stage + # serve from different Cloudflare custom domains; the catalog must + # reference the host where its own bytes are reachable. + BASE_URL: ${{ github.ref_name == 'main' && vars.BASE_URL_MAIN || vars.BASE_URL_STAGE }} + steps: + - name: Checkout repository + uses: actions/checkout@v4 + + - name: Restore Petrobras 3W generated tree (if sources unchanged) + id: cache-3w + uses: actions/cache@v4 + with: + path: parquet/petrobras_3w + key: >- + petrobras-3w-${{ hashFiles( + 'scripts/transform/petrobras_3w/**', + 'scripts/export/petrobras_3w/**', + 'scripts/run_pipeline/petrobras_3w/**' + ) }}-${{ env.BASE_URL }} + + - name: Install uv + if: steps.cache-3w.outputs.cache-hit != 'true' + uses: astral-sh/setup-uv@v3 + with: + enable-cache: true + + - name: Set up Python + if: steps.cache-3w.outputs.cache-hit != 'true' + run: uv python install 3.12 + + - name: Install dependencies + if: steps.cache-3w.outputs.cache-hit != 'true' + run: uv sync --frozen + + - name: Build Petrobras 3W parquet tree + if: steps.cache-3w.outputs.cache-hit != 'true' + run: uv run python -m scripts.run_pipeline.petrobras_3w + + - name: Deploy to Cloudflare Pages + uses: cloudflare/wrangler-action@v3 + with: + accountId: ${{ secrets.CLOUDFLARE_ACCOUNT_ID }} + apiToken: ${{ secrets.CLOUDFLARE_API_TOKEN }} + command: >- + pages deploy parquet/ + --project-name=${{ vars.CLOUDFLARE_PAGES_PROJECT }} + --branch=${{ github.ref_name }} diff --git a/.github/workflows/deploy-pages.yml b/.github/workflows/deploy-pages.yml deleted file mode 100644 index 3e267f7..0000000 --- a/.github/workflows/deploy-pages.yml +++ /dev/null @@ -1,43 +0,0 @@ -name: Deploy Parquet to Pages - -on: - push: - branches: [ main ] - paths: - - 'parquet/**' - workflow_dispatch: - -permissions: - contents: read - pages: write - id-token: write - -concurrency: - group: "pages" - cancel-in-progress: false - -jobs: - deploy: - environment: - name: github-pages - url: ${{ steps.deployment.outputs.page_url }} - runs-on: ubuntu-latest - steps: - - name: Checkout repository - uses: actions/checkout@v4 - - - name: Ensure CNAME file exists - run: echo "volve-db.oscarcortez.me" > parquet/CNAME - - - name: Setup Pages - uses: actions/configure-pages@v4 - - - name: Upload artifact - uses: actions/upload-pages-artifact@v3 - with: - # Only upload the parquet directory - path: 'parquet/' - - - name: Deploy to GitHub Pages - id: deployment - uses: actions/deploy-pages@v4 diff --git a/.gitignore b/.gitignore index b68e24c..5c5bbca 100644 --- a/.gitignore +++ b/.gitignore @@ -210,6 +210,11 @@ __marimo__/ .duckdb/ database/argentina.duckdb database/argentina.duckdb.wal +database/petrobras_3w.duckdb +database/petrobras_3w.duckdb.wal + +# Petrobras 3W upstream shallow clone (see ADR-0002 for the pin policy) +data/petrobras_3w/ # Tool cache/config directories (from container) .config/ @@ -220,3 +225,9 @@ database/argentina.duckdb.wal .DS_Store .gemini/ data/Argentina + +# Petrobras 3W published tree — regenerated in CI on every deploy +# (see .github/workflows/deploy-cloudflare-pages.yml). The pipeline output +# (catalogs, hive-partitioned Observations, manifest, schema docs, README, +# license mirror) is never committed; other datasets remain tracked. +parquet/petrobras_3w/ diff --git a/CONTEXT.md b/CONTEXT.md index 0af8c66..d4e800a 100644 --- a/CONTEXT.md +++ b/CONTEXT.md @@ -16,14 +16,43 @@ The human-readable code for a well (e.g. `YPF.BLO.x-8`). Treated as a label, not **Formprod** (producing formation): The geological formation a given `idpozo` produces from. Static per `idpozo` by construction (encoded into the ID itself), so it lives in the master table, never in the time-series. +### Petrobras 3W dataset + +**Well** (3W sense, `well_id`): +A physical offshore producing well operated by Petrobras, identified by an integer anonymized by upstream (e.g. `1`, `19`). 1-Hz sensor histories from the well are sliced into many **Instances**. The dataset covers ~40 distinct real wells. _Not the same as Argentina's Well_: there is no `formprod` split here, and no human-readable label. + +**Instance** (`instance_id`): +A single contiguous 1-Hz time-window of sensor data drawn from one source, framed around at most one labeled anomaly event. Identified by the upstream source filename without extension (e.g. `WELL-00019_20120601165020`). An instance is the natural unit of an experiment: one file = one window to train, evaluate, or visualize. Length varies from ~6 hours (~21k rows) to ~3 days (~243k rows). + +**Well kind** (`well_kind`): +The provenance of an Instance. One of `{real, simulated, drawn}` — `real` for instances sourced from a physical Well (~40 wells across the corpus), `simulated` for synthetic instances generated by upstream tooling, `drawn` for hand-drawn series. Simulated and drawn instances have **no** `well_id`. + +**Event class** (`event_class`): +An integer code in `0..9` identifying the operational regime an Instance is framed around. `0 = NORMAL`, `1..9 = anomaly categories` (Abrupt Increase of BSW, Spurious Closure of DHSV, Severe Slugging, Flow Instability, Rapid Productivity Loss, Quick Restriction in PCK, Scaling in PCK, Hydrate in Production Line, Hydrate in Service Line). The canonical names come from `dataset.ini` upstream. _Not the same as Argentina's "event"_, which means an operational state transition row. + +**Transient label**: +A per-observation `class` value equal to `event_class + 100`, marking the developing phase of an anomaly before it reaches steady state. Codes seen in the data: `101, 102, 105, 106, 107, 108, 109`. Defined only for the seven anomalies that upstream marks `TRANSIENT=True`. Events `3` (Severe Slugging) and `4` (Flow Instability) are `TRANSIENT=False` — no transient codes exist for them. + +**Observation**: +One row of an `observations` table — a single 1-Hz timestamped sample with 27 sensor floats, a `class` label (per-observation regime: NULL, 0, anomaly class, or transient code), and a `state` label (well operational status). + +**Source file**: +The upstream parquet file backing a single Instance. Filenames encode provenance: `WELL-_.parquet` for real, `SIMULATED_.parquet`, `DRAWN_.parquet`. In our published catalog the source filename is preserved as a column so consumers can cross-reference with upstream. + ## Relationships - A **Well** (`idpozo`) is the unit of identity for both the master table and the production time-series. - **Formprod** is a static attribute of a **Well** — recorded once in the master, never repeated in monthly rows. +- A **Well** (3W sense, `well_id`) is the parent of zero-or-more **Instances** (only when `well_kind = real`). +- An **Instance** is the parent of many **Observations** (1-Hz rows). +- An **Event class** is referenced by an **Instance** (the regime the window is framed around) and by an **Observation** (the per-row label, possibly via its transient offset). +- Simulated and drawn **Instances** have no parent **Well** — their `well_id` is NULL by design. ## Flagged ambiguities -- "well" was used to refer to both the physical wellbore and the producing-formation-specific record — resolved: in this dataset a **Well** = `idpozo` = wellbore × formprod. Use "physical wellbore" if the bore-only concept is needed. +- "well" was used to refer to both the physical wellbore and the producing-formation-specific record — resolved: in the Argentina dataset a **Well** = `idpozo` = wellbore × formprod. Use "physical wellbore" if the bore-only concept is needed. +- "Well" is overloaded across datasets — resolved by scope: Argentina **Well** = `idpozo`; Petrobras 3W **Well** = anonymized offshore wellbore `well_id`. Always qualify with the dataset name when the context is mixed. +- "Event" is overloaded — resolved: in the Argentina dataset an **Event** is a row in `well_events.parquet` marking an operational-state transition; in the Petrobras 3W dataset an **Event class** is an anomaly category referenced by an Instance and by every Observation. Different concepts, different tables. ## Argentina dataset — column buckets @@ -100,3 +129,70 @@ The export step asserts the following before writing Parquets; failure aborts pu 4. Geometry in `wells.parquet` is parseable WKB for every row that carries a `geom`. 5. Year-partition count is 22 and the sum of partition row counts equals the staged-source row count. 6. Soft-warn if any year-partition Parquet exceeds 50 MB (Cloudflare cache headroom). + +## Petrobras 3W dataset — tables + +A four-table relational shape, derived from upstream's per-instance Parquet files plus `dataset.ini`. Layouts and column names below are settled in the grill; pipeline implementation details (e.g. exact column-name policy for hyphenated source columns) are still open. + +**Static master** (`wells.parquet`, one row per real Well, ~40 rows): +`well_id`, `n_instances` (count of `instances` rows), `first_ts`, `last_ts`, `n_observations` (sum across instances). Upstream anonymizes physical well attributes (no basin, field, depth, or location is published), so this table is mostly an identity-plus-statistics master. Only `well_kind = real` instances have a parent row here. + +**Lookup** (`event_types.parquet`, exactly 10 rows): +`event_class` (PK, 0..9), `name` (canonical PascalCase from `dataset.ini`, e.g. `HYDRATE_IN_PRODUCTION_LINE`), `description` (human-readable, e.g. `Hydrate in Production Line`), `has_transient` (boolean, false for `{0, 3, 4}`), `transient_code` (`event_class + 100` when `has_transient`, NULL otherwise), `has_normal_prefix` (boolean, true for events that include a class=0 precursor in the data — correlates with `has_transient`). + +**Instance catalog** (`instances.parquet`, one row per `(instance_id, event_class)` pair): +`instance_id` (source filename without extension; **not** unique on its own — upstream re-publishes each synthetic `SIMULATED_*` / `DRAWN_*` series under several event classes, so the composite `(instance_id, event_class)` is the primary key; the hive partition path on `observations/` encodes the same identity), `well_kind` (enum: `real | simulated | drawn`), `well_id` (FK to `wells.parquet`, NULL when `well_kind != real`), `event_class` (FK to `event_types.parquet`), `start_ts`, `end_ts`, `duration_s` (derived), `n_rows`, `n_rows_warmup_null` (rows where `class IS NULL`), `n_rows_normal` (rows where `class = 0` AND `event_class <> 0` — i.e. the NORMAL precursor before an anomaly; event 0's `class = 0` rows are its labelled regime and roll into `n_rows_steady` instead so the four buckets always partition `n_rows`), `n_rows_transient` (rows where `class = event_class + 100`; NULL when `has_transient = false`), `n_rows_steady` (rows where `class = event_class`), `source_file` (upstream filename for cross-reference), `source_url` (URL to the published Observations parquet). The four `n_rows_*` columns let corpus-wide balance and labeled-mass queries run purely against this catalog without scanning Observations. + +**Observations time-series** (`observations/event_class=N/.parquet`, ~2,228 files): +Hive-partitioned by `event_class` only. Each file is a single Instance's 1-Hz rows. Columns: the 27 source sensor columns + `class` (per-observation label, may include the +100 transient codes) + `state` (well operational status) + `timestamp` + three added constant columns: `instance_id`, `well_id`, `well_kind`. `event_class` is provided by the hive partition, not stored in the file body. + +## Petrobras 3W dataset — operating principles + +- **Source column names preserved verbatim**, including hyphens (`P-PDG`, `ABER-CKGL`, `ESTADO-SDV-GL`). Consumers must double-quote these identifiers in SQL. Same fidelity principle applied to Argentina, applied here to hyphenated forms instead of Spanish forms. +- **Source fidelity over smoothing**: the `class` column's NULL warmup prefix (~1 hour at the start of every real-Well instance) is preserved; the `+100` transient offset is preserved as a raw code (lookup table explains the convention); per-instance file boundaries match upstream's per-instance file boundaries. No row-level reinterpretation. +- **Per-row provenance**: `instance_id`, `well_id`, `well_kind` are duplicated into each Observations file as constant columns (RLE-encoded, negligible storage). The dataset stays self-describing without filename-parsing tribal knowledge. +- **TRANSIENT semantics**: instances of an event with `has_transient = true` carry a `NORMAL → TRANSIENT → STEADY` arc in their `class` column. Instances of an event with `has_transient = false` (events 3 and 4) carry **only** the steady class — no `NORMAL` precursor either. Consumers training early-detection models should filter on `has_transient`. +- **3W toolkit artefacts are excluded**: per-event detector hyperparameters (`WINDOW`, `STEP`) and fold-config knobs (`EXTRA_INSTANCES_TRAINING`) live in upstream's `dataset.ini` but belong to the *3W toolkit*, not to the dataset itself. Petrodb publishes the dataset; toolkit and folds remain the consumer's responsibility. +- **Pure DuckDB SQL transform**: pipeline is read-with-`filename=true` → parse filename into catalog columns → `COPY ... PARTITION_BY (event_class)`. Polars permitted only where SQL becomes unreadable; pandas is not used. Same rule as the Argentina pipeline. +- **Upstream pinned to a git tag that ships a specific dataset version.** Currently pinned to git tag `v.1.70.0`, which ships upstream dataset version `2.0.0`. Refreshes are event-driven on new upstream tags, not calendar-driven. Past upstream releases have retroactively changed sensor values inside the same `instance_id`, so tracking `main` would silently mutate already-published parquet bytes. Upstream uses two version namespaces — git tags (`v.1.NN.0`) and dataset semver in `dataset/README.md` (`1.0.0`, `1.1.0`, `1.1.1`, `2.0.0`, …); we pin the git tag because that is what `git clone --branch` accepts, and we record the dataset version because that is what identifies the data shape. See [ADR-0002](docs/adr/0002-petrobras-3w-pin-upstream-release-tag.md). + +## Petrobras 3W dataset — output layout + +``` +parquet/petrobras_3w/ +├── wells.parquet +├── event_types.parquet +├── instances.parquet +├── observations/ +│ ├── event_class=0/.parquet # 594 files +│ ├── event_class=1/.parquet # 128 files +│ ├── … +│ └── event_class=9/.parquet # 207 files +├── observations/_files.json # manifest of all Observations URLs for httpfs consumers +├── schema.md / schema.json / schema.sql +├── README.md +└── LICENSE-3W-DATA.md # CC BY 4.0 mirror with upstream attribution +``` + +Rationale: 2,228 instances × ~0.4–1 MB each keep every file well under Cloudflare's edge-cache size target. Hive partitioning by `event_class` prunes the dominant ML query pattern ("train on class N"). One-file-per-instance preserves the upstream conceptual unit and makes single-instance fetch a direct GET (the ML training-loop pattern). Catalog tables answer corpus-wide questions without touching Observations. + +## Petrobras 3W dataset — pre-publish validation + +The export step asserts the following before writing Parquets; failure aborts publish. + +1. `instances` has unique `(instance_id, event_class)`. `instance_id` alone is not unique because upstream re-publishes each synthetic `SIMULATED_*` / `DRAWN_*` series under several event classes (~225 such pairs at the pinned tag). +2. Every `instance_id` referenced by an Observations file exists in `instances.parquet`. +3. Every non-NULL `instances.well_id` exists in `wells.parquet` (FK integrity); NULL only when `well_kind != real`. +4. Every `instances.event_class` exists in `event_types.parquet` (FK integrity). +5. Within each Observations file: only one distinct `(instance_id, well_id, well_kind)` triple; `class ∈ {NULL, 0, event_class, transient_code(event_class)}` for the file's `event_class`; row count equals `instances.n_rows`; `timestamp` is strictly monotonic at exactly 1-second cadence. +6. For Instances with `has_transient = false` event class (events 3, 4): no `class` value ≥ 100 appears; no `class = 0` appears. +7. Real-Well coverage equals the count derived from upstream filename prefixes at the pinned git tag (currently 40 distinct `WELL-NNNNN` prefixes at `v.1.70.0` / dataset version 2.0.0 — fail-loud if our derived `wells.parquet` rowcount disagrees, to catch upstream version drift). Upstream's `dataset/README.md` states "42 real wells covered", but only 40 IDs (`00001..00016`, `00019..00042`) actually appear in instance filenames; IDs `00017` and `00018` are absent. The validator pins on the observed 40. +8. `wells.parquet` rows are limited to the union of `instances.well_id WHERE well_kind = 'real'`. +9. Soft-warn if any Observations Parquet exceeds 50 MB (Cloudflare cache headroom). + +## Petrobras 3W dataset — environments + +Two hosting paths sit alongside each other, never sharing state: + +- **Local dev** — `deploy.sh` rsyncs `parquet/` into the Caddy-mounted volume; Caddy serves `dev-petrodb.ocortez.com` straight from disk. Constants fall back to this host when `BASE_URL` is unset. Iterating on a feature branch goes here, never to Cloudflare. +- **Cloudflare Pages** — one project (`vars.CLOUDFLARE_PAGES_PROJECT`) with two deployments selected by `wrangler pages deploy --branch `. `.github/workflows/deploy-cloudflare-pages.yml` runs on push to `main` (→ production at `petrodb.ocortez.com`) and to `stage` (→ preview deployment). Push-to-`main` is the publish gate; there is no separate promote step. The workflow triggers on any change under `parquet/` or the 3W pipeline sources and always uploads the whole `parquet/` directory so committed datasets (Argentina, FORCE 2020, Volve) ride along with the 3W tree. `BASE_URL` is per-branch (`vars.BASE_URL_MAIN` / `vars.BASE_URL_STAGE`) and holds only the host (e.g. `https://petrodb.ocortez.com`) — the `/petrobras_3w` dataset segment is appended by the constants module to match this dataset's directory under `parquet/`. The host is per-branch because it's stamped into `instances.parquet#source_url` at build time and the catalog must reference the host where its own bytes are reachable. The 3W pipeline re-run itself is gated by a cache keyed on the 3W sources + the resolved `BASE_URL`: a docs-only or non-3W parquet change deploys from the previously-generated tree without rebuilding. Cloudflare deduplicates by content hash so unchanged files don't re-transfer. See [ADR-0003](docs/adr/0003-cloudflare-pages-hosting.md). diff --git a/README.md b/README.md index bd8ce5b..0932522 100644 --- a/README.md +++ b/README.md @@ -53,6 +53,43 @@ four-bucket rationale, and three more canonical query patterns live in + + +### Petrobras 3W Dataset +Labelled 1-Hz sensor-data windows from the Petrobras 3W dataset, sliced +into per-Instance Parquet files. Pinned at upstream git tag `v.1.70.0` +(dataset version `2.0.0`). This release publishes the +event-class lookup, the real-Well master, the full Instance catalog, and +the per-Instance Observations time-series (hive-partitioned by event class). + +Measure the labelled-data balance across the corpus from the catalog alone +(no Observations scan needed): + +```python +import duckdb + +base = 'https://dev-petrodb.ocortez.com/petrobras_3w' +result = duckdb.sql(f""" + SELECT + et.event_class, + et.description, + COUNT(*) AS n_instances, + SUM(i.n_rows) AS n_observations + FROM '{base}/instances.parquet' i + JOIN '{base}/event_types.parquet' et + ON et.event_class = i.event_class + GROUP BY et.event_class, et.description + ORDER BY et.event_class +""").df() +``` + +Full per-column English docs (including the 27-sensor glossary mirrored +from upstream `dataset.ini`) live in +[`parquet/petrobras_3w/README.md`](parquet/petrobras_3w/README.md). Upstream +source: (CC BY 4.0). + + + ## Access Data Browse and download files at: **https://dev-petrodb.ocortez.com** diff --git a/docs/adr/0001-petrobras-3w-observations-layout.md b/docs/adr/0001-petrobras-3w-observations-layout.md new file mode 100644 index 0000000..a915042 --- /dev/null +++ b/docs/adr/0001-petrobras-3w-observations-layout.md @@ -0,0 +1,24 @@ +# Petrobras 3W observations layout: one file per Instance, hive on `event_class` + +**Status:** accepted + +## Context + +The Petrobras 3W dataset is a corpus of ~2,228 labeled 1-Hz sensor-data windows ("Instances"), totalling ~1.74 GB compressed. Instance length varies from ~21k rows (~6 h) to ~243k rows (~3 days). The ML workload that dominates is event-detector training, which has two distinct access patterns: (a) "load all instances of event class N" and (b) "load one specific Instance for visualization or single-window training." Petrodb is published as static Parquet files behind Cloudflare with a soft per-file size target of 50 MB (CONTEXT.md `Argentina dataset — pre-publish validation` rule 6). + +## Decision + +Publish `observations/` as `observations/event_class=N/.parquet` — hive-partitioned by `event_class` only, with one file per Instance preserving upstream's per-instance file boundaries. Inside each file, in addition to the 30 upstream columns, store `instance_id`, `well_id`, `well_kind` as constant columns (RLE-encoded, negligible cost) so the dataset stays self-describing under any future restructuring. + +## Considered alternatives + +- **Single coalesced `observations.parquet`** with row-group sort on `(event_class, instance_id, timestamp)`. Best for corpus-wide aggregate scans; rejected because the 1.74 GB single file busts the per-file cache target and a cold CDN miss serves the entire file even when DuckDB only needs a row-group range. +- **Hive on `event_class` only, coalesced within partition** (10 files, ~50–500 MB each). Rejected: partition 0 (NORMAL, 594 instances) alone is hundreds of MB; busts the cache target; also discards the per-Instance file boundary, which is the natural unit of the ML workload. +- **Hive on `(event_class × well_id)`** (~100–150 partitions of ~10–20 MB). Fits the cache target and is efficient for combined class+well filters, but the dominant pattern is single-Instance fetch — pulling a 15 MB partition to read a 0.5 MB Instance is wasted bandwidth and worse cold-start latency. +- **Thin mirror of upstream** (`dataset/N/*.parquet`). Rejected: leaves `well_id`, `event_class`, `well_kind`, `instance_id` encoded only in filenames and directory names, requiring tribal knowledge to query. Contradicts petrodb's mission of clean, non-redundant relational schemas (CONTEXT.md, line 3). + +## Consequences + +- Corpus-wide aggregate scans require 2,228 parallel HTTP requests rather than a single large GET. DuckDB httpfs handles this concurrently, and the catalog tables (`instances.parquet`, `wells.parquet`, `event_types.parquet`) answer most aggregate questions without touching `observations/` at all. +- The published URL space is committed: `observations/event_class=N/.parquet` becomes part of the public API. Reorganizing later breaks consumers. +- Adding `instance_id` / `well_id` / `well_kind` as constant columns means future coalescing (if ever needed) is a transparent operation — no schema migration for downstream consumers. diff --git a/docs/adr/0002-petrobras-3w-pin-upstream-release-tag.md b/docs/adr/0002-petrobras-3w-pin-upstream-release-tag.md new file mode 100644 index 0000000..c44fde5 --- /dev/null +++ b/docs/adr/0002-petrobras-3w-pin-upstream-release-tag.md @@ -0,0 +1,24 @@ +# Pin Petrobras 3W upstream to a release tag, not `main` + +**Status:** accepted + +## Context + +Petrobras 3W uses semantic versioning (`VERSIONING.md` upstream). Past minor releases have reshaped instance contents in-place: 1.1.0 added/removed instances, adjusted expert labels, and corrected historian-tag misconfigurations that retroactively changed sensor values. The same `instance_id` can therefore have different bytes between upstream commits, with no rename or URL change to signal it. Petrodb publishes parquet at stable URLs, so silently-changing bytes would propagate as silently-changing training data to downstream consumers. + +## Decision + +The Petrobras 3W pipeline reads from a pinned upstream git tag that ships a specific upstream *dataset version*. The currently pinned git tag is `v.1.70.0`, which ships dataset version `2.0.0`. Refreshes are event-driven (when upstream cuts a new git tag — typically corresponding to a new dataset version — and we've reviewed the release notes), not calendar-driven. The current pinned git tag and dataset version are recorded in `parquet/petrobras_3w/README.md` and emitted in the pipeline's validation logs. + +Note on upstream versioning: upstream uses two distinct version namespaces. Git tags are formatted `v.1.NN.0` (with the dot after `v`); these are the only mechanism a clone can pin to. The *dataset* itself carries a separate semver in `dataset/README.md` (`1.0.0`, `1.1.0`, `1.1.1`, `2.0.0`, …) — this is the version that identifies the data shape and content. The dataset version is what consumers care about; the git tag is the pinning mechanism that delivers it byte-stably. A single dataset version is typically shipped by many consecutive git tags (toolkit/docs fixes don't bump the dataset); for stability we pin to the latest available git tag for the chosen dataset version. + +## Considered alternatives + +- **Track upstream `main`.** Rejected — re-pulling at any time risks silently mutating already-published `instance_id`s (sensor-value corrections, label adjustments). Consumers relying on bytes-stable URLs would observe non-reproducible behaviour. +- **Pin to a commit SHA.** Strongest reproducibility, but upstream cuts release tags deliberately at coherent dataset states; an arbitrary SHA boundary is harder to reason about and harder to compare against the release notes. The marginal reproducibility gain over a tag is small enough not to justify the extra friction. + +## Consequences + +- Pre-publish validation rule 7 (real-Well coverage equals upstream's stated count) implicitly enforces the pin — an accidental tag change shows up as a row-count mismatch, not a silent corruption. +- Petrodb's release cadence for this dataset is bounded above by upstream's release cadence. If upstream goes dormant we publish stale data; that's acceptable for the reproducibility guarantee it buys. +- The pinned tag is part of the dataset's published metadata; bumping it is a deliberate, reviewable action with release-note context, not an automated refresh. diff --git a/docs/adr/0003-cloudflare-pages-hosting.md b/docs/adr/0003-cloudflare-pages-hosting.md new file mode 100644 index 0000000..d199e96 --- /dev/null +++ b/docs/adr/0003-cloudflare-pages-hosting.md @@ -0,0 +1,31 @@ +# Host the prod parquet tree on Cloudflare Pages + +**Status:** accepted + +## Context + +The prod parquet tree is published at a stable public URL so consumers can run DuckDB `httpfs` queries directly against it. ADR-0001 commits that URL space — `observations/event_class=N/.parquet` and friends — as the dataset's public API, so the chosen host has to honour those paths verbatim. + +The Petrobras 3W dataset, generated end-to-end (event-class lookup, real-Well master, Instance catalog, per-Instance Observations tree at the pinned upstream tag — see ADR-0002), weighs ~2.3 GB across ~2,228 parquet files. The current prod deploy targets GitHub Pages, which enforces a hard 1 GB published-site cap; the next push including the full 3W tree would fail at `actions/upload-pages-artifact`. GitHub Pages also imposes a 100 GB/month soft bandwidth cap, which a public ML dataset is well-positioned to blow through. + +Refreshes are event-driven (a new upstream git tag — ADR-0002), not calendar-driven, so the deploy model needs to handle "one immutable publish per push" gracefully — not "stream tiny diffs continuously." + +## Decision + +Publish the prod parquet tree to Cloudflare Pages at `petrodb.ocortez.com`. The `parquet/` directory is the deploy root; Wrangler uploads it from a GitHub Actions workflow on push-to-main. Cloudflare's content-hash deduplication on the Wrangler upload path means subsequent deploys only re-transfer files whose bytes changed. + +The dev environment (`dev-petrodb.ocortez.com`, Caddy reverse proxy fed by `deploy.sh` rsync) is unaffected — Cloudflare Pages replaces the prod host only. + +## Considered alternatives + +- **Stay on GitHub Pages.** Rejected at the forcing function: the 3W tree is over twice the 1 GB site cap and would fail to publish. Even if it fit, the 100 GB/month bandwidth ceiling is a foreseeable problem for a public, queryable parquet dataset that ML consumers may scan from many regions. +- **Cloudflare R2.** R2 is excellent object storage and would fit the byte volume, but it lacks the atomic "one immutable deploy per push" model that Pages provides. A refresh against R2 is a per-file PUT loop with no transactional boundary — a partially completed refresh can serve a mix of old and new bytes against the committed URL space (ADR-0001), and there is no first-class rollback to a prior known-good snapshot. Pages's atomic deploys with instant rollback fit the event-driven refresh cadence (ADR-0002) far better, where each upstream tag bump should be one reviewable publish. +- **Self-host behind Caddy on the dev VM.** Rejected — prod traffic on a single home-lab host is a liability for a dataset meant to be publicly queryable, and the dev VM's bandwidth and uptime are not the right SLA for the public API. + +## Consequences + +- The published URL space committed in ADR-0001 stays exactly what consumers expect — Pages serves `parquet/` at the deploy root, so `https://petrodb.ocortez.com/petrobras_3w/observations/event_class=N/.parquet` resolves unchanged. +- Cloudflare Pages enforces a **25 MiB per-file hard cap** on uploads. The pre-publish validator's existing 50 MB soft-warn (CONTEXT.md *Petrobras 3W dataset — pre-publish validation* rule 9; mirrored as rule 6 for Argentina) sits comfortably below this — at the currently pinned upstream tag every Observations file is under 25 MiB, so the soft-warn fires well before Pages would reject a file. If a future upstream change ever produces an outlier Instance, the soft-warn surfaces it during pre-publish, not as a 4xx during the Wrangler upload. +- Cloudflare Pages free plan caps deploys at 500/month and concurrent builds at 1 — both comfortably above the event-driven refresh cadence (ADR-0002), where a publish is gated on upstream cutting a new tag. +- DNS for `ocortez.com` already lives in Cloudflare, so wiring `petrodb.ocortez.com` to the Pages project is a one-time UI action with no zone-transfer or split-DNS complexity. +- The deploy contract becomes "every push to `main` is a publish." This matches the existing reviewable-merge workflow but is worth naming: there is no separate "promote to prod" step. diff --git a/parquet/index.html b/parquet/index.html index 1e60c7f..fef3c81 100644 --- a/parquet/index.html +++ b/parquet/index.html @@ -794,6 +794,12 @@

PETRODATA REPOSITORY

4 files + + + @@ -2001,6 +2007,192 @@

Source & License

+ + +
+ +
+

Download Petrobras 3W Files

+

+ Labelled 1-Hz sensor-data windows from the Petrobras 3W dataset. + Pinned at upstream git tag v.1.70.0 + (dataset version 2.0.0). This release + publishes the event-class lookup, the real-Well master, the full + Instance catalog, and the per-Instance Observations time-series + hive-partitioned by event_class. +

+ + + +

Lookup + Wells master + Instance catalog + per-Instance Observations (hive-partitioned by event_class) · pinned upstream identity logged on every publish

+
+ + +
+

About This Dataset

+

+ The Petrobras 3W dataset is a corpus of + ~2,228 labelled 1-Hz sensor-data windows recorded on + Petrobras's offshore wells, framed around at most one + anomaly event per window. The full corpus covers ten + operational regimes (NORMAL plus nine anomaly categories + such as Hydrate in Production Line and + Severe Slugging) across ~40 distinct real wells, + supplemented by simulated and hand-drawn instances. +

+

+ Petrodb pins the upstream repository at git tag + v.1.70.0 (dataset version + 2.0.0) — refreshes are + event-driven on new upstream releases, never silent. +

+
+ + +
+

Quick Start with DuckDB

+

+ Measure the labelled-data balance across the corpus from the + Instance catalog alone (no Observations scan needed): +

+
+
+ + + +
+
import duckdb
+
+base = 'https://dev-petrodb.ocortez.com/petrobras_3w'
+result = duckdb.sql(f"""
+    SELECT
+        et.event_class,
+        et.description,
+        COUNT(*)             AS n_instances,
+        SUM(i.n_rows)        AS n_observations
+    FROM '{base}/instances.parquet' i
+    JOIN '{base}/event_types.parquet' et
+        ON et.event_class = i.event_class
+    GROUP BY et.event_class, et.description
+    ORDER BY et.event_class
+""").df()
+
+

+ The per-Instance Observations files are accessible via the + hive-partitioned URL pattern + observations/event_class=N/<instance_id>.parquet. + Each file embeds instance_id, well_id, + and well_kind as constant columns, so corpus-wide + queries against a single event class do not need to join the + catalog: +

+
+
+ + + +
+
-- All real-Well Hydrate-in-Production-Line observations
+SELECT instance_id, well_id, "timestamp", "P-PDG", "T-PDG", class
+FROM 'https://dev-petrodb.ocortez.com/petrobras_3w/observations/event_class=8/*.parquet'
+WHERE well_kind = 'real';
+
+
+ + +
+

Schema Documents

+

+ Full per-column documentation is published alongside the parquets: +

+
+
+

README.md

+

Dataset overview, pinned upstream identity, query examples

+
+ → Open README.md +
+
+
+

schema.md

+

English column docs + 27-sensor glossary mirrored from upstream dataset.ini

+
+ → Open schema.md +
+
+
+

schema.json

+

Machine-readable column list, types, primary & foreign keys

+ +
+
+

schema.sql

+

DDL that mirrors the published structure in a fresh DuckDB

+
+ → Open schema.sql +
+
+
+
+ + +
+

Source & License

+

+ Upstream repository: https://github.com/petrobras/3W.git + (pinned at git tag v.1.70.0, dataset version + 2.0.0). +

+

+ Licensed under Creative Commons Attribution 4.0. + All credit for the underlying measurements, labelling, and dataset + design belongs to Petrobras and the upstream maintainers. +

+
+
+ + +