From 1c9e3c4d94d7fda534b406c90193e68b99f06ced Mon Sep 17 00:00:00 2001 From: Maarten Jacobs Date: Wed, 23 Sep 2026 16:26:09 +0200 Subject: [PATCH] pipeline-sql: run one SQL query across a stage as a merged table Running SQL through pipeline-cmd ('pg:psql -c ...') squeezes each app's own psql table into the appname | output layout, which is unreadable. pipeline-sql runs a single query against every app's DATABASE_URL via pg:psql -f and prints ONE table with the app name as the first column (aligned on a terminal, appname;col1;col2 when piped). psql is driven entirely through meta-commands in the script file, since heroku's pg:psql exposes no psql flags: unaligned tab-separated output, no footer, ON_ERROR_STOP, and a sentinel \echo line that lets the worker discard anything ~/.psqlrc printed first. stdout and stderr are captured separately so NOTICEs never become rows; they are forwarded to stderr with the app name. Failed apps get their error in place of rows, zero-row apps are skipped with a count (or one empty row with -a), and a header that differs between apps is reported as schema drift. The chunked fan-out core of run_pipeline_workers is extracted into make_run_tmpdir / fan_out_workers / report_skipped so all three commands share it; pipeline-cmd and config-replace output is unchanged. Co-Authored-By: Claude Fable 5.1 --- README.md | 103 ++++++++ bin/heroku-scripts | 551 +++++++++++++++++++++++++++++++++++---- test/heroku-scripts.bats | 303 +++++++++++++++++++++ 3 files changed, 900 insertions(+), 57 deletions(-) diff --git a/README.md b/README.md index f0e4bac..89e2a04 100644 --- a/README.md +++ b/README.md @@ -70,6 +70,7 @@ HEROKU_API_KEY="op://Private/Heroku/credential" op run -- heroku-scripts apps my heroku-scripts apps heroku-scripts pipeline-cmd "" [--concurrency=N] [--retries=N] [--no-stream] [-a] [--table|--csv] heroku-scripts config-replace [--concurrency=N] [--dry-run] [-a|--all] [--table|--csv] [--no-stream] +heroku-scripts pipeline-sql ("" | --file=) [--concurrency=N] [-a|--all] [--table|--csv] heroku-scripts pipeline-task [--concurrency=N] heroku-scripts promote [--dry-run] [--yes] heroku-scripts deploy-slug [--from=] [--dry-run] [--yes] @@ -149,6 +150,108 @@ last error as its record. The default is `--retries=0` (unchanged behavior). One caveat: a mid-session drop can happen *after* the remote command started running, so only use `--retries` with commands that are safe to run twice. +### Run one SQL query across a stage, as one merged table + +Running SQL through `pipeline-cmd` works — `pipeline-cmd my-pipe production +'pg:psql -c "select count(*) from users"'` — but reads badly: each app's +output is psql's own aligned table (header, dashes, rows, `(N rows)` footer), +and that multi-line blob gets squeezed into pipeline-cmd's `appname | output` +layout. `pipeline-sql` runs a single query against the default database +(`DATABASE_URL`) of every app in the stage and prints ONE table with the app +name as the first column: + +```sh +heroku-scripts pipeline-sql my-pipe production "select count(*) as users, max(inserted_at) as newest from users" +``` + +``` +# on a terminal +appname | users | newest +--------------+-------+-------------------------- +my-app | 1204 | 2026-09-22 14:03:11.51239 +my-app-worker | 1204 | 2026-09-22 14:03:11.51239 +my-other-app | 87 | 2026-09-19 09:12:40.00417 + +# piped +appname;users;newest +my-app;1204;2026-09-22 14:03:11.51239 +my-app-worker;1204;2026-09-22 14:03:11.51239 +my-other-app;87;2026-09-19 09:12:40.00417 +``` + +As with `pipeline-cmd`, the format is a table on a terminal and CSV when +piped; force either with `--table` or `--csv`. Longer queries can come from a +file instead of the command line: + +```sh +heroku-scripts pipeline-sql my-pipe production --file=reports/orphaned-invoices.sql +``` + +Inline SQL that starts with a `-- comment` line would be mistaken for an +option; put `--` (end of options) in front of it: + +```sh +heroku-scripts pipeline-sql my-pipe production -- "-- locked accounts +select id, email from users where locked" +``` + +Either way the SQL is expected to be **one statement that returns a result +set**. Statements that return nothing (an `UPDATE`, a `DO` block) are +reported as an error row rather than silently succeeding, and several +statements in one file are not supported. + +The output is **buffered, not streamed**: the merged header and every column +width depend on all apps' results, so nothing is printed until the last app +finishes, and the rows come out sorted by app name. There are no +`--stream`/`--no-stream` flags. `--concurrency=N` works as in `pipeline-cmd`. + +Apps whose query returns **no rows** are skipped, with a count on stderr; pass +`-a`/`--all` to give each of them one row with empty cells instead: + +```sh +heroku-scripts pipeline-sql my-pipe production "select id, email from users where locked" -a +# appname;id;email +# my-app;42;someone@example.com +# my-app-worker;; +``` + +Apps whose query **fails** — a SQL error, or heroku itself failing (no +database attached, no access) — get an error row in place of data, spanning +the data columns, with psql's `psql:::` prefix and heroku's +`Error: psql exited with code N` trailer removed so only the message remains. +The other apps are unaffected: + +``` +appname | id | name +----------+----+------ +app-one | 1 | alice +app-one | 2 | bob +app-three | ERROR: relation "users" does not exist + | LINE 1: select * from users + | ^ +``` + +The merged header is the first app's (in name order). If a later app returns +different columns — schema drift between apps — its rows are still printed, +but a warning like `pipeline-sql: app-two: columns differ from app-one (id, +mail vs id, email)` goes to stderr, since its cells will sit under the wrong +column names. + +Only the query result goes into the table. Anything psql says on the side +during a *successful* query — a `NOTICE` from a `RAISE NOTICE`, a `WARNING`, +an identifier-truncation notice — is forwarded to stderr, one line each, +prefixed with the app name (`pipeline-sql: my-app: NOTICE: ...`), so it is +neither lost nor mistaken for data. + +Two caveats. Cells are tab-separated internally and rows newline-separated, +so a value that itself contains a tab or newline will break its row — like +`pipeline-cmd`'s CSV, this output is for reading and grepping, not strict +parsing. And because `heroku pg:psql` offers no way to pass psql's `-X`/`-q` +flags, your `~/.psqlrc` is still read: pipeline-sql neutralises its output +and formatting settings (`\timing`, `\pset` changes, `\o` redirection, +`\echo` chatter) from inside the query script, but a psqlrc that errors out +or changes `\set ON_ERROR_STOP` semantics will still get in the way. + ### Replace a config var's value across a stage, only where it currently matches `config-replace` is a guarded `config:set`: it reads each app's current value diff --git a/bin/heroku-scripts b/bin/heroku-scripts index cd650f0..81be230 100755 --- a/bin/heroku-scripts +++ b/bin/heroku-scripts @@ -51,6 +51,24 @@ Commands: the same empty line for an unset var and one set to the empty string, so a var set to "" is treated as not set. + pipeline-sql ("" | --file=) [--concurrency=N] [-a|--all] [--table|--csv] + Run one SQL statement against the default database (DATABASE_URL) of + every app in / via \`pg:psql\`, and print the results + as ONE merged table with the app name as the first column — where + pipeline-cmd would show each app's own psql table. The SQL, inline or + from --file, must be a single statement that returns a result set. + Output is always buffered and sorted by app name (the merged header + and column widths depend on every app's result), so there are no + streaming flags. Apps whose query returns no rows are skipped (a + count is reported on stderr); -a/--all gives them one empty row. An + app whose query fails gets its error message in place of rows. + + Table on a terminal, CSV (\`appname;col1;col2\`) when piped; force + either with --table or --csv. Cells are tab-separated internally, so + values containing tabs or newlines will break rows: the output is + for reading/grepping, not strict parsing. Your ~/.psqlrc is still + read, but its output and formatting settings are neutralised. + pipeline-task [--concurrency=N] Run \`mix \` via \`heroku run\` against every app in /. Defaults to 3 concurrent runs. @@ -307,53 +325,75 @@ emit_record() { done <<< "$output" } -# Shared skeleton for the commands that fan a per-app worker out across a -# pipeline stage: chunked parallelism, streaming or buffered record emission, -# table/CSV rendering, and the skipped-apps summary. pipeline-cmd and -# config-replace differ only in what they do per app, so that part comes in as -# a worker function name. Worker-specific parameters (the heroku command, the -# config var, ...) are the calling command's locals, which bash's dynamic -# scoping keeps visible inside the worker. +# Resolves a command's output format — `auto` becomes a table on a terminal +# and CSV when piped or redirected; --table/--csv pass through untouched — and +# stores it in the variable named by $1 (printf -v, as in make_run_tmpdir). # -# The worker is called with one argument (the app name), prints the app's -# record on stdout, and returns non-zero to skip the app: no record is emitted -# and the app is counted in the stderr summary, rendered as -# " $skip_summary". -run_pipeline_workers() { - local worker="$1" apps="$2" concurrency="$3" stream="$4" format="$5" skip_summary="$6" - - # Resolve auto -> table on a terminal, CSV when piped. The table's left column - # is sized from the (already known) app list, so table output still streams. - local render="$format" - if [[ "$render" == "auto" ]]; then - if [[ -t 1 ]]; then render=table; else render=csv; fi +# It must NOT be called inside `$(...)`: a command substitution's stdout is +# the capture pipe, so `[[ -t 1 ]]` would always be false and every auto run +# would render CSV, even on a terminal. Hence the out-variable. +resolve_render_format() { + local varname="$1" format="$2" + if [[ "$format" == "auto" ]]; then + if [[ -t 1 ]]; then format=table; else format=csv; fi fi - local width=7 - [[ "$render" == "table" ]] && width="$(app_column_width "$apps")" + printf -v "$varname" '%s' "$format" +} - local tmpdir - tmpdir="$(mktemp -d)" - # Interpolate $tmpdir into the trap *now*, not at fire time — by the time - # the EXIT trap runs, this function has returned and the local is gone. +# Creates the scratch directory for one fan-out run and stores its path in the +# variable named by $1 (printf -v, so no eval). Installs the cleanup traps too: +# they are process-global, and the directory outlives this function because the +# caller keeps reading per-app files out of it after the workers finish. +# +# Also creates `.skipped/`, where workers drop a marker file for each app they +# skip so the parent can report a count once they all finish. A dotfile dir — +# like everything else the runner itself puts in here — so the collection glob +# (`$tmpdir/*`) only ever sees per-app files. +make_run_tmpdir() { + # The local is deliberately NOT called tmpdir: callers pass that name, and + # a same-named local here would shadow theirs, leaving printf -v writing + # into our own variable and the caller's empty. + local varname="$1" dir + dir="$(mktemp -d)" + # Interpolate $dir into the trap *now*, not at fire time — by the time the + # EXIT trap runs, this function has returned and the local is gone. # shellcheck disable=SC2064 - trap "rm -rf '$tmpdir'" EXIT + trap "rm -rf '$dir'" EXIT # On an external INT/TERM (cron, a wrapper script, plain `kill`, systemd # stop) the signal reaches only the parent. Without this, the backgrounded # heroku subshells are orphaned and keep running while the EXIT trap deletes # the tmpdir out from under them. Tear the jobs down first, then clean up. # shellcheck disable=SC2064 - trap "trap - INT TERM EXIT; kill \$(jobs -p) 2>/dev/null || true; rm -rf '$tmpdir'; exit 130" INT TERM - - # Workers drop a marker file here for each app they skip, so the parent can - # report a count after they all finish. A dotfile dir, so the --no-stream - # collection glob (`$tmpdir/*`) never picks it up. - mkdir "$tmpdir/.skipped" + trap "trap - INT TERM EXIT; kill \$(jobs -p) 2>/dev/null || true; rm -rf '$dir'; exit 130" INT TERM + mkdir "$dir/.skipped" + printf -v "$varname" '%s' "$dir" +} - if [[ "$render" == "table" ]]; then - print_table_header "$width" - else - echo "appname;output" - fi +# The chunked fan-out core shared by every command that runs a per-app worker +# across a pipeline stage (pipeline-cmd, config-replace, pipeline-sql). It only +# runs workers and files their results; what a record means and how it is +# rendered is the caller's business, which is what lets pipeline-sql (one +# merged table, rendered after everything is in) share it with pipeline-cmd +# (one record per app, streamed) without either bending to the other. +# +# The worker is called with one argument (the app name) and prints the app's +# record on stdout. Its exit status decides what happens to the record: +# 0 stored at $tmpdir/ (the filename IS the app name, so the +# caller's `$tmpdir/*` glob yields records sorted by app), and — +# when $on_record is a non-empty function name — also handed to +# `$on_record ` right away, serialized under a lock +# so concurrent finishers can't interleave their output. +# non-zero "skip this app": no record, a marker in $tmpdir/.skipped/ instead. +# Worker-specific parameters (the heroku command, the config var, the SQL +# file, ...) are the calling command's locals, which bash's dynamic scoping +# keeps visible inside the worker. +fan_out_workers() { + local worker="$1" apps="$2" concurrency="$3" tmpdir="$4" on_record="$5" + # An empty/missing tmpdir would send the workers' writes to / and make + # the streaming lock (`mkdir "$tmpdir/.lock"`) fail forever — a silent hang, + # so check it up front. + [[ -n "$tmpdir" && -d "$tmpdir/.skipped" ]] \ + || die "internal error: fan_out_workers needs a directory from make_run_tmpdir" # Process apps in chunks of $concurrency. Waiting for each chunk to drain # before starting the next caps resource use; running every app in parallel @@ -374,7 +414,10 @@ run_pipeline_workers() { # killing the subshell before it can emit its record. set +e if output="$("$worker" "$app")"; then - if [[ "$stream" == "true" ]]; then + # Always file the record (the caller reads these back for buffered, + # sorted rendering); the write is trivially cheap even when streaming. + printf '%s' "$output" > "$tmpdir/$app" + if [[ -n "$on_record" ]]; then # Print as soon as this app finishes. Serialize with an atomic mkdir # lock so a record streams out in one piece (emit_record may write # several lines for multi-line/table output) even when apps finish at @@ -384,12 +427,8 @@ run_pipeline_workers() { # breaks on the first failed write so the lock is released promptly. trap '' PIPE while ! mkdir "$tmpdir/.lock" 2>/dev/null; do sleep 0.05; done - emit_record "$render" "$width" "$app" "$output" + "$on_record" "$app" "$output" rmdir "$tmpdir/.lock" - else - # Buffered: store the raw output (filename is the app name) and - # format it sorted at the end. - printf '%s' "$output" > "$tmpdir/$app" fi else # Non-zero from the worker means "skip this app": record it for the @@ -407,6 +446,55 @@ run_pipeline_workers() { fi done 3<<< "$apps" wait || true +} + +# Reports how many apps the workers skipped, on stderr so it never pollutes +# the data stream piped from stdout. Separated from the output by a blank +# line and rendered dim + italic — but only when stderr is a terminal and +# NO_COLOR is unset, so redirected/piped output stays plain text. Silent when +# nothing was skipped. +report_skipped() { + local tmpdir="$1" skip_summary="$2" skipped + skipped="$(find "$tmpdir/.skipped" -type f | wc -l | tr -d ' ')" + (( skipped > 0 )) || return 0 + local msg="$skipped $skip_summary" + if [[ -t 2 && -z "${NO_COLOR:-}" ]]; then + printf '\n\033[2;3m%s\033[0m\n' "$msg" >&2 + else + printf '\n%s\n' "$msg" >&2 + fi +} + +# Shared skeleton for the commands that emit one record per app (pipeline-cmd, +# config-replace): fan_out_workers plus streaming or buffered `appname | +# output` rendering and the skipped-apps summary, rendered as +# " $skip_summary". The two commands differ only in what they do per +# app, so that part comes in as a worker function name — see fan_out_workers +# for the worker contract. +run_pipeline_workers() { + local worker="$1" apps="$2" concurrency="$3" stream="$4" format="$5" skip_summary="$6" + + # The table's left column is sized from the (already known) app list, which + # is what lets table output stream. $render and $width are read by + # stream_pipeline_record through dynamic scoping. + local render width=7 + resolve_render_format render "$format" + [[ "$render" == "table" ]] && width="$(app_column_width "$apps")" + + local tmpdir + make_run_tmpdir tmpdir + + if [[ "$render" == "table" ]]; then + print_table_header "$width" + else + echo "appname;output" + fi + + # Streaming hands each record to stream_pipeline_record the moment its app + # finishes; --no-stream passes no callback and renders from the files below. + local on_record="" + [[ "$stream" == "true" ]] && on_record=stream_pipeline_record + fan_out_workers "$worker" "$apps" "$concurrency" "$tmpdir" "$on_record" # Streaming mode already printed each record. For --no-stream, emit the # buffered per-app rows now, in a stable app-name order. Bash sorts glob @@ -422,20 +510,13 @@ run_pipeline_workers() { done fi - # Report how many apps the workers skipped, on stderr so it never pollutes - # the data stream piped from stdout. Separated from the output by a blank - # line and rendered dim + italic — but only when stderr is a terminal and - # NO_COLOR is unset, so redirected/piped output stays plain text. - local skipped - skipped="$(find "$tmpdir/.skipped" -type f | wc -l | tr -d ' ')" - if (( skipped > 0 )); then - local msg="$skipped $skip_summary" - if [[ -t 2 && -z "${NO_COLOR:-}" ]]; then - printf '\n\033[2;3m%s\033[0m\n' "$msg" >&2 - else - printf '\n%s\n' "$msg" >&2 - fi - fi + report_skipped "$tmpdir" "$skip_summary" +} + +# fan_out_workers callback for run_pipeline_workers' streaming mode. $render +# and $width are run_pipeline_workers' locals (dynamic scoping). +stream_pipeline_record() { + emit_record "$render" "$width" "$1" "$2" } # Per-app worker for pipeline-cmd: run the heroku command and use its combined @@ -585,6 +666,359 @@ cmd_config_replace() { "app(s) without $var skipped (use -a/--all to include them)" } +# Marks where a pipeline-sql query's real output begins. The user's ~/.psqlrc +# runs before our script and may print to stdout (e.g. "Timing is on.") +# before our `\set QUIET on` can silence anything, so the script ends its +# setup with `\echo ` and the worker discards everything up to and +# including that line. Fixed, printable, and unlikely to appear in a result. +PIPELINE_SQL_SENTINEL='--- heroku-scripts pipeline-sql ---' + +# Prints the psql script pipeline-sql hands to every app: formatting +# meta-commands, the sentinel, then the user's SQL verbatim. +# +# heroku's pg:psql accepts only -c or -f — no way to pass psql's own -X/-q/-A +# flags — so all formatting has to happen through meta-commands inside the +# file, and -c can't mix meta-commands with SQL, hence -f. Line by line: +# QUIET on no "Output format is unaligned." style confirmations for +# the \pset calls that follow +# ON_ERROR_STOP a SQL error makes psql exit non-zero, which heroku turns +# into a non-zero exit of its own ("psql exited with code N") +# — that is how the worker tells an error from a result +# ECHO none don't echo the query back +# \timing off undo a ~/.psqlrc `\timing`, or every app appends "Time: …" +# unaligned, tab fieldsep, newline recordsep +# one header line + one line per row, tab-separated, so the +# renderer can merge results without parsing psql's ASCII art +# tuples_only off keep the header line: it names the merged table's columns +# footer off no "(N rows)" trailer +# expanded off, \pset title (unset) +# undo psqlrc settings that would change the line layout +# null '' NULLs print as empty cells +# \o (bare) send query output back to stdout, undoing a +# psqlrc `\o somefile` that would swallow the result +write_pipeline_sql_script() { + local sql="$1" + cat <<'EOF' +\set QUIET on +\set ON_ERROR_STOP on +\set ECHO none +\timing off +\o +\pset format unaligned +\pset fieldsep '\t' +\pset recordsep '\n' +\pset tuples_only off +\pset footer off +\pset expanded off +\pset title +\pset null '' +EOF + printf '\\echo %s\n' "$PIPELINE_SQL_SENTINEL" + printf '%s\n' "$sql" +} + +# Per-app worker for pipeline-sql: runs the shared psql script against the +# app's default database and prints the app's result as tagged lines, one +# per line (the tag and payload are tab-separated): +# H
the tab-separated column names (always first, at most one) +# R one tab-separated data row +# E one line of an error message (record has only E lines) +# A per-line tag rather than a one-line-marker-then-payload layout because +# the runner captures the record with `$(...)`, which strips trailing +# newlines: a last row whose only cell is empty (`select null`) would vanish +# untagged, whereas "R" survives. +# +# Always returns 0 and always emits the H line when the query ran: the +# "no rows" skip is decided in cmd_pipeline_sql's merge loop, because even a +# zero-row app's header is needed to name (or drift-check) the merged +# table's columns. $sql_file and $raw_dir are cmd_pipeline_sql's locals. +pipeline_sql_worker() { + local app="$1" status + local out="$raw_dir/$app.out" err="$raw_dir/$app.err" + # A real argv call — the file path and app are discrete values, not a shell + # command. Its output goes to files, not `$(...)`, and stdout and stderr go + # to SEPARATE files, for three reasons: + # - `$(...)` strips trailing newlines, and a trailing empty row (`select + # null as x` prints "x", then an empty line) would be lost before it + # could be tagged. awk reading the file does see that final empty line. + # - heroku's pg:psql pipes psql's stdout (the result, nothing else) + # through its own stdout, while psql's NOTICE/WARNING/ERROR lines, the + # "--> Connecting to ..." banner, and heroku's own errors all go to + # stderr. Merged, a NOTICE raised by a successful query would land after + # the sentinel and be taken for the header or a row. + # - the result must never pass through strip_heroku_noise: a data value + # that happens to look like the connecting banner would be deleted. + # Only stderr is filtered. + # The exit status is what tells a SQL error apart from a result. + heroku pg:psql -f "$sql_file" -a "$app" > "$out" 2> "$err" + status=$? + + local found=false + grep -qFx -e "$PIPELINE_SQL_SENTINEL" "$out" && found=true + + if (( status != 0 )) || [[ "$found" != "true" ]]; then + # Error: a SQL error (non-zero, sentinel present) or heroku itself failing + # before psql ran (no database attached, no such app — usually non-zero, + # but the missing sentinel alone is proof that no result was produced). + # The message is stderr, minus heroku's ambient noise and minus what the + # operator doesn't need: the `psql::: ` prefix psql puts + # on errors from -f files (the temp path means nothing to anyone), + # heroku's own "Error: psql exited with code N" trailer (the real + # `ERROR: ...` line above it is the message), and the ` › ` bullet the + # heroku CLI puts in front of its own messages. `LINE 1: ...` and the + # caret line stay. + local message + message="$(strip_heroku_noise < "$err" | awk ' + /^[[:space:]]*(›[[:space:]]+)?Error: psql exited with code [0-9]+/ { next } + { + sub(/^psql:[^[:space:]]+:[0-9]+: /, "") + sub(/^[[:space:]]*›[[:space:]]+/, "") + print "E\t" $0 + } + ')" + if [[ -z "$message" ]]; then + # Nothing usable on stderr (e.g. heroku killed mid-run): still say why + # there are no rows, rather than emitting an empty record. + message="$(printf 'E\tpg:psql exited with status %s and no error message' "$status")" + fi + printf '%s' "$message" + return 0 + fi + + # Success. psql may still have written NOTICE/WARNING lines to stderr (a + # RAISE NOTICE, an identifier-truncation notice): forward them to our own + # stderr, prefixed with the app, so they aren't silently dropped — they + # can't go in the table without being mistaken for data. The + # `psql::: ` prefix goes for the same reason as on errors. + strip_heroku_noise < "$err" \ + | awk -v app="$app" ' + NF { sub(/^psql:[^[:space:]]+:[0-9]+: /, ""); print "pipeline-sql: " app ": " $0 } + ' >&2 + + # The header line and rows are whatever follows the sentinel on stdout. + local result + result="$(awk -v s="$PIPELINE_SQL_SENTINEL" ' + seen { print (n++ ? "R" : "H") "\t" $0 } + $0 == s { seen = 1 } + ' "$out")" + if [[ -z "$result" ]]; then + # Nothing at all after the sentinel: the statement produced no result set + # (an UPDATE/INSERT, whose command tag QUIET suppresses, or a bare + # meta-command). Not "no rows" — there were no columns either — so report + # it rather than silently skipping the app. + printf 'E\tno result set (pipeline-sql expects one statement that returns rows)' + return 0 + fi + printf '%s' "$result" +} + +# Renders pipeline-sql's collected records — a TSV of +# `` lines in app order, see pipeline_sql_worker +# for the tags — as one merged table (fmt=table) or CSV (fmt=csv). +# +# One awk program in two passes: the whole file is read into arrays first +# because the merged header and every column width depend on every app's +# result; nothing can be printed before the last record is in. Kept to POSIX +# awk so it runs unchanged on macOS's BWK awk and Linux's gawk: widths go +# through sprintf("%-" w "s") rather than the `*` width, and counters stand +# in for length(array). +# +# The merged header is the first app's. A later app whose header differs +# (schema drift) still gets its rows printed but is called out on stderr, +# since its cells will sit under the wrong column names. With no header at +# all (every app errored) the single data column is called "output". +# +# Table: everything left-aligned, cells joined by " | ", a "-+-" rule, no +# right border and no padding on the last cell (so no trailing whitespace). +# Error lines span the data columns after the app cell; continuation lines +# get a blank app cell and count toward no width. An app with a header but +# no rows still contributes its header (see pipeline_sql_worker) but gets a +# row — one of empty cells — only when all=true (-a/--all). +# CSV: `appname;col1;col2`, rows with tabs turned into `;`, error records as +# `app;` plus the remaining lines verbatim (the same shape +# emit_record gives multi-line output in csv mode). +render_pipeline_sql() { + local fmt="$1" all="$2" records="$3" + awk -v fmt="$fmt" -v all="$all" ' + function pad(s, n) { return sprintf("%-" n "s", s) } + function dashes(n, s) { s = sprintf("%-" n "s", ""); gsub(/ /, "-", s); return s } + function commas(s) { gsub(/\t/, ", ", s); return s } + function semis(s) { gsub(/\t/, ";", s); return s } + # One data row in the chosen format: cell[1..n] under maxcols columns. + function row(app, n, line, i, c) { + if (fmt != "table") { + line = app + for (i = 1; i <= maxcols; i++) line = line ";" (i <= n ? cell[i] : "") + print line + return + } + line = pad(app, w0) + for (i = 1; i <= maxcols; i++) { + c = (i <= n ? cell[i] : "") + if (i < maxcols) line = line " | " pad(c, w[i]) + else if (c != "") line = line " | " c + else line = line " |" + } + print line + } + BEGIN { FS = "\t"; w0 = 7; ncols = 0; maxcols = 0; nrec = 0 } # 7 = len("appname") + { + app = $1; tag = $2 + payload = substr($0, length(app) + length(tag) + 3) # past "app\t" and "tag\t" + nrec++; rapp[nrec] = app; rtag[nrec] = tag; rpay[nrec] = payload + if (length(app) > w0) w0 = length(app) + if (tag == "H") { + if (ncols == 0) { + ncols = split(payload, col, "\t"); first_app = app; first_hdr = payload + } else if (payload != first_hdr) { + print "pipeline-sql: " app ": columns differ from " first_app \ + " (" commas(payload) " vs " commas(first_hdr) ")" | "cat 1>&2" + } + } else if (tag == "R") { + rows[app]++ + n = split(payload, cell, "\t") + if (n > maxcols) maxcols = n + for (i = 1; i <= n; i++) if (length(cell[i]) > w[i]) w[i] = length(cell[i]) + } + } + END { + close("cat 1>&2") + if (ncols == 0) { ncols = 1; col[1] = "output" } + if (ncols > maxcols) maxcols = ncols + for (i = 1; i <= maxcols; i++) { + if (i > ncols) col[i] = "" + if (length(col[i]) > w[i]) w[i] = length(col[i]) + } + # Header (and rule). The header is a row too, so reuse row() for it. + for (i = 1; i <= maxcols; i++) cell[i] = col[i] + row("appname", maxcols) + if (fmt == "table") { + line = dashes(w0) + for (i = 1; i <= maxcols; i++) line = line "-+-" dashes(w[i]) + print line + } + for (r = 1; r <= nrec; r++) { + if (rtag[r] == "R") { + n = split(rpay[r], cell, "\t") + row(rapp[r], n) + } else if (rtag[r] == "H") { + # rows[app] is unset (compares equal to 0) for a no-rows app. + if (all == "true" && rows[rapp[r]] == 0) row(rapp[r], 0) + } else if (rtag[r] == "E") { + cont = (r > 1 && rtag[r - 1] == "E" && rapp[r - 1] == rapp[r]) + if (fmt != "table") print (cont ? "" : rapp[r] ";") rpay[r] + else if (rpay[r] != "") print pad(cont ? "" : rapp[r], w0) " | " rpay[r] + else print pad(cont ? "" : rapp[r], w0) " |" + } + } + } + ' "$records" +} + +cmd_pipeline_sql() { + local usage_line="Usage: $SCRIPT_NAME pipeline-sql (\"\" | --file=) [--concurrency=N] [-a|--all] [--table|--csv]" + [[ $# -ge 2 ]] || die "$usage_line" + local pipeline="$1" stage="$2" + shift 2 + + # The SQL comes either inline as the one remaining positional or from + # --file; anything else starting with a dash that we don't know goes to + # parse_concurrency, which owns the "Unknown option" error. No streaming + # flags: the merged table's header and widths depend on every app's result, + # so output is always buffered and sorted by app. + local sql_file_arg="" include_empty=false format=auto + local -a opts=() positional=() + while [[ $# -gt 0 ]]; do + case "$1" in + --file=*) sql_file_arg="${1#*=}";; + -a|--all) include_empty=true;; + --table) format=table;; + --csv) format=csv;; + # End of options: whatever follows is the SQL, even when it starts with + # a dash — a statement that opens with a `-- comment` line would + # otherwise be taken for an option and rejected. + --) shift; while [[ $# -gt 0 ]]; do positional+=("$1"); shift; done; break;; + -*) opts+=("$1");; + *) positional+=("$1");; + esac + shift + done + + if [[ ${#positional[@]} -gt 1 ]]; then + die "$usage_line" + fi + local sql="" + if [[ ${#positional[@]} -eq 1 && -n "$sql_file_arg" ]]; then + die "Give the SQL either inline or with --file, not both"$'\n'"$usage_line" + elif [[ ${#positional[@]} -eq 1 ]]; then + sql="${positional[0]}" + elif [[ -n "$sql_file_arg" ]]; then + [[ -f "$sql_file_arg" && -r "$sql_file_arg" ]] || die "Cannot read SQL file: $sql_file_arg" + # `cat <` rather than `cat file`: a file named `-something.sql` would + # otherwise be taken for an option. + sql="$(cat < "$sql_file_arg")" + else + die "No SQL given: pass it inline or with --file="$'\n'"$usage_line" + fi + [[ -n "${sql//[[:space:]]/}" ]] || die "The SQL is empty" + + local concurrency=3 + if [[ ${#opts[@]} -gt 0 ]]; then + concurrency="$(parse_concurrency "${opts[@]}")" || exit 1 + fi + + resolve_heroku_api_key + + local apps + apps="$(require_apps "$pipeline" "$stage")" || exit 1 + + local render + resolve_render_format render "$format" + + # The psql script is written once and shared read-only by every worker. It + # lives in the run's tmpdir (so the traps clean it up) as a dotfile, so the + # per-app collection glob below never mistakes it for a record. Workers + # also park heroku's raw stdout/stderr per app under `.raw/` (see + # pipeline_sql_worker for why files rather than `$(...)`). $sql_file and + # $raw_dir are read by pipeline_sql_worker through dynamic scoping. + local tmpdir sql_file raw_dir + make_run_tmpdir tmpdir + sql_file="$tmpdir/.query.sql" + write_pipeline_sql_script "$sql" > "$sql_file" + raw_dir="$tmpdir/.raw" + mkdir "$raw_dir" + + # Buffered on purpose (no per-record callback): see render_pipeline_sql. + fan_out_workers pipeline_sql_worker "$apps" "$concurrency" "$tmpdir" "" + + # Merge the per-app records into one TSV, prefixing each line with its app. + # The glob is sorted (see run_pipeline_workers), so the renderer sees apps + # in name order and the first header it meets is the first app's. awk + # rather than cat: the files have no trailing newline (printf '%s'), so cat + # would glue one app's last line to the next app's first. + # + # This is also where "no rows" is decided: an app whose query ran (it has + # a header) but returned no data row is counted as skipped unless -a/--all + # — but its header line is still fed to the renderer, so a zero-row app + # can name the merged table's columns and take part in the drift check + # like any other. The renderer prints an empty row for it only with -a. + local records="$tmpdir/.records" f app + : > "$records" + for f in "$tmpdir"/*; do + [[ -f "$f" ]] || continue + app="${f##*/}" + if [[ "$include_empty" != "true" ]] \ + && grep -q $'^H\t' "$f" && ! grep -q $'^R\t' "$f"; then + : > "$tmpdir/.skipped/$app" + fi + awk -v app="$app" '{ print app "\t" $0 }' "$f" >> "$records" + done + + render_pipeline_sql "$render" "$include_empty" "$records" + + report_skipped "$tmpdir" "app(s) with no rows skipped (use -a/--all to include them)" +} + cmd_pipeline_task() { [[ $# -ge 3 ]] || die "Usage: $SCRIPT_NAME pipeline-task [--concurrency=N]" local pipeline="$1" stage="$2" task="$3" @@ -1002,6 +1436,9 @@ main() { config-replace) cmd_config_replace "$@" ;; + pipeline-sql) + cmd_pipeline_sql "$@" + ;; pipeline-task) cmd_pipeline_task "$@" ;; diff --git a/test/heroku-scripts.bats b/test/heroku-scripts.bats index 9495762..3ebe267 100644 --- a/test/heroku-scripts.bats +++ b/test/heroku-scripts.bats @@ -692,3 +692,306 @@ STUB [ "$status" -eq 1 ] [[ "$output" == *"1Password CLI"* ]] } + +# --------------------------------------------------------------------------- +# pipeline-sql +# --------------------------------------------------------------------------- + +# heroku stub for pipeline-sql. pipelines:info lists several stages so a test +# picks its app mix by stage name. pg:psql insists on the exact call shape +# (`pg:psql -f -a `), logs its argv to ./psql-calls, copies the +# script it was handed to ./psql-file- (so tests can assert on the file +# contents), and prints the sentinel by reading it out of the script's `\echo` +# line — so tests never hardcode it. Streams are split like the real CLI's: +# the query result (and psqlrc chatter) on stdout; the update banner, the +# "--> Connecting to ..." line, psql's NOTICE/ERROR lines and heroku's own +# errors on stderr. +_heroku_stub_pg_psql() { + cat > "$TESTDIR/bin/heroku" <<'STUB' +#!/usr/bin/env bash +if [[ "$1" == "pipelines:info" ]]; then + printf '=== %s\n' "$2" + printf 'app-two basic\napp-one basic\n' + printf 'app-one mixed\napp-empty mixed\napp-err mixed\n' + printf 'app-noisy noise\n' + printf 'app-one drift\napp-drift drift\n' + printf 'app-nodb nodb\n' + printf 'app-notice notice\n' + printf 'app-banner banner\n' + printf 'app-nullrow nullrow\n' + printf 'app-empty empties\napp-empty2 empties\n' + exit 0 +fi +if [[ "$1" != "pg:psql" || "$2" != "-f" || "$4" != "-a" || $# -ne 5 ]]; then + echo "unexpected call: $*" >&2 + exit 1 +fi +file="$3"; app="$5" +# Build the whole log line first and append it with ONE write: apps run in +# parallel, and several small writes per app can interleave in the file. +line='PSQL'; for a in "$@"; do line="$line [$a]"; done +printf '%s\n' "$line" >> ./psql-calls +cp "$file" "./psql-file-$app" +sentinel="$(sed -n 's/^\\echo //p' "$file")" +echo " › Warning: heroku update available from 8.0.0 to 9.0.0." >&2 +echo "--> Connecting to postgresql-curved-12345" >&2 +case "$app" in + app-one) printf '%s\nid\tname\n1\talice\n2\tbob\n' "$sentinel";; + app-two) printf '%s\nid\tname\n3\tcarol\n' "$sentinel";; + app-empty) printf '%s\nid\tname\n' "$sentinel";; + app-empty2) printf '%s\nid\tname\n' "$sentinel";; + # psqlrc chatter lands on stdout BEFORE the sentinel. + app-noisy) printf 'Timing is on.\nNull display is "(null)".\n%s\nid\tname\n7\tzed\n' "$sentinel";; + app-drift) printf '%s\nid\temail\n9\tx@y.z\n' "$sentinel";; + # A successful query that also raised a NOTICE (on stderr). + app-notice) + echo "psql:$file:15: NOTICE: identifier will be truncated" >&2 + printf '%s\nid\tname\n5\teve\n' "$sentinel";; + # A data value that looks exactly like heroku's connecting banner. + app-banner) printf '%s\nid\tnote\n6\t--> Connecting to postgresql-curved-12345\n' "$sentinel";; + # `select null as x`: a header and one row whose only cell is empty. + app-nullrow) printf '%s\nx\n\n' "$sentinel";; + # A SQL error: psql prefixes the first line with the -f file and line + # number, then heroku adds its own exit trailer — all on stderr. + app-err) + printf '%s\n' "$sentinel" + printf 'psql:%s:14: ERROR: relation "users" does not exist\nLINE 1: select * from users\n ^\n › Error: psql exited with code 3\n' "$file" >&2 + exit 1;; + # heroku fails before psql ever runs: no sentinel at all. + app-nodb) echo " › Error: No database found for app-nodb" >&2; exit 1;; +esac +STUB + chmod +x "$TESTDIR/bin/heroku" +} + +@test "pipeline-sql merges every app's rows under one header, sorted by app" { + _heroku_stub_pg_psql + run --separate-stderr "$SCRIPT" pipeline-sql mypipe basic "select id, name from users" + [ "$status" -eq 0 ] + [ "${lines[0]}" = "appname;id;name" ] + [ "${lines[1]}" = "app-one;1;alice" ] + [ "${lines[2]}" = "app-one;2;bob" ] + [ "${lines[3]}" = "app-two;3;carol" ] + [ "${#lines[@]}" -eq 4 ] + [ -z "$stderr" ] +} + +@test "pipeline-sql calls pg:psql -f -a once per app" { + _heroku_stub_pg_psql + run "$SCRIPT" pipeline-sql mypipe basic "select 1" + [ "$status" -eq 0 ] + [ "$(wc -l < psql-calls | tr -d ' ')" = "2" ] + grep -qE '^PSQL \[pg:psql\] \[-f\] \[[^]]+/\.[^]/]+\] \[-a\] \[app-one\]$' psql-calls + grep -qE '^PSQL \[pg:psql\] \[-f\] \[[^]]+/\.[^]/]+\] \[-a\] \[app-two\]$' psql-calls +} + +@test "pipeline-sql writes the inline SQL and the psql settings into the -f file" { + _heroku_stub_pg_psql + run "$SCRIPT" pipeline-sql mypipe basic "select count(*) from users where name = 'o''brien'" + [ "$status" -eq 0 ] + grep -qxF "select count(*) from users where name = 'o''brien'" psql-file-app-one + grep -qxF '\set ON_ERROR_STOP on' psql-file-app-one + grep -qxF '\set QUIET on' psql-file-app-one + grep -qxF '\pset format unaligned' psql-file-app-one + grep -qxF "\\pset fieldsep '\\t'" psql-file-app-one + grep -qxF "\\pset null ''" psql-file-app-one + # The sentinel echo comes after every setting and before the SQL. + awk '/^\\echo / { echo = NR } /^select count/ { sql = NR } /^\\pset null/ { last = NR } + END { exit !(last < echo && echo < sql) }' psql-file-app-one + # Both apps got the identical script. + cmp -s psql-file-app-one psql-file-app-two +} + +@test "pipeline-sql --file reads the SQL from a file" { + _heroku_stub_pg_psql + printf 'select id,\n name\nfrom users\n' > query.sql + run "$SCRIPT" pipeline-sql mypipe basic --file=query.sql + [ "$status" -eq 0 ] + [ "${lines[1]}" = "app-one;1;alice" ] + grep -qxF 'select id,' psql-file-app-one + grep -qxF ' name' psql-file-app-one + grep -qxF 'from users' psql-file-app-one +} + +@test "pipeline-sql --table renders one aligned table with error rows spanning the columns" { + _heroku_stub_pg_psql + run --separate-stderr "$SCRIPT" pipeline-sql mypipe mixed "select id, name from users" --table -a + [ "$status" -eq 0 ] + # Widths: app column 9 (app-empty), id 2, name 5 (alice). + [ "${lines[0]}" = "appname | id | name" ] + [ "${lines[1]}" = "----------+----+------" ] + [ "${lines[2]}" = "app-empty | |" ] + [ "${lines[3]}" = 'app-err | ERROR: relation "users" does not exist' ] + [ "${lines[4]}" = " | LINE 1: select * from users" ] + [ "${lines[5]}" = " | ^" ] + [ "${lines[6]}" = "app-one | 1 | alice" ] + [ "${lines[7]}" = "app-one | 2 | bob" ] + [ "${#lines[@]}" -eq 8 ] + # No trailing whitespace anywhere. + ! grep -q '[[:space:]]$' <<< "$output" +} + +@test "pipeline-sql turns a failed query into an error record with the psql prefix and heroku trailer removed" { + _heroku_stub_pg_psql + run --separate-stderr "$SCRIPT" pipeline-sql mypipe mixed "select * from users" --csv + [ "$status" -eq 0 ] + [ "${lines[1]}" = 'app-err;ERROR: relation "users" does not exist' ] + [ "${lines[2]}" = "LINE 1: select * from users" ] + [ "${lines[3]}" = " ^" ] + [[ "$output" != *"psql:"* ]] + [[ "$output" != *"psql exited"* ]] + # The error did not stop the other apps from being reported. + [[ "$output" == *"app-one;1;alice"* ]] +} + +@test "pipeline-sql reports heroku failing before psql ran (no sentinel) as an error record" { + _heroku_stub_pg_psql + run --separate-stderr "$SCRIPT" pipeline-sql mypipe nodb "select 1" --csv + [ "$status" -eq 0 ] + # No app produced a header, so the fallback header is used. + [ "${lines[0]}" = "appname;output" ] + [ "${lines[1]}" = "app-nodb;Error: No database found for app-nodb" ] + [[ "$output" != *"Connecting to"* ]] + [[ "$output" != *"update available"* ]] +} + +@test "pipeline-sql skips apps with no rows and reports a count on stderr" { + _heroku_stub_pg_psql + "$SCRIPT" pipeline-sql mypipe mixed "select id, name from users" --csv >stdout.txt 2>stderr.txt + ! grep -q "app-empty" stdout.txt + grep -q "app-one;1;alice" stdout.txt + # A blank line separates the output from the summary; no ANSI styling when + # stderr is not a terminal. + [ -z "$(head -n 1 stderr.txt)" ] + grep -q "1 app(s) with no rows skipped" stderr.txt + grep -q -- "-a/--all" stderr.txt + ! grep -qF $'\033' stderr.txt +} + +@test "pipeline-sql -a includes a no-rows app as one empty row and prints no skip summary" { + _heroku_stub_pg_psql + run --separate-stderr "$SCRIPT" pipeline-sql mypipe mixed "select id, name from users" --csv -a + [ "$status" -eq 0 ] + [ "${lines[1]}" = "app-empty;;" ] + [[ "$stderr" != *"skipped"* ]] +} + +@test "pipeline-sql discards psqlrc output printed before the sentinel" { + _heroku_stub_pg_psql + run --separate-stderr "$SCRIPT" pipeline-sql mypipe noise "select id, name from users" + [ "$status" -eq 0 ] + [ "${lines[0]}" = "appname;id;name" ] + [ "${lines[1]}" = "app-noisy;7;zed" ] + [ "${#lines[@]}" -eq 2 ] + [[ "$output" != *"Timing"* ]] + [[ "$output" != *"Null display"* ]] +} + +@test "pipeline-sql warns on stderr when an app's columns differ but still emits its rows" { + _heroku_stub_pg_psql + run --separate-stderr "$SCRIPT" pipeline-sql mypipe drift "select id, name from users" + [ "$status" -eq 0 ] + # app-drift sorts first, so its header wins. + [ "${lines[0]}" = "appname;id;email" ] + [ "${lines[1]}" = "app-drift;9;x@y.z" ] + [ "${lines[2]}" = "app-one;1;alice" ] + [ "${lines[3]}" = "app-one;2;bob" ] + [ "$stderr" = "pipeline-sql: app-one: columns differ from app-drift (id, name vs id, email)" ] +} + +@test "pipeline-sql accepts SQL that starts with a -- comment after an end-of-options --" { + _heroku_stub_pg_psql + run "$SCRIPT" pipeline-sql mypipe basic --csv -- $'-- who is there\nselect 1' + [ "$status" -eq 0 ] + [ "${lines[0]}" = "appname;id;name" ] + [[ "$output" == *"app-one;1;alice"* ]] + # The comment line and the statement both reach the psql script. + grep -qx -- '-- who is there' psql-file-app-one + grep -qx 'select 1' psql-file-app-one +} + +@test "pipeline-sql requires the SQL, inline or via --file" { + _heroku_stub_pg_psql + run "$SCRIPT" pipeline-sql mypipe basic + [ "$status" -eq 1 ] + [[ "$output" == *"No SQL given"* ]] + [[ "$output" == *"Usage:"* ]] + [ ! -e psql-calls ] +} + +@test "pipeline-sql rejects inline SQL combined with --file" { + _heroku_stub_pg_psql + printf 'select 1\n' > query.sql + run "$SCRIPT" pipeline-sql mypipe basic "select 1" --file=query.sql + [ "$status" -eq 1 ] + [[ "$output" == *"not both"* ]] + [[ "$output" == *"Usage:"* ]] + [ ! -e psql-calls ] +} + +@test "pipeline-sql fails clearly on an unreadable --file" { + _heroku_stub_pg_psql + run "$SCRIPT" pipeline-sql mypipe basic --file=does-not-exist.sql + [ "$status" -eq 1 ] + [[ "$output" == *"Cannot read SQL file: does-not-exist.sql"* ]] + [ ! -e psql-calls ] +} + +@test "pipeline-sql rejects an unknown option" { + _heroku_stub_pg_psql + run "$SCRIPT" pipeline-sql mypipe basic "select 1" --no-stream + [ "$status" -eq 1 ] + [[ "$output" == *"Unknown option: --no-stream"* ]] + [ ! -e psql-calls ] +} + +@test "pipeline-sql rejects a non-positive concurrency" { + _heroku_stub_pg_psql + run "$SCRIPT" pipeline-sql mypipe basic "select 1" --concurrency=0 + [ "$status" -eq 1 ] + [[ "$output" == *"positive integer"* ]] +} + +@test "pipeline-sql forwards a NOTICE on stderr with the app name instead of treating it as data" { + _heroku_stub_pg_psql + run --separate-stderr "$SCRIPT" pipeline-sql mypipe notice "select id, name from users" + [ "$status" -eq 0 ] + [ "${lines[0]}" = "appname;id;name" ] + [ "${lines[1]}" = "app-notice;5;eve" ] + [ "${#lines[@]}" -eq 2 ] + [ "$stderr" = "pipeline-sql: app-notice: NOTICE: identifier will be truncated" ] +} + +@test "pipeline-sql keeps a result value that looks like heroku's connecting banner" { + _heroku_stub_pg_psql + run --separate-stderr "$SCRIPT" pipeline-sql mypipe banner "select id, note from notes" + [ "$status" -eq 0 ] + [ "${lines[1]}" = "app-banner;6;--> Connecting to postgresql-curved-12345" ] + [ "${#lines[@]}" -eq 2 ] +} + +@test "pipeline-sql keeps a trailing row whose only cell is empty" { + _heroku_stub_pg_psql + run --separate-stderr "$SCRIPT" pipeline-sql mypipe nullrow "select null as x" + [ "$status" -eq 0 ] + [ "${lines[0]}" = "appname;x" ] + [ "${lines[1]}" = "app-nullrow;" ] + [ "${#lines[@]}" -eq 2 ] + # It is a real row, not a "no rows" app. + [[ "$stderr" != *"skipped"* ]] +} + +@test "pipeline-sql -a treats an empty-celled row as data, not as a no-rows placeholder" { + _heroku_stub_pg_psql + run --separate-stderr "$SCRIPT" pipeline-sql mypipe nullrow "select null as x" -a + [ "$status" -eq 0 ] + [ "${lines[1]}" = "app-nullrow;" ] + [ "${#lines[@]}" -eq 2 ] +} + +@test "pipeline-sql keeps the header when every app returns zero rows" { + _heroku_stub_pg_psql + "$SCRIPT" pipeline-sql mypipe empties "select id, name from users where false" --csv >stdout.txt 2>stderr.txt + [ "$(cat stdout.txt)" = "appname;id;name" ] + grep -q "2 app(s) with no rows skipped" stderr.txt +}