feat(ga4): Google Analytics integration with delta reconciliation #911
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: Ingest Rust Tests | |
| on: | |
| push: | |
| branches: [main] | |
| paths: | |
| - "apps/ingest/**" | |
| - "packages/db/drizzle/**" | |
| - "packages/db/src/schema/**" | |
| - ".github/scripts/format-ingest-benchmark-comment.py" | |
| - ".github/workflows/ingest-rust-tests.yml" | |
| pull_request: | |
| paths: | |
| - "apps/ingest/**" | |
| - "packages/db/drizzle/**" | |
| - "packages/db/src/schema/**" | |
| - ".github/scripts/format-ingest-benchmark-comment.py" | |
| - ".github/workflows/ingest-rust-tests.yml" | |
| workflow_dispatch: | |
| concurrency: | |
| group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} | |
| cancel-in-progress: true | |
| permissions: | |
| contents: read | |
| pull-requests: write | |
| env: | |
| CARGO_TERM_COLOR: never | |
| LOAD_TEST_REQUESTS: "2000" | |
| LOAD_TEST_CONCURRENCY: "64" | |
| LOAD_TEST_BATCH_LOGS: "10" | |
| LOAD_BENCH_ITERATIONS: "3" | |
| HEAD_INGEST_MODE: tinybird | |
| jobs: | |
| test: | |
| name: Cargo test and load benchmark | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 45 | |
| defaults: | |
| run: | |
| working-directory: apps/ingest | |
| env: | |
| # The perf comparison — head bench, Criterion, the base-branch release | |
| # build, its probe and its bench — is opt-in. It was ~60% of this job's | |
| # runtime on every ingest PR, and this job was the single most expensive | |
| # workflow in the repo. Label a PR `bench-ingest` when you are actually | |
| # changing the hot path; `cargo test` and the release build still run | |
| # unconditionally. Same opt-in shape as eval.yml's `run-evals`. | |
| RUN_BENCH: ${{ github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'bench-ingest') }} | |
| steps: | |
| - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v6 | |
| # Installs the rust version pinned in the root mise.toml (was a floating | |
| # `stable`). Keep rust-cache below — mise caches the toolchain, not the | |
| # cargo registry / target dir. | |
| - uses: jdx/mise-action@c2a87611a18de5b3828c5652fe268e992400cb5c # v4.3.0 | |
| - uses: Swatinem/rust-cache@f0d9c3887740aee45f6153b24b3a6b815192ec16 # v2 | |
| with: | |
| workspaces: apps/ingest | |
| - name: Run ingest unit and fake Tinybird e2e tests | |
| run: | | |
| set +e | |
| set -o pipefail | |
| cargo test --locked 2>&1 | tee "$RUNNER_TEMP/ingest-cargo-test.txt" | |
| echo "${PIPESTATUS[0]}" > "$RUNNER_TEMP/ingest-cargo-test.exit" | |
| exit 0 | |
| - name: Build PR ingest and benchmark binaries | |
| run: | | |
| set +e | |
| # load_test only exists to drive the benchmark, so only build it | |
| # when the benchmark is going to run. | |
| bins="--bin maple-ingest" | |
| if [ "$RUN_BENCH" = "true" ]; then | |
| bins="$bins --bin load_test" | |
| fi | |
| # shellcheck disable=SC2086 | |
| cargo build --locked --release $bins 2>&1 | tee "$RUNNER_TEMP/ingest-head-build.txt" | |
| echo "${PIPESTATUS[0]}" > "$RUNNER_TEMP/ingest-head-build.exit" | |
| mkdir -p "$RUNNER_TEMP/head" | |
| if [ -x target/release/maple-ingest ]; then | |
| cp target/release/maple-ingest "$RUNNER_TEMP/head/maple-ingest" | |
| fi | |
| if [ -x target/release/load_test ]; then | |
| cp target/release/load_test "$RUNNER_TEMP/head/load_test" | |
| fi | |
| exit 0 | |
| # Every step above records its exit code and returns 0 so the run reaches | |
| # the aggregation step and reports test and build failures together. That | |
| # is worth doing for the cheap steps; it is not worth spending the | |
| # benchmark budget on a tree that does not compile. | |
| - name: Skip benchmarks if tests or build failed | |
| id: bench-gate | |
| if: ${{ env.RUN_BENCH == 'true' }} | |
| run: | | |
| for file in "$RUNNER_TEMP"/ingest-cargo-test.exit "$RUNNER_TEMP"/ingest-head-build.exit; do | |
| [ -f "$file" ] || continue | |
| if [ "$(cat "$file")" != "0" ]; then | |
| echo "::notice::skipping benchmarks — $(basename "$file") is non-zero" | |
| echo "ok=false" >> "$GITHUB_OUTPUT" | |
| exit 0 | |
| fi | |
| done | |
| echo "ok=true" >> "$GITHUB_OUTPUT" | |
| - name: Run PR ingest load benchmark (median of N) | |
| if: ${{ steps.bench-gate.outputs.ok == 'true' }} | |
| run: | | |
| set +e | |
| set -o pipefail | |
| rm -f "$RUNNER_TEMP"/ingest-head-load-*.json | |
| iterations="${LOAD_BENCH_ITERATIONS:-3}" | |
| overall=0 | |
| for i in $(seq 1 "$iterations"); do | |
| report="$RUNNER_TEMP/ingest-head-load-$i.json" | |
| queue_dir="$RUNNER_TEMP/head-wal-$i" | |
| LOAD_TEST_INGEST_BIN="$RUNNER_TEMP/head/maple-ingest" \ | |
| LOAD_TEST_INGEST_MODE="$HEAD_INGEST_MODE" \ | |
| LOAD_TEST_REQUESTS="$LOAD_TEST_REQUESTS" \ | |
| LOAD_TEST_CONCURRENCY="$LOAD_TEST_CONCURRENCY" \ | |
| LOAD_TEST_BATCH_LOGS="$LOAD_TEST_BATCH_LOGS" \ | |
| LOAD_TEST_INGEST_PORT="$((3475 + i))" \ | |
| LOAD_TEST_QUEUE_DIR="$queue_dir" \ | |
| LOAD_TEST_REPORT_PATH="$report" \ | |
| "$RUNNER_TEMP/head/load_test" 2>&1 \ | |
| | tee "$RUNNER_TEMP/ingest-head-load-$i.log" | |
| code="${PIPESTATUS[0]}" | |
| if [ "$code" != "0" ]; then | |
| overall="$code" | |
| fi | |
| done | |
| echo "$overall" > "$RUNNER_TEMP/ingest-head-load.exit" | |
| exit 0 | |
| - name: Run Criterion WAL-ack microbench | |
| if: ${{ steps.bench-gate.outputs.ok == 'true' }} | |
| run: | | |
| set +e | |
| set -o pipefail | |
| cargo bench --bench ingest_bench -- --output-format=bencher 2>&1 \ | |
| | tee "$RUNNER_TEMP/ingest-criterion.txt" | |
| echo "${PIPESTATUS[0]}" > "$RUNNER_TEMP/ingest-criterion.exit" | |
| exit 0 | |
| - name: Build base ingest binary | |
| if: ${{ steps.bench-gate.outputs.ok == 'true' && github.event_name == 'pull_request' }} | |
| run: | | |
| set +e | |
| git fetch origin "${{ github.base_ref }}" --depth=1 | |
| git worktree add "$RUNNER_TEMP/base" FETCH_HEAD | |
| # Only build maple-ingest; load_test may not exist on the base | |
| # branch (it's added by the PR that introduces the harness). | |
| # The probe + benchmark steps always use the head's load_test | |
| # binary pointed at the base's maple-ingest. | |
| # | |
| # Build into the head's target dir rather than the worktree's own. | |
| # Swatinem/rust-cache only covers `workspaces: apps/ingest`, so a | |
| # worktree-local target/ under $RUNNER_TEMP was a full cold compile | |
| # of every dependency on every PR. Sharing the dir reuses those | |
| # cached dep artifacts and recompiles only what actually differs. | |
| # Safe because the head binaries were already copied to | |
| # $RUNNER_TEMP/head above, so overwriting target/release is fine. | |
| ( | |
| cd "$RUNNER_TEMP/base/apps/ingest" | |
| CARGO_TARGET_DIR="$GITHUB_WORKSPACE/apps/ingest/target" \ | |
| cargo build --locked --release --bin maple-ingest | |
| ) 2>&1 | tee "$RUNNER_TEMP/ingest-base-build.txt" | |
| echo "${PIPESTATUS[0]}" > "$RUNNER_TEMP/ingest-base-build.exit" | |
| mkdir -p "$RUNNER_TEMP/base-bin" | |
| if [ -x "$GITHUB_WORKSPACE/apps/ingest/target/release/maple-ingest" ]; then | |
| cp "$GITHUB_WORKSPACE/apps/ingest/target/release/maple-ingest" "$RUNNER_TEMP/base-bin/maple-ingest" | |
| fi | |
| exit 0 | |
| - name: Probe whether base supports head ingest mode | |
| if: ${{ steps.bench-gate.outputs.ok == 'true' && github.event_name == 'pull_request' }} | |
| run: | | |
| set +e | |
| echo "0" > "$RUNNER_TEMP/base-supports-mode" | |
| if [ ! -x "$RUNNER_TEMP/base-bin/maple-ingest" ] || [ ! -x "$RUNNER_TEMP/head/load_test" ]; then | |
| echo "base binaries missing — skipping probe" | |
| exit 0 | |
| fi | |
| probe_dir="$RUNNER_TEMP/base-probe-wal" | |
| rm -rf "$probe_dir" | |
| LOAD_TEST_INGEST_BIN="$RUNNER_TEMP/base-bin/maple-ingest" \ | |
| LOAD_TEST_INGEST_MODE="$HEAD_INGEST_MODE" \ | |
| LOAD_TEST_REQUESTS=10 \ | |
| LOAD_TEST_CONCURRENCY=4 \ | |
| LOAD_TEST_BATCH_LOGS=1 \ | |
| LOAD_TEST_INGEST_PORT=3490 \ | |
| LOAD_TEST_QUEUE_DIR="$probe_dir" \ | |
| LOAD_TEST_REPORT_PATH="$RUNNER_TEMP/base-probe.json" \ | |
| timeout 60 "$RUNNER_TEMP/head/load_test" >/dev/null 2>&1 | |
| if [ -s "$RUNNER_TEMP/base-probe.json" ]; then | |
| echo "1" > "$RUNNER_TEMP/base-supports-mode" | |
| fi | |
| rm -rf "$probe_dir" | |
| exit 0 | |
| - name: Run base ingest load benchmark (median of N) | |
| if: ${{ steps.bench-gate.outputs.ok == 'true' && github.event_name == 'pull_request' }} | |
| run: | | |
| set +e | |
| set -o pipefail | |
| rm -f "$RUNNER_TEMP"/ingest-base-load-*.json | |
| if [ ! -f "$RUNNER_TEMP/base-supports-mode" ] || [ "$(cat "$RUNNER_TEMP/base-supports-mode")" != "1" ]; then | |
| echo "base does not support $HEAD_INGEST_MODE mode — skipping baseline benchmark" | |
| echo "0" > "$RUNNER_TEMP/ingest-base-load.exit" | |
| exit 0 | |
| fi | |
| iterations="${LOAD_BENCH_ITERATIONS:-3}" | |
| overall=0 | |
| for i in $(seq 1 "$iterations"); do | |
| report="$RUNNER_TEMP/ingest-base-load-$i.json" | |
| queue_dir="$RUNNER_TEMP/base-wal-$i" | |
| LOAD_TEST_INGEST_BIN="$RUNNER_TEMP/base-bin/maple-ingest" \ | |
| LOAD_TEST_INGEST_MODE="$HEAD_INGEST_MODE" \ | |
| LOAD_TEST_REQUESTS="$LOAD_TEST_REQUESTS" \ | |
| LOAD_TEST_CONCURRENCY="$LOAD_TEST_CONCURRENCY" \ | |
| LOAD_TEST_BATCH_LOGS="$LOAD_TEST_BATCH_LOGS" \ | |
| LOAD_TEST_INGEST_PORT="$((3500 + i))" \ | |
| LOAD_TEST_QUEUE_DIR="$queue_dir" \ | |
| LOAD_TEST_REPORT_PATH="$report" \ | |
| "$RUNNER_TEMP/head/load_test" 2>&1 \ | |
| | tee "$RUNNER_TEMP/ingest-base-load-$i.log" | |
| code="${PIPESTATUS[0]}" | |
| if [ "$code" != "0" ]; then | |
| overall="$code" | |
| fi | |
| done | |
| echo "$overall" > "$RUNNER_TEMP/ingest-base-load.exit" | |
| exit 0 | |
| - name: Render PR comment body | |
| if: ${{ always() && steps.bench-gate.outputs.ok == 'true' && github.event_name == 'pull_request' }} | |
| env: | |
| TEST_OUTPUT: ${{ runner.temp }}/ingest-cargo-test.txt | |
| CRITERION_OUTPUT: ${{ runner.temp }}/ingest-criterion.txt | |
| COMMENT_OUTPUT: ${{ runner.temp }}/ingest-results-comment.md | |
| run: | | |
| set -e | |
| head_json=$(ls "$RUNNER_TEMP"/ingest-head-load-*.json 2>/dev/null | paste -sd, -) | |
| base_json=$(ls "$RUNNER_TEMP"/ingest-base-load-*.json 2>/dev/null | paste -sd, -) | |
| HEAD_LOAD_JSON="$head_json" \ | |
| BASE_LOAD_JSON="$base_json" \ | |
| python "$GITHUB_WORKSPACE/.github/scripts/format-ingest-benchmark-comment.py" | |
| - name: Post or update PR comment (sticky) | |
| if: ${{ always() && steps.bench-gate.outputs.ok == 'true' && github.event_name == 'pull_request' }} | |
| env: | |
| GH_TOKEN: ${{ github.token }} | |
| PR_NUMBER: ${{ github.event.pull_request.number }} | |
| COMMENT_OUTPUT: ${{ runner.temp }}/ingest-results-comment.md | |
| REPO: ${{ github.repository }} | |
| run: | | |
| set -e | |
| if [ ! -s "$COMMENT_OUTPUT" ]; then | |
| echo "no comment body rendered, skipping" | |
| exit 0 | |
| fi | |
| existing=$(gh api "repos/$REPO/issues/$PR_NUMBER/comments" --paginate \ | |
| --jq '[.[] | select(.body | startswith("<!-- ingest-bench-comment -->"))][0].id') | |
| if [ -n "$existing" ] && [ "$existing" != "null" ]; then | |
| payload="$RUNNER_TEMP/ingest-comment-payload.json" | |
| jq -Rs '{body: .}' < "$COMMENT_OUTPUT" > "$payload" | |
| gh api -X PATCH "repos/$REPO/issues/comments/$existing" \ | |
| --input "$payload" >/dev/null | |
| echo "updated existing comment $existing" | |
| else | |
| gh pr comment "$PR_NUMBER" --body-file "$COMMENT_OUTPUT" | |
| fi | |
| - name: Fail if ingest checks failed | |
| if: ${{ always() }} | |
| run: | | |
| failed=0 | |
| for file in "$RUNNER_TEMP"/ingest-*.exit; do | |
| [ -f "$file" ] || continue | |
| code=$(cat "$file") | |
| if [ "$code" != "0" ]; then | |
| echo "::error file=$file::step failed with exit code $code" | |
| failed=1 | |
| fi | |
| done | |
| exit "$failed" |