From 3faf078581da6e3e8cbcb890c450f468b7033c69 Mon Sep 17 00:00:00 2001 From: Vlad Scherbich Date: Fri, 25 Sep 2026 14:50:46 -0400 Subject: [PATCH 1/7] chore(ci): raise S3 wheel poll timeout to 120m for prof-correctness GitLab install.sh uploads have reached ~78m; the previous 63m poll still flakes trigger-downstream before wheels land. --- .github/workflows/prof-correctness.yml | 21 +++++++++++++-------- 1 file changed, 13 insertions(+), 8 deletions(-) diff --git a/.github/workflows/prof-correctness.yml b/.github/workflows/prof-correctness.yml index 52f362e3bb0..ac93e4a8a01 100644 --- a/.github/workflows/prof-correctness.yml +++ b/.github/workflows/prof-correctness.yml @@ -29,11 +29,11 @@ permissions: jobs: trigger-downstream: runs-on: ubuntu-latest - # 60m S3 poll + 60m downstream watch = 120, plus ~10m slack + # 120m S3 poll + 60m downstream watch = 180, plus ~10m slack # for octo-sts / trigger / GHA overhead. # Job timeout must stay strictly above POLL_TIMEOUT so a late upload # does not consume the entire budget before downstream-python starts. - timeout-minutes: 130 + timeout-minutes: 190 steps: - name: Resolve commit SHA id: sha @@ -71,8 +71,8 @@ jobs: echo "run=true" >> "$GITHUB_OUTPUT" fi - # GitLab uploads install.sh after wheel build (often 20–63+ min under - # load). This workflow uses its own 60m poll budget for late GitLab + # GitLab uploads install.sh after wheel build (often 20–78+ min under + # load). This workflow uses its own 120m poll budget for late GitLab # uploads (independent of scripts/download-s3-wheels.sh). Poll before # dispatch so downstream does not 404. - name: Wait for S3 install.sh @@ -81,20 +81,25 @@ jobs: DD_TRACE_PY_SHA: ${{ steps.sha.outputs.value }} run: | INSTALL_URL="https://dd-trace-py-builds.s3.amazonaws.com/${DD_TRACE_PY_SHA}/install.sh" - POLL_TIMEOUT="${POLL_TIMEOUT:-3800}" + POLL_TIMEOUT="${POLL_TIMEOUT:-7200}" POLL_INTERVAL="${POLL_INTERVAL:-30}" echo "Polling for: ${INSTALL_URL}" + echo "Timeout: ${POLL_TIMEOUT}s, Interval: ${POLL_INTERVAL}s" elapsed=0 + last_status="none" while true; do - if curl -sf -o /dev/null "${INSTALL_URL}"; then + last_status="$(curl -sS -o /dev/null -w '%{http_code}' "${INSTALL_URL}" || true)" + last_status="${last_status:-curl_error}" + if [ "$last_status" = "200" ]; then echo "install.sh found after ${elapsed}s" break fi if [ "$elapsed" -ge "$POLL_TIMEOUT" ]; then - echo "error: timed out after ${POLL_TIMEOUT}s waiting for ${INSTALL_URL}" >&2 + echo "error: timed out after ${POLL_TIMEOUT}s waiting for ${INSTALL_URL} (last HTTP ${last_status})" >&2 + echo "GitLab may still be publishing wheels. Re-run this job after ${INSTALL_URL} returns 200." >&2 exit 1 fi - echo "Not available yet (${elapsed}s elapsed), retrying in ${POLL_INTERVAL}s..." + echo "Not available yet (${elapsed}s elapsed, HTTP ${last_status}), retrying in ${POLL_INTERVAL}s..." sleep "${POLL_INTERVAL}" elapsed=$((elapsed + POLL_INTERVAL)) done From 2757eda636d4c21c3fa5336617bd7be3fdf85161 Mon Sep 17 00:00:00 2001 From: Vlad Scherbich Date: Fri, 25 Sep 2026 15:13:50 -0400 Subject: [PATCH 2/7] chore(ci): cap S3 wheel poll at 90m for prof-correctness A 120m poll doubles the budget again after a 14m miss. 90m covers the observed ~78m GitLab upload plus slack without parking a runner for two hours. --- .github/workflows/prof-correctness.yml | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/.github/workflows/prof-correctness.yml b/.github/workflows/prof-correctness.yml index ac93e4a8a01..f0d31be469e 100644 --- a/.github/workflows/prof-correctness.yml +++ b/.github/workflows/prof-correctness.yml @@ -29,11 +29,11 @@ permissions: jobs: trigger-downstream: runs-on: ubuntu-latest - # 120m S3 poll + 60m downstream watch = 180, plus ~10m slack + # 90m S3 poll + 60m downstream watch = 150, plus ~10m slack # for octo-sts / trigger / GHA overhead. # Job timeout must stay strictly above POLL_TIMEOUT so a late upload # does not consume the entire budget before downstream-python starts. - timeout-minutes: 190 + timeout-minutes: 160 steps: - name: Resolve commit SHA id: sha @@ -72,7 +72,7 @@ jobs: fi # GitLab uploads install.sh after wheel build (often 20–78+ min under - # load). This workflow uses its own 120m poll budget for late GitLab + # load). This workflow uses its own 90m poll budget for late GitLab # uploads (independent of scripts/download-s3-wheels.sh). Poll before # dispatch so downstream does not 404. - name: Wait for S3 install.sh @@ -81,7 +81,7 @@ jobs: DD_TRACE_PY_SHA: ${{ steps.sha.outputs.value }} run: | INSTALL_URL="https://dd-trace-py-builds.s3.amazonaws.com/${DD_TRACE_PY_SHA}/install.sh" - POLL_TIMEOUT="${POLL_TIMEOUT:-7200}" + POLL_TIMEOUT="${POLL_TIMEOUT:-5400}" POLL_INTERVAL="${POLL_INTERVAL:-30}" echo "Polling for: ${INSTALL_URL}" echo "Timeout: ${POLL_TIMEOUT}s, Interval: ${POLL_INTERVAL}s" From b3dc993ed2a6c7a5b95f9bdbac3439ff7ee11ea8 Mon Sep 17 00:00:00 2001 From: Vlad Scherbich Date: Sat, 26 Sep 2026 09:49:59 -0400 Subject: [PATCH 3/7] chore(ci): poll earliest S3 install script for prof-correctness Unsuffixed install.sh is only after upload all, so a 90m wait still 404s while install-manylinux2014_x86_64.sh is already up. Match dd-trace-doe and pass the first 200 to downstream. --- .github/workflows/prof-correctness.yml | 50 +++++++++++++++++--------- 1 file changed, 33 insertions(+), 17 deletions(-) diff --git a/.github/workflows/prof-correctness.yml b/.github/workflows/prof-correctness.yml index f0d31be469e..d88e1e44aa7 100644 --- a/.github/workflows/prof-correctness.yml +++ b/.github/workflows/prof-correctness.yml @@ -71,35 +71,49 @@ jobs: echo "run=true" >> "$GITHUB_OUTPUT" fi - # GitLab uploads install.sh after wheel build (often 20–78+ min under - # load). This workflow uses its own 90m poll budget for late GitLab - # uploads (independent of scripts/download-s3-wheels.sh). Poll before - # dispatch so downstream does not 404. - - name: Wait for S3 install.sh + # GitLab publishes per-platform indexes first + # (install-manylinux2014_x86_64.sh, then install-manylinux2014.sh). + # Unsuffixed install.sh is only after "upload all" (often 20–78+ min). + # Poll the same three as dd-trace-doe so we dispatch when linux amd64 + # wheels are up. 90m is the fallback if even the earliest index is late. + - name: Wait for S3 install script if: steps.gate.outputs.run == 'true' + id: s3 env: DD_TRACE_PY_SHA: ${{ steps.sha.outputs.value }} run: | - INSTALL_URL="https://dd-trace-py-builds.s3.amazonaws.com/${DD_TRACE_PY_SHA}/install.sh" + S3_BASE="https://dd-trace-py-builds.s3.amazonaws.com/${DD_TRACE_PY_SHA}" + # Same order as dd-trace-doe images/python/scripts/install_tracer.sh. + CANDIDATES="install-manylinux2014_x86_64.sh install-manylinux2014.sh install.sh" POLL_TIMEOUT="${POLL_TIMEOUT:-5400}" POLL_INTERVAL="${POLL_INTERVAL:-30}" - echo "Polling for: ${INSTALL_URL}" + echo "Polling ${S3_BASE} for: ${CANDIDATES}" echo "Timeout: ${POLL_TIMEOUT}s, Interval: ${POLL_INTERVAL}s" elapsed=0 - last_status="none" + found_url="" while true; do - last_status="$(curl -sS -o /dev/null -w '%{http_code}' "${INSTALL_URL}" || true)" - last_status="${last_status:-curl_error}" - if [ "$last_status" = "200" ]; then - echo "install.sh found after ${elapsed}s" + statuses="" + found_url="" + for name in ${CANDIDATES}; do + url="${S3_BASE}/${name}" + code="$(curl -sS -o /dev/null -w '%{http_code}' "${url}" || true)" + code="${code:-curl_error}" + statuses="${statuses}${statuses:+, }${name}=${code}" + if [ "$code" = "200" ] && [ -z "$found_url" ]; then + found_url="$url" + fi + done + if [ -n "$found_url" ]; then + echo "found ${found_url} after ${elapsed}s (${statuses})" + echo "url=${found_url}" >> "$GITHUB_OUTPUT" break fi if [ "$elapsed" -ge "$POLL_TIMEOUT" ]; then - echo "error: timed out after ${POLL_TIMEOUT}s waiting for ${INSTALL_URL} (last HTTP ${last_status})" >&2 - echo "GitLab may still be publishing wheels. Re-run this job after ${INSTALL_URL} returns 200." >&2 + echo "error: timed out after ${POLL_TIMEOUT}s waiting for any of: ${CANDIDATES} under ${S3_BASE} (last ${statuses})" >&2 + echo "GitLab may still be publishing wheels. Re-run this job after a candidate returns 200." >&2 exit 1 fi - echo "Not available yet (${elapsed}s elapsed, HTTP ${last_status}), retrying in ${POLL_INTERVAL}s..." + echo "Not available yet (${elapsed}s elapsed, ${statuses}), retrying in ${POLL_INTERVAL}s..." sleep "${POLL_INTERVAL}" elapsed=$((elapsed + POLL_INTERVAL)) done @@ -117,14 +131,16 @@ jobs: GH_TOKEN: ${{ steps.octo-sts.outputs.token }} DD_TRACE_PY_SHA: ${{ steps.sha.outputs.value }} TEST_SCENARIOS: ${{ github.event.inputs.test_scenarios || 'python_(mem_domain|exceptions|async_gen|lock)_3\.(14|15)' }} + INSTALL_URL: ${{ steps.s3.outputs.url }} run: | # downstream-python.yml run-name: "dd-trace-py downstream ()" TRIGGERED_AT="$(date -u -d "@$(($(date +%s) - 30))" +%Y-%m-%dT%H:%M:%SZ)" - echo "Triggering prof-correctness for commit ${DD_TRACE_PY_SHA}" + echo "Triggering prof-correctness for commit ${DD_TRACE_PY_SHA} via ${INSTALL_URL}" gh workflow run downstream-python.yml \ --repo DataDog/prof-correctness \ -f "dd_trace_py_commit_sha=${DD_TRACE_PY_SHA}" \ - -f "test_scenarios=${TEST_SCENARIOS}" + -f "test_scenarios=${TEST_SCENARIOS}" \ + -f "ddtrace_install_url=${INSTALL_URL}" RUN_ID="" for _ in $(seq 1 30); do From a6417ff116a7b9dbb3abde6bd7958ef364e4dbe0 Mon Sep 17 00:00:00 2001 From: Vlad Scherbich Date: Mon, 28 Sep 2026 15:02:08 -0400 Subject: [PATCH 4/7] chore(ci): own prof-correctness workflow by profiling-python Route CODEOWNERS reviews for .github/workflows/prof-correctness.yml to @DataDog/profiling-python, matching the other profiling GHA workflows. --- .github/CODEOWNERS | 1 + 1 file changed, 1 insertion(+) diff --git a/.github/CODEOWNERS b/.github/CODEOWNERS index e13d22ebe19..b95f94843a5 100644 --- a/.github/CODEOWNERS +++ b/.github/CODEOWNERS @@ -228,6 +228,7 @@ ddtrace/internal/settings/profiling.pyi @DataDog/profiling-python ddtrace/internal/datadog/profiling @DataDog/profiling-python tests/profiling @DataDog/profiling-python .github/workflows/profiling-native.yml @DataDog/profiling-python +.github/workflows/prof-correctness.yml @DataDog/profiling-python .gitlab/tests/profiling.yml @DataDog/profiling-python .github/workflows/pytorch_gpu_tests.yml @DataDog/profiling-python .github/PULL_REQUEST_TEMPLATE/profiler_change.md @DataDog/profiling-python From 6a4df4c623eafb1ec4b2e322dd317a7d13dd1ad3 Mon Sep 17 00:00:00 2001 From: Vlad Scherbich Date: Mon, 28 Sep 2026 15:08:34 -0400 Subject: [PATCH 5/7] chore(ci): bound S3 poll curls and stop on first hit Keep hung TCP probes from blowing past POLL_TIMEOUT, and skip remaining install-script candidates once one returns 200. --- .github/workflows/prof-correctness.yml | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/.github/workflows/prof-correctness.yml b/.github/workflows/prof-correctness.yml index d88e1e44aa7..5f0f3f1a3de 100644 --- a/.github/workflows/prof-correctness.yml +++ b/.github/workflows/prof-correctness.yml @@ -91,16 +91,21 @@ jobs: echo "Timeout: ${POLL_TIMEOUT}s, Interval: ${POLL_INTERVAL}s" elapsed=0 found_url="" + # Bound each probe so a hung S3/TCP stall cannot blow past POLL_TIMEOUT. + CURL_CONNECT_TIMEOUT="${CURL_CONNECT_TIMEOUT:-5}" + CURL_MAX_TIME="${CURL_MAX_TIME:-15}" while true; do statuses="" found_url="" for name in ${CANDIDATES}; do url="${S3_BASE}/${name}" - code="$(curl -sS -o /dev/null -w '%{http_code}' "${url}" || true)" + code="$(curl -sS --connect-timeout "${CURL_CONNECT_TIMEOUT}" --max-time "${CURL_MAX_TIME}" \ + -o /dev/null -w '%{http_code}' "${url}" || true)" code="${code:-curl_error}" statuses="${statuses}${statuses:+, }${name}=${code}" - if [ "$code" = "200" ] && [ -z "$found_url" ]; then + if [ "$code" = "200" ]; then found_url="$url" + break fi done if [ -n "$found_url" ]; then From e0365636e0aa4df3906c81e32aa7604e2ff07a9d Mon Sep 17 00:00:00 2001 From: Vlad Scherbich Date: Mon, 28 Sep 2026 15:42:11 -0400 Subject: [PATCH 6/7] chore(ci): use wall-clock SECONDS for S3 poll timeout Interval-only elapsed ignored curl probe time and could overrun the job timeout before dispatch. --- .github/workflows/prof-correctness.yml | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/.github/workflows/prof-correctness.yml b/.github/workflows/prof-correctness.yml index 5f0f3f1a3de..b5441302f5d 100644 --- a/.github/workflows/prof-correctness.yml +++ b/.github/workflows/prof-correctness.yml @@ -89,11 +89,13 @@ jobs: POLL_INTERVAL="${POLL_INTERVAL:-30}" echo "Polling ${S3_BASE} for: ${CANDIDATES}" echo "Timeout: ${POLL_TIMEOUT}s, Interval: ${POLL_INTERVAL}s" - elapsed=0 found_url="" # Bound each probe so a hung S3/TCP stall cannot blow past POLL_TIMEOUT. CURL_CONNECT_TIMEOUT="${CURL_CONNECT_TIMEOUT:-5}" CURL_MAX_TIME="${CURL_MAX_TIME:-15}" + # Wall-clock budget: elapsed += POLL_INTERVAL ignores curl probe time + # (up to 3 * CURL_MAX_TIME per round) and can overrun the job timeout. + SECONDS=0 while true; do statuses="" found_url="" @@ -109,18 +111,17 @@ jobs: fi done if [ -n "$found_url" ]; then - echo "found ${found_url} after ${elapsed}s (${statuses})" + echo "found ${found_url} after ${SECONDS}s (${statuses})" echo "url=${found_url}" >> "$GITHUB_OUTPUT" break fi - if [ "$elapsed" -ge "$POLL_TIMEOUT" ]; then + if [ "$SECONDS" -ge "$POLL_TIMEOUT" ]; then echo "error: timed out after ${POLL_TIMEOUT}s waiting for any of: ${CANDIDATES} under ${S3_BASE} (last ${statuses})" >&2 echo "GitLab may still be publishing wheels. Re-run this job after a candidate returns 200." >&2 exit 1 fi - echo "Not available yet (${elapsed}s elapsed, ${statuses}), retrying in ${POLL_INTERVAL}s..." + echo "Not available yet (${SECONDS}s elapsed, ${statuses}), retrying in ${POLL_INTERVAL}s..." sleep "${POLL_INTERVAL}" - elapsed=$((elapsed + POLL_INTERVAL)) done - uses: DataDog/dd-octo-sts-action@96a25462dbcb10ebf0bfd6e2ccc917d2ab235b9a # v1.0.4 From 61a0aa4067eae9a3063778db75c12210f220921c Mon Sep 17 00:00:00 2001 From: Vlad Scherbich Date: Wed, 30 Sep 2026 11:33:19 -0400 Subject: [PATCH 7/7] chore(ci): drop private dd-trace-doe refs from S3 poll comments MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Stand-alone publish-order rationale only — platform indexes land before unsuffixed install.sh. --- .github/workflows/prof-correctness.yml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/.github/workflows/prof-correctness.yml b/.github/workflows/prof-correctness.yml index b5441302f5d..e83e57dd1e7 100644 --- a/.github/workflows/prof-correctness.yml +++ b/.github/workflows/prof-correctness.yml @@ -74,8 +74,8 @@ jobs: # GitLab publishes per-platform indexes first # (install-manylinux2014_x86_64.sh, then install-manylinux2014.sh). # Unsuffixed install.sh is only after "upload all" (often 20–78+ min). - # Poll the same three as dd-trace-doe so we dispatch when linux amd64 - # wheels are up. 90m is the fallback if even the earliest index is late. + # Poll in that order so we dispatch when linux amd64 wheels are up. + # 90m is the fallback if even the earliest index is late. - name: Wait for S3 install script if: steps.gate.outputs.run == 'true' id: s3 @@ -83,7 +83,7 @@ jobs: DD_TRACE_PY_SHA: ${{ steps.sha.outputs.value }} run: | S3_BASE="https://dd-trace-py-builds.s3.amazonaws.com/${DD_TRACE_PY_SHA}" - # Same order as dd-trace-doe images/python/scripts/install_tracer.sh. + # Platform indexes land before unsuffixed install.sh. CANDIDATES="install-manylinux2014_x86_64.sh install-manylinux2014.sh install.sh" POLL_TIMEOUT="${POLL_TIMEOUT:-5400}" POLL_INTERVAL="${POLL_INTERVAL:-30}"