diff --git a/.github/CODEOWNERS b/.github/CODEOWNERS index 356bc6230db..5a9116c716f 100644 --- a/.github/CODEOWNERS +++ b/.github/CODEOWNERS @@ -227,6 +227,8 @@ ddtrace/internal/settings/profiling.py @DataDog/profiling-python ddtrace/internal/settings/profiling.pyi @DataDog/profiling-python ddtrace/internal/datadog/profiling @DataDog/profiling-python tests/profiling @DataDog/profiling-python +.github/workflows/profiling-native.yml @DataDog/profiling-python +.github/workflows/prof-correctness.yml @DataDog/profiling-python .gitlab/tests/profiling.yml @DataDog/profiling-python .github/workflows/pytorch_gpu_tests.yml @DataDog/profiling-python .github/PULL_REQUEST_TEMPLATE/profiler_change.md @DataDog/profiling-python diff --git a/.github/workflows/prof-correctness.yml b/.github/workflows/prof-correctness.yml index 52f362e3bb0..e83e57dd1e7 100644 --- a/.github/workflows/prof-correctness.yml +++ b/.github/workflows/prof-correctness.yml @@ -29,11 +29,11 @@ permissions: jobs: trigger-downstream: runs-on: ubuntu-latest - # 60m S3 poll + 60m downstream watch = 120, plus ~10m slack + # 90m S3 poll + 60m downstream watch = 150, plus ~10m slack # for octo-sts / trigger / GHA overhead. # Job timeout must stay strictly above POLL_TIMEOUT so a late upload # does not consume the entire budget before downstream-python starts. - timeout-minutes: 130 + timeout-minutes: 160 steps: - name: Resolve commit SHA id: sha @@ -71,32 +71,57 @@ jobs: echo "run=true" >> "$GITHUB_OUTPUT" fi - # GitLab uploads install.sh after wheel build (often 20–63+ min under - # load). This workflow uses its own 60m poll budget for late GitLab - # uploads (independent of scripts/download-s3-wheels.sh). Poll before - # dispatch so downstream does not 404. - - name: Wait for S3 install.sh + # GitLab publishes per-platform indexes first + # (install-manylinux2014_x86_64.sh, then install-manylinux2014.sh). + # Unsuffixed install.sh is only after "upload all" (often 20–78+ min). + # Poll in that order so we dispatch when linux amd64 wheels are up. + # 90m is the fallback if even the earliest index is late. + - name: Wait for S3 install script if: steps.gate.outputs.run == 'true' + id: s3 env: DD_TRACE_PY_SHA: ${{ steps.sha.outputs.value }} run: | - INSTALL_URL="https://dd-trace-py-builds.s3.amazonaws.com/${DD_TRACE_PY_SHA}/install.sh" - POLL_TIMEOUT="${POLL_TIMEOUT:-3800}" + S3_BASE="https://dd-trace-py-builds.s3.amazonaws.com/${DD_TRACE_PY_SHA}" + # Platform indexes land before unsuffixed install.sh. + CANDIDATES="install-manylinux2014_x86_64.sh install-manylinux2014.sh install.sh" + POLL_TIMEOUT="${POLL_TIMEOUT:-5400}" POLL_INTERVAL="${POLL_INTERVAL:-30}" - echo "Polling for: ${INSTALL_URL}" - elapsed=0 + echo "Polling ${S3_BASE} for: ${CANDIDATES}" + echo "Timeout: ${POLL_TIMEOUT}s, Interval: ${POLL_INTERVAL}s" + found_url="" + # Bound each probe so a hung S3/TCP stall cannot blow past POLL_TIMEOUT. + CURL_CONNECT_TIMEOUT="${CURL_CONNECT_TIMEOUT:-5}" + CURL_MAX_TIME="${CURL_MAX_TIME:-15}" + # Wall-clock budget: elapsed += POLL_INTERVAL ignores curl probe time + # (up to 3 * CURL_MAX_TIME per round) and can overrun the job timeout. + SECONDS=0 while true; do - if curl -sf -o /dev/null "${INSTALL_URL}"; then - echo "install.sh found after ${elapsed}s" + statuses="" + found_url="" + for name in ${CANDIDATES}; do + url="${S3_BASE}/${name}" + code="$(curl -sS --connect-timeout "${CURL_CONNECT_TIMEOUT}" --max-time "${CURL_MAX_TIME}" \ + -o /dev/null -w '%{http_code}' "${url}" || true)" + code="${code:-curl_error}" + statuses="${statuses}${statuses:+, }${name}=${code}" + if [ "$code" = "200" ]; then + found_url="$url" + break + fi + done + if [ -n "$found_url" ]; then + echo "found ${found_url} after ${SECONDS}s (${statuses})" + echo "url=${found_url}" >> "$GITHUB_OUTPUT" break fi - if [ "$elapsed" -ge "$POLL_TIMEOUT" ]; then - echo "error: timed out after ${POLL_TIMEOUT}s waiting for ${INSTALL_URL}" >&2 + if [ "$SECONDS" -ge "$POLL_TIMEOUT" ]; then + echo "error: timed out after ${POLL_TIMEOUT}s waiting for any of: ${CANDIDATES} under ${S3_BASE} (last ${statuses})" >&2 + echo "GitLab may still be publishing wheels. Re-run this job after a candidate returns 200." >&2 exit 1 fi - echo "Not available yet (${elapsed}s elapsed), retrying in ${POLL_INTERVAL}s..." + echo "Not available yet (${SECONDS}s elapsed, ${statuses}), retrying in ${POLL_INTERVAL}s..." sleep "${POLL_INTERVAL}" - elapsed=$((elapsed + POLL_INTERVAL)) done - uses: DataDog/dd-octo-sts-action@96a25462dbcb10ebf0bfd6e2ccc917d2ab235b9a # v1.0.4 @@ -112,14 +137,16 @@ jobs: GH_TOKEN: ${{ steps.octo-sts.outputs.token }} DD_TRACE_PY_SHA: ${{ steps.sha.outputs.value }} TEST_SCENARIOS: ${{ github.event.inputs.test_scenarios || 'python_(mem_domain|exceptions|async_gen|lock)_3\.(14|15)' }} + INSTALL_URL: ${{ steps.s3.outputs.url }} run: | # downstream-python.yml run-name: "dd-trace-py downstream ()" TRIGGERED_AT="$(date -u -d "@$(($(date +%s) - 30))" +%Y-%m-%dT%H:%M:%SZ)" - echo "Triggering prof-correctness for commit ${DD_TRACE_PY_SHA}" + echo "Triggering prof-correctness for commit ${DD_TRACE_PY_SHA} via ${INSTALL_URL}" gh workflow run downstream-python.yml \ --repo DataDog/prof-correctness \ -f "dd_trace_py_commit_sha=${DD_TRACE_PY_SHA}" \ - -f "test_scenarios=${TEST_SCENARIOS}" + -f "test_scenarios=${TEST_SCENARIOS}" \ + -f "ddtrace_install_url=${INSTALL_URL}" RUN_ID="" for _ in $(seq 1 30); do