Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions .github/CODEOWNERS
Original file line number Diff line number Diff line change
Expand Up @@ -227,6 +227,8 @@ ddtrace/internal/settings/profiling.py @DataDog/profiling-python
ddtrace/internal/settings/profiling.pyi @DataDog/profiling-python
ddtrace/internal/datadog/profiling @DataDog/profiling-python
tests/profiling @DataDog/profiling-python
.github/workflows/profiling-native.yml @DataDog/profiling-python
.github/workflows/prof-correctness.yml @DataDog/profiling-python
.gitlab/tests/profiling.yml @DataDog/profiling-python
.github/workflows/pytorch_gpu_tests.yml @DataDog/profiling-python
.github/PULL_REQUEST_TEMPLATE/profiler_change.md @DataDog/profiling-python
Expand Down
65 changes: 46 additions & 19 deletions .github/workflows/prof-correctness.yml
Comment thread
vlad-scherbich marked this conversation as resolved.
Original file line number Diff line number Diff line change
Expand Up @@ -29,11 +29,11 @@ permissions:
jobs:
trigger-downstream:
runs-on: ubuntu-latest
# 60m S3 poll + 60m downstream watch = 120, plus ~10m slack
# 90m S3 poll + 60m downstream watch = 150, plus ~10m slack
# for octo-sts / trigger / GHA overhead.
# Job timeout must stay strictly above POLL_TIMEOUT so a late upload
# does not consume the entire budget before downstream-python starts.
timeout-minutes: 130
timeout-minutes: 160
steps:
- name: Resolve commit SHA
id: sha
Expand Down Expand Up @@ -71,32 +71,57 @@ jobs:
echo "run=true" >> "$GITHUB_OUTPUT"
fi

# GitLab uploads install.sh after wheel build (often 20–63+ min under
# load). This workflow uses its own 60m poll budget for late GitLab
# uploads (independent of scripts/download-s3-wheels.sh). Poll before
# dispatch so downstream does not 404.
- name: Wait for S3 install.sh
# GitLab publishes per-platform indexes first
# (install-manylinux2014_x86_64.sh, then install-manylinux2014.sh).
# Unsuffixed install.sh is only after "upload all" (often 20–78+ min).
# Poll in that order so we dispatch when linux amd64 wheels are up.
# 90m is the fallback if even the earliest index is late.
- name: Wait for S3 install script
if: steps.gate.outputs.run == 'true'
id: s3
env:
DD_TRACE_PY_SHA: ${{ steps.sha.outputs.value }}
run: |
INSTALL_URL="https://dd-trace-py-builds.s3.amazonaws.com/${DD_TRACE_PY_SHA}/install.sh"
POLL_TIMEOUT="${POLL_TIMEOUT:-3800}"
S3_BASE="https://dd-trace-py-builds.s3.amazonaws.com/${DD_TRACE_PY_SHA}"
# Platform indexes land before unsuffixed install.sh.
CANDIDATES="install-manylinux2014_x86_64.sh install-manylinux2014.sh install.sh"
POLL_TIMEOUT="${POLL_TIMEOUT:-5400}"
POLL_INTERVAL="${POLL_INTERVAL:-30}"
echo "Polling for: ${INSTALL_URL}"
elapsed=0
echo "Polling ${S3_BASE} for: ${CANDIDATES}"
echo "Timeout: ${POLL_TIMEOUT}s, Interval: ${POLL_INTERVAL}s"
found_url=""
# Bound each probe so a hung S3/TCP stall cannot blow past POLL_TIMEOUT.
CURL_CONNECT_TIMEOUT="${CURL_CONNECT_TIMEOUT:-5}"
CURL_MAX_TIME="${CURL_MAX_TIME:-15}"
# Wall-clock budget: elapsed += POLL_INTERVAL ignores curl probe time
# (up to 3 * CURL_MAX_TIME per round) and can overrun the job timeout.
SECONDS=0
while true; do
if curl -sf -o /dev/null "${INSTALL_URL}"; then
echo "install.sh found after ${elapsed}s"
statuses=""
found_url=""
for name in ${CANDIDATES}; do
url="${S3_BASE}/${name}"
code="$(curl -sS --connect-timeout "${CURL_CONNECT_TIMEOUT}" --max-time "${CURL_MAX_TIME}" \
-o /dev/null -w '%{http_code}' "${url}" || true)"
code="${code:-curl_error}"
statuses="${statuses}${statuses:+, }${name}=${code}"
if [ "$code" = "200" ]; then
found_url="$url"
break
fi
done
if [ -n "$found_url" ]; then
echo "found ${found_url} after ${SECONDS}s (${statuses})"
echo "url=${found_url}" >> "$GITHUB_OUTPUT"
break
fi
if [ "$elapsed" -ge "$POLL_TIMEOUT" ]; then
echo "error: timed out after ${POLL_TIMEOUT}s waiting for ${INSTALL_URL}" >&2
if [ "$SECONDS" -ge "$POLL_TIMEOUT" ]; then
echo "error: timed out after ${POLL_TIMEOUT}s waiting for any of: ${CANDIDATES} under ${S3_BASE} (last ${statuses})" >&2
echo "GitLab may still be publishing wheels. Re-run this job after a candidate returns 200." >&2
exit 1
fi
echo "Not available yet (${elapsed}s elapsed), retrying in ${POLL_INTERVAL}s..."
echo "Not available yet (${SECONDS}s elapsed, ${statuses}), retrying in ${POLL_INTERVAL}s..."
sleep "${POLL_INTERVAL}"
elapsed=$((elapsed + POLL_INTERVAL))
done

- uses: DataDog/dd-octo-sts-action@96a25462dbcb10ebf0bfd6e2ccc917d2ab235b9a # v1.0.4
Expand All @@ -112,14 +137,16 @@ jobs:
GH_TOKEN: ${{ steps.octo-sts.outputs.token }}
DD_TRACE_PY_SHA: ${{ steps.sha.outputs.value }}
TEST_SCENARIOS: ${{ github.event.inputs.test_scenarios || 'python_(mem_domain|exceptions|async_gen|lock)_3\.(14|15)' }}
INSTALL_URL: ${{ steps.s3.outputs.url }}
run: |
# downstream-python.yml run-name: "dd-trace-py downstream (<sha>)"
TRIGGERED_AT="$(date -u -d "@$(($(date +%s) - 30))" +%Y-%m-%dT%H:%M:%SZ)"
echo "Triggering prof-correctness for commit ${DD_TRACE_PY_SHA}"
echo "Triggering prof-correctness for commit ${DD_TRACE_PY_SHA} via ${INSTALL_URL}"
gh workflow run downstream-python.yml \
--repo DataDog/prof-correctness \
-f "dd_trace_py_commit_sha=${DD_TRACE_PY_SHA}" \
-f "test_scenarios=${TEST_SCENARIOS}"
-f "test_scenarios=${TEST_SCENARIOS}" \
-f "ddtrace_install_url=${INSTALL_URL}"
Comment thread
vlad-scherbich marked this conversation as resolved.

RUN_ID=""
for _ in $(seq 1 30); do
Expand Down
Loading