Skip to content
Merged
Show file tree
Hide file tree
Changes from 2 commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 17 additions & 3 deletions .github/scripts/activity-labeler.js
Original file line number Diff line number Diff line change
Expand Up @@ -90,13 +90,13 @@ module.exports = async ({ github, context, core }) => {
}
core.info(`Scanning ${prs.length} open PRs`);

for (const pr of prs) {
async function processOnePr(pr) {
const pr_number = pr.number;
const labelNames = pr.labels.map((l) => l.name);

if (pr.draft) {
core.info(`#${pr_number}: draft, skipping`);
continue;
return;
}
if (labelNames.some((n) => EXEMPT_LABELS.includes(n))) {
// A manual outcome label means a human has already made the call —
Expand All @@ -113,7 +113,7 @@ module.exports = async ({ github, context, core }) => {
} else {
core.info(`#${pr_number}: has manual outcome label, skipping`);
}
continue;
return;
}

// Gather every timestamped human event on the PR.
Expand Down Expand Up @@ -224,4 +224,18 @@ module.exports = async ({ github, context, core }) => {
}
}
}

for (const pr of prs) {
try {
await processOnePr(pr);
} catch (e) {
// One PR's failure (a deleted PR, a transient API error, anything)
// must never abort the scan for every other open PR still queued
// behind it. This matters most here — more than in the backfill
// workflow's own loop — because the nightly cron and a full manual
// run both scan every open PR in a single pass; without this, one
// bad PR would silently kill that night's entire backstop scan.
core.warning(`#${pr.number}: failed during activity scan, skipping (${e.message})`);
Comment thread
akshitpatel1732 marked this conversation as resolved.
}
}
};
57 changes: 52 additions & 5 deletions .github/scripts/pr-metadata-labeler.js
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,57 @@ function sizeTier(totalLines, filesChanged) {
return `${SIZE_PREFIX}XS`;
}

// --- first-contribution ---
// Deliberately NOT using GitHub's Search API here — it has a much
// stricter *secondary* rate limit (30/min) than everything else this
// codebase calls, and hitting it once per PR is what caused a real
// backfill to fail partway the first time this ran at scale.
//
// Also deliberately NOT fetching the repo's full PR history and caching
// it in memory (an earlier version of this file did exactly that): that
// works fine for a single backfill process looping over many PRs, but
// pr.area-labeler.yml runs this in a FRESH process for every single
// real-time PR event — so that cache was empty every time it mattered,
// and "fixing" the backfill case reintroduced the same scaling problem
// on the far more frequent real-time path (every open/push/reopen event,
// on every PR, forever, would refetch the entire repo's PR history just
// to check one author).
//
// Instead: `creator` filters server-side to just this one author's items
// — cheap in both contexts regardless of total repo size — and this
// stops paginating the instant it's seen enough to know the answer.
async function isFirstContribution(github, owner, repo, author) {
try {
let prCount = 0;
let page = 1;
// eslint-disable-next-line no-constant-condition
while (true) {
const { data } = await github.rest.issues.listForRepo({
owner,
repo,
creator: author,
state: "all",
per_page: 100,
page,
});
for (const item of data) {
// listForRepo returns issues AND PRs by this author — `pull_request`
// is only present on the PR ones, which is all we're counting.
if (item.pull_request) prCount++;
if (prCount > 1) return false; // already confirmed not their first — stop here, no need to see the rest
}
if (data.length < 100) break; // last page
page++;
}
return prCount <= 1;
} catch (e) {
// A missing "nice to have" label is a much smaller problem than
// letting this crash the caller's loop — log and move on.
console.warn(`first-contribution check failed for ${author}: ${e.message}`);
return false;
}
}

async function labelOne({ github, owner, repo, pr }) {
const pr_number = pr.number;
const { data: current } = await github.rest.issues.get({ owner, repo, issue_number: pr_number });
Expand Down Expand Up @@ -48,11 +99,7 @@ async function labelOne({ github, owner, repo, pr }) {

// --- first-contribution (sticky once set, cheap to skip re-checking) ---
if (!labelNames.includes("first-contribution")) {
const author = pr.user.login;
const { data: pastPRs } = await github.rest.search.issuesAndPullRequests({
q: `repo:${owner}/${repo} type:pr author:${author}`,
});
if (pastPRs.total_count <= 1) {
if (await isFirstContribution(github, owner, repo, pr.user.login)) {
await github.rest.issues.addLabels({ owner, repo, issue_number: pr_number, labels: ["first-contribution"] });
}
}
Expand Down
16 changes: 14 additions & 2 deletions .github/workflows/pr.labels-backfill.yml
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,7 @@
if: needs.list-open-prs.outputs.json != '[]'
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7 # without this, actions/labeler can't find .github/labeler.yml locally and re-fetches it via the API once per PR in the list — harmless but noisy, and this avoids it
Comment thread
github-advanced-security[bot] marked this conversation as resolved.
Fixed
Comment thread
github-advanced-security[bot] marked this conversation as resolved.
Fixed
- uses: actions/labeler@v7
with:
configuration-path: .github/labeler.yml
Expand All @@ -61,12 +62,23 @@
const prNumbers = ${{ needs.list-open-prs.outputs.json }};

for (const pr_number of prNumbers) {
const { data: pr } = await github.rest.pulls.get({ owner, repo, pull_number: pr_number });
await labelOne({ github, owner, repo, pr });
try {
const { data: pr } = await github.rest.pulls.get({ owner, repo, pull_number: pr_number });
await labelOne({ github, owner, repo, pr });
} catch (e) {
// One PR's failure (rate limit, deleted PR, anything) must
// never abort processing for every PR still queued behind
// it — this is exactly what happened before this fix: a
// single Search API rate-limit hit crashed the whole loop
// partway through a real backfill, silently leaving the
// rest of the PRs untouched.
core.warning(`#${pr_number}: failed during metadata backfill, skipping (${e.message})`);
}
}

activity-labels-backfill:
needs: metadata-labels-backfill
if: always() # runs even if metadata-labels-backfill reported a failure — the two are independent (this script does its own full open-PR scan and doesn't depend on metadata-labels-backfill's output), so one weak spot shouldn't also block the other job that would otherwise work fine on its own
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
Expand Down
2 changes: 2 additions & 0 deletions tools/pr-labeler/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,8 @@ tools/pr-labeler/
2. Merge this to your default branch. Merging alone triggers `ops.label-sync.yml` (it watches `.github/labels.yml`) — check the Actions tab and confirm the full label catalog now exists under Issues → Labels. If it doesn't fire automatically, run it manually: **Actions → Sync Labels → Run workflow**.
3. Run **Actions → Backfill All PR Labels → Run workflow** once. This applies every label type — area, size, multi, first-contribution, and activity status — to every PR that was already open before this system existed.

**Note on scale**: the first-contribution check doesn't call GitHub's Search API at all — that endpoint has a much stricter *secondary* rate limit (30 requests/minute) than everything else this system uses, and calling it once per PR is exactly what caused a real backfill to fail partway the first time this ran at scale. It also doesn't fetch the repo's entire PR history to work this out (an earlier version did exactly that, which fixed the backfill case but reintroduced the same scaling problem on the far more frequent real-time path — see the comment above `isFirstContribution` in `pr-metadata-labeler.js` for why that didn't hold up). Instead, it uses GitHub's server-side `creator` filter to scope the query to just the one author being checked, and stops paginating the instant it's confirmed the answer — cheap in both the real-time (one PR per run) and backfill (many PRs per run) contexts, regardless of how large the repo's overall PR history is. On top of that, both the backfill loop and the activity scan's own loop isolate failures per PR: if any single PR errors for any reason, that failure is logged and skipped rather than aborting every other PR still queued behind it. If a run does still fail outright, it's always safe to just re-run it: every operation here checks current label state before changing anything, so re-running only picks up what's still missing.

From there it's automatic: new/updated PRs get area and size labels within seconds, and activity status updates instantly on new comments, new commits, and reviews, with a nightly scan as a backstop for anything time-based (a tier aging from day 6 to day 7 with no new activity, for instance) or anything the real-time triggers can't reach — see the caveat on forked PRs below.

**Caveat on forked PRs**: `pull_request_review` and `pull_request_review_comment` have no fork-safe "_target" variant, so GitHub gives them a read-only token when the PR is from a fork (a platform limitation, not something fixable here). For PRs from branches within this repo — the normal case — this doesn't apply. If this repo ever accepts outside-fork contributions, a review left on a fork's PR won't update labels instantly through this specific trigger, but the nightly scan still catches it correctly within 24h.
Expand Down
Loading