Skip to content

Index general search in Elasticsearch #932

Index general search in Elasticsearch

Index general search in Elasticsearch #932

name: Index general search in Elasticsearch
# Keep general search indexes current by scraping docs content and indexing the records
# into Elasticsearch.
on:
workflow_dispatch:
inputs:
version:
description: "Version to exclusively generate the search index for. E.g. 'dotcom', 'ghes-3.12'"
required: false
default: ''
languages:
description: "Comma separated languages. E.g. 'en,es,ja,pt,zh,ru,fr,ko,de' (defaults to all)"
required: false
default: ''
schedule:
- cron: '20 16 * * 1-5'
workflow_run:
workflows: ['Purge Fastly']
types:
- completed
permissions:
contents: read
# Include the Purge Fastly conclusion so skipped purge runs cannot cancel valid indexing runs.
concurrency:
group: '${{ github.workflow }} @ ${{ github.head_ref }} ${{ github.event_name }} ${{ github.event.workflow_run.conclusion }}'
cancel-in-progress: true
env:
ELASTICSEARCH_URL: ${{ secrets.ELASTICSEARCH_URL }}
# Empty Hydro credentials keep production-mode indexing runs from sending analytics.
HYDRO_ENDPOINT: ''
HYDRO_SECRET: ''
jobs:
figureOutMatrix:
# Skip non-successful Purge Fastly workflow_run events.
if: ${{ github.repository == 'github/docs-internal' && (github.event_name != 'workflow_run' || github.event.workflow_run.conclusion == 'success') }}
runs-on: ubuntu-latest
outputs:
matrix: ${{ steps.set-matrix.outputs.result }}
steps:
- uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
id: set-matrix
with:
script: |
// allNonEnglish is the source of truth for non-English Elasticsearch languages.
const allNonEnglish = 'es,ja,pt,zh,ru,fr,ko,de'.split(',')
const allPossible = ["en", ...allNonEnglish]
if (context.eventName === "workflow_run") {
// The job filter keeps non-successful workflow_run events out, so treat this as a guard.
if (context.payload.workflow_run.conclusion === "success") {
return ["en"]
}
console.warn(`Unexpected: workflow_run with conclusion '${context.payload.workflow_run.conclusion}'`)
return []
}
if (context.eventName === "workflow_dispatch") {
if (context.payload.inputs.languages) {
const clean = context.payload.inputs.languages.split(',').map(x => x.trim()).filter(Boolean)
const notRecognized = clean.find(x => !allPossible.includes(x))
if (notRecognized) {
throw new Error(`'${notRecognized}' is not a recognized language code`)
}
return clean
}
return allPossible
}
if (context.eventName === "schedule") {
return allNonEnglish
}
console.log(context)
throw new Error(`Unable figure out what languages to run (${context.eventName})`)
- name: Debug output
run: echo "${{ steps.set-matrix.outputs.result }}"
- name: Check out repo
if: ${{ failure() && github.event_name != 'workflow_dispatch' }}
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
- uses: ./.github/actions/create-workflow-failure-issue
id: create-failure-issue
if: ${{ failure() && github.event_name != 'workflow_dispatch' }}
with:
token: ${{ secrets.DOCS_BOT_PAT_BASE }}
- uses: ./.github/actions/slack-alert
if: ${{ failure() && github.event_name != 'workflow_dispatch' }}
with:
slack_token: ${{ secrets.SLACK_DOCS_BOT_TOKEN }}
issue_url: ${{ steps.create-failure-issue.outputs.issue_url }}
updateElasticsearchIndexes:
needs: figureOutMatrix
name: Update indexes
if: ${{ github.repository == 'github/docs-internal' && needs.figureOutMatrix.outputs.matrix != '[]' }}
runs-on: ubuntu-latest
strategy:
fail-fast: false
# English-only runs have one matrix item, so max-parallel does not matter.
# Full runs index eight non-English languages, each taking more than 10 minutes.
# Elasticsearch rate-limits higher concurrency, so keep this at 2.
max-parallel: 2
matrix:
language: ${{ fromJSON(needs.figureOutMatrix.outputs.matrix) }}
steps:
- name: Check out repo
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
- name: Clone docs-internal-data
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
repository: github/docs-internal-data
# docs-bot can read the private github/docs-internal-data repository.
token: ${{ secrets.DOCS_BOT_PAT_BASE }}
path: docs-internal-data
- name: Clone all translations
if: ${{ matrix.language != 'en' }}
uses: ./.github/actions/clone-translations
with:
token: ${{ secrets.DOCS_BOT_PAT_BASE }}
- uses: ./.github/actions/node-npm-setup
- uses: ./.github/actions/cache-nextjs
- name: Run build scripts
run: npm run build
- name: Start the server in the background
env:
ENABLE_DEV_LOGGING: false
run: |
npm run general-search-scrape-server > /tmp/stdout.log 2> /tmp/stderr.log &
sleep 6
curl --retry-connrefused --retry 6 -I http://localhost:4002/
- if: ${{ failure() }}
name: Debug server outputs on errors
run: |
echo "____STDOUT____"
cat /tmp/stdout.log
echo "____STDERR____"
cat /tmp/stderr.log
- name: Scrape records into a temp directory
env:
# Deleting a reusable or data/* entry can leave translations pointing at missing
# site.data keys and trigger RenderError. Accept empty strings until the next
# translation pipeline removes those references.
THROW_ON_EMPTY: false
# Empty VERSION makes the scrape script use its default version selection.
VERSION: ${{ inputs.version }}
DOCS_INTERNAL_DATA: docs-internal-data
run: |
mkdir /tmp/records
npm run general-search-scrape -- /tmp/records \
--language ${{ matrix.language }}
ls -lh /tmp/records
- name: Check for scraping failures
id: check-failures
run: |
if [ -f /tmp/records/failures-summary.json ]; then
FAILED_PAGES=$(jq -r '.totalFailedPages' /tmp/records/failures-summary.json)
echo "failed_pages=$FAILED_PAGES" >> $GITHUB_OUTPUT
echo "has_failures=true" >> $GITHUB_OUTPUT
echo "⚠️ Warning: $FAILED_PAGES page(s) failed to scrape"
else
echo "has_failures=false" >> $GITHUB_OUTPUT
echo "✅ All pages scraped successfully"
fi
- name: Check that Elasticsearch is accessible
run: |
curl --fail --retry-connrefused --retry 5 -I ${{ env.ELASTICSEARCH_URL }}
- name: Index into Elasticsearch
env:
# Match the scrape version so the indexer reads the directory the scraper wrote.
VERSION: ${{ inputs.version }}
run: |
npm run index-general-search -- /tmp/records \
--language ${{ matrix.language }} \
--stagger-seconds 5 \
--retries 5
- name: Check created indexes and aliases
run: |
# Avoid --fail because an empty index list can return a 404 instead of 200 OK.
curl --retry-connrefused --retry 5 ${{ env.ELASTICSEARCH_URL }}/_cat/indices?v
curl --retry-connrefused --retry 5 ${{ env.ELASTICSEARCH_URL }}/_cat/indices?v
- name: Purge Fastly edge cache
env:
FASTLY_TOKEN: ${{ secrets.FASTLY_TOKEN }}
FASTLY_SERVICE_ID: ${{ secrets.FASTLY_SERVICE_ID }}
run: npm run purge-fastly -- --surrogate-key api-search:${{ matrix.language }}
- name: Upload failures artifact
if: ${{ steps.check-failures.outputs.has_failures == 'true' }}
uses: actions/upload-artifact@bbbca2ddaa5d8feaa63e36b76fdaad77386f024f # v7.0.0
with:
name: search-failures-${{ matrix.language }}
path: /tmp/records/failures-summary.json
retention-days: 1
- uses: ./.github/actions/create-workflow-failure-issue
id: create-failure-issue
if: ${{ failure() && github.event_name != 'workflow_dispatch' }}
with:
token: ${{ secrets.DOCS_BOT_PAT_BASE }}
- uses: ./.github/actions/slack-alert
if: ${{ failure() && github.event_name != 'workflow_dispatch' }}
with:
slack_token: ${{ secrets.SLACK_DOCS_BOT_TOKEN }}
issue_url: ${{ steps.create-failure-issue.outputs.issue_url }}
notifyScrapingFailures:
name: Notify scraping failures
needs: updateElasticsearchIndexes
if: ${{ always() && github.repository == 'github/docs-internal' && github.event_name != 'workflow_dispatch' && needs.updateElasticsearchIndexes.result != 'cancelled' }}
runs-on: ubuntu-latest
steps:
- name: Check out repo
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
- name: Download all failure artifacts
uses: actions/download-artifact@70fc10c6e5e1ce46ad2ea6f2b72d43f7d47b13c3 # v8.0.0
with:
pattern: search-failures-*
path: /tmp/failures
continue-on-error: true
- name: Check if any failures were downloaded
id: check-artifacts
run: |
if [ -d /tmp/failures ] && [ "$(ls -A /tmp/failures 2>/dev/null)" ]; then
echo "has_artifacts=true" >> $GITHUB_OUTPUT
else
echo "has_artifacts=false" >> $GITHUB_OUTPUT
fi
- uses: ./.github/actions/node-npm-setup
if: ${{ steps.check-artifacts.outputs.has_artifacts == 'true' }}
- name: Aggregate failures and format message
if: ${{ steps.check-artifacts.outputs.has_artifacts == 'true' }}
id: aggregate
run: |
RESULT=$(npx tsx src/search/scripts/aggregate-search-index-failures.ts /tmp/failures \
--workflow-url "${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}")
{
echo 'result<<EOF'
echo "$RESULT"
echo 'EOF'
} >> "$GITHUB_OUTPUT"
- name: Comment on or create scraping failure issue
if: ${{ steps.check-artifacts.outputs.has_artifacts == 'true' && fromJSON(steps.aggregate.outputs.result || '{"hasFailures":false}').hasFailures }}
env:
GH_TOKEN: ${{ secrets.DOCS_BOT_PAT_BASE }}
FAILURE_MESSAGE: ${{ fromJSON(steps.aggregate.outputs.result || '{"message":""}').message }}
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
FILE_URL: ${{ github.server_url }}/${{ github.repository }}/blob/main/.github/workflows/index-general-search.yml
WORKFLOW_NAME: ${{ github.workflow }}
run: |
# Reuse the oldest open scraping-failures issue to keep the issue list quiet.
existing_issue=$(gh issue list \
--repo github/technical-content \
--label "search-scraping-failures" \
--state open \
--limit 200 \
--json number,createdAt \
--jq 'sort_by(.createdAt) | .[0].number // empty')
today=$(date -u +%Y-%m-%d)
if [ -n "$existing_issue" ]; then
comment_body=$(cat <<EOF
### Search index scraping failures ($today)
$FAILURE_MESSAGE
**Workflow run:** $RUN_URL
EOF
)
gh issue comment "$existing_issue" \
--repo github/technical-content \
--body "$comment_body"
exit 0
fi
body=$(cat <<EOF
### Search index scraping failures
$FAILURE_MESSAGE
---
**Workflow run:** $RUN_URL
**Workflow file:** $FILE_URL
This issue was automatically created by the \`$WORKFLOW_NAME\` workflow.
Subsequent failures from later workflow runs will be added as comments
on this issue rather than opening a new issue each day.
EOF
)
gh issue create \
--repo github/technical-content \
--label "search-scraping-failures" \
--title "[Search Scraping Failures] $today" \
--body "$body"
- name: Send consolidated Slack notification
if: ${{ steps.check-artifacts.outputs.has_artifacts == 'true' && fromJSON(steps.aggregate.outputs.result || '{"hasFailures":false}').hasFailures }}
uses: ./.github/actions/slack-alert
with:
slack_token: ${{ secrets.SLACK_DOCS_BOT_TOKEN }}
message: ${{ fromJSON(steps.aggregate.outputs.result || '{"message":""}').message }}
- uses: ./.github/actions/create-workflow-failure-issue
id: create-failure-issue
if: ${{ failure() }}
with:
token: ${{ secrets.DOCS_BOT_PAT_BASE }}
- uses: ./.github/actions/slack-alert
if: ${{ failure() }}
with:
slack_token: ${{ secrets.SLACK_DOCS_BOT_TOKEN }}
issue_url: ${{ steps.create-failure-issue.outputs.issue_url }}