diff --git a/.github/workflows/daily-geo-optimizer.lock.yml b/.github/workflows/daily-geo-optimizer.lock.yml index c315a066062..b5ad28d2ad2 100644 --- a/.github/workflows/daily-geo-optimizer.lock.yml +++ b/.github/workflows/daily-geo-optimizer.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"9127fb4afa031eeaf2edc230b93fcb6be580335673a305be89ae99f1849f3a2d","body_hash":"f1aa1c7c52f8e02e4a52afaba117aad10ce54d2df1f13afebecb8f6dabf60f8c","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80","copilot-sdk":"1.0.11"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"0f9354670d36c0a5e21d2a9f1b5092e2d79c5f1c1994bf78007c0cc744f3d37f","body_hash":"c165e4d63e33992e1eb8b50ef8178c0c1b7caeb0bde7ea91b4231206c3b5f05b","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80","copilot-sdk":"1.0.11"}} # gh-aw-manifest: {"version":1,"secrets":["GH_AW_AGENT_TOKEN","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GH_AW_OTEL_GRAFANA_AUTHORIZATION","GH_AW_OTEL_GRAFANA_ENDPOINT","GH_AW_OTEL_SENTRY_AUTHORIZATION","GH_AW_OTEL_SENTRY_ENDPOINT","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/setup-python","sha":"5fda3b95a4ea91299a34e894583c3862153e4b97","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.28.1","digest":"sha256:5e3f6ee27eeae07195838b97ac4aa2f8aea42a7c55f1c0d3e17d8e88e294ad0d","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.28.1@sha256:5e3f6ee27eeae07195838b97ac4aa2f8aea42a7c55f1c0d3e17d8e88e294ad0d"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.1","digest":"sha256:288e7d2a12d5b430500d739f9c16e20bb1ed51b91f986f3f3eccde189f489f5c","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.1@sha256:288e7d2a12d5b430500d739f9c16e20bb1ed51b91f986f3f3eccde189f489f5c"},{"image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.1","digest":"sha256:f931e5e1e13f765605d03ef9511fc755d779a51b76581ea14586e9871506a610","pinned_image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.1@sha256:f931e5e1e13f765605d03ef9511fc755d779a51b76581ea14586e9871506a610"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.28.1","digest":"sha256:9d428af47899bf18ef2d5618075777d76ef344c91e76c1f44ec1aaa0ee347e5f","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.28.1@sha256:9d428af47899bf18ef2d5618075777d76ef344c91e76c1f44ec1aaa0ee347e5f"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.9","digest":"sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.9@sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196","pinned_image":"ghcr.io/github/gh-aw-node@sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196"},{"image":"ghcr.io/github/github-mcp-server:v1.9.0","digest":"sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e","pinned_image":"ghcr.io/github/github-mcp-server:v1.9.0@sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e"}]} # This file was automatically generated by gh-aw. DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # @@ -1951,6 +1951,76 @@ jobs: --argjson found "$FOUND" \ '{url: $url, http_status: ($http_status | tonumber), content_type: $content_type, found: $found}' \ > /tmp/gh-aw/agent/geo-optimizer/docs-robots-verification.json + - name: Verify documentation AI discovery files + run: | + PROJECT_SITE_BASE_URL="https://github.github.com/gh-aw" + CHECKS_JSONL="/tmp/gh-aw/agent/geo-optimizer/docs-ai-discovery-verification.jsonl" + : > "$CHECKS_JSONL" + + check_url() { + local key="$1" + local path="$2" + local body_path="$3" + local url="${PROJECT_SITE_BASE_URL}/${path}" + local curl_metadata + local http_status + local content_type + local found=false + + rm -f "$body_path" + # runner-guard:ignore RGS-012 -- unauthenticated GET from the public documentation site; no secrets are sent. + curl_metadata="$(curl --silent --show-error --location --max-time 30 \ + --output "$body_path" --write-out '%{http_code}\t%{content_type}' \ + "$url" || true)" + IFS=$'\t' read -r http_status content_type <<< "$curl_metadata" + if [[ "$http_status" == "200" ]]; then + found=true + else + rm -f "$body_path" + fi + jq -n \ + --arg key "$key" \ + --arg url "$url" \ + --arg http_status "${http_status:-000}" \ + --arg content_type "$content_type" \ + --arg body_path "$body_path" \ + --argjson found "$found" \ + '{ + key: $key, + url: $url, + http_status: ($http_status | tonumber), + content_type: $content_type, + found: $found, + body_path: (if $found then $body_path else null end) + }' >> "$CHECKS_JSONL" + } + + check_url "llms_txt" "llms.txt" "/tmp/gh-aw/agent/geo-optimizer/docs-llms.txt" + # Verify the project-scoped AI discovery signal that the docs site serves below /gh-aw. + check_url "ai_txt" ".well-known/ai.txt" "/tmp/gh-aw/agent/geo-optimizer/docs-ai.txt" + check_url "ai_summary_json" "ai/summary.json" "/tmp/gh-aw/agent/geo-optimizer/docs-ai-summary.json" + check_url "ai_faq_json" "ai/faq.json" "/tmp/gh-aw/agent/geo-optimizer/docs-ai-faq.json" + check_url "ai_service_json" "ai/service.json" "/tmp/gh-aw/agent/geo-optimizer/docs-ai-service.json" + + jq -s \ + --arg base_url "$PROJECT_SITE_BASE_URL" \ + '{ + base_url: $base_url, + checks: ((map(. as $check | {($check.key): ($check | del(.key))}) | add) // {}), + summary: { + llms_txt_found: ( + map(select(.key == "llms_txt")) | + if length > 0 then .[0].found else false end + ), + ai_discovery_found_count: (map(select(.key | startswith("ai_"))) | map(select(.found)) | length), + ai_discovery_total: (map(select(.key | startswith("ai_"))) | length), + ai_discovery_all_found: ( + map(select(.key | startswith("ai_"))) as $ai_checks | + (($ai_checks | length) > 0 and ($ai_checks | all(.found))) + ) + } + }' "$CHECKS_JSONL" \ + > /tmp/gh-aw/agent/geo-optimizer/docs-ai-discovery-verification.json - name: Write audit metadata run: | python3 - <<'EOF' diff --git a/.github/workflows/daily-geo-optimizer.md b/.github/workflows/daily-geo-optimizer.md index 4aacf36063d..0bee0c6d26e 100644 --- a/.github/workflows/daily-geo-optimizer.md +++ b/.github/workflows/daily-geo-optimizer.md @@ -95,6 +95,77 @@ jobs: '{url: $url, http_status: ($http_status | tonumber), content_type: $content_type, found: $found}' \ > /tmp/gh-aw/agent/geo-optimizer/docs-robots-verification.json + - name: Verify documentation AI discovery files + run: | + PROJECT_SITE_BASE_URL="https://github.github.com/gh-aw" + CHECKS_JSONL="/tmp/gh-aw/agent/geo-optimizer/docs-ai-discovery-verification.jsonl" + : > "$CHECKS_JSONL" + + check_url() { + local key="$1" + local path="$2" + local body_path="$3" + local url="${PROJECT_SITE_BASE_URL}/${path}" + local curl_metadata + local http_status + local content_type + local found=false + + rm -f "$body_path" + # runner-guard:ignore RGS-012 -- unauthenticated GET from the public documentation site; no secrets are sent. + curl_metadata="$(curl --silent --show-error --location --max-time 30 \ + --output "$body_path" --write-out '%{http_code}\t%{content_type}' \ + "$url" || true)" + IFS=$'\t' read -r http_status content_type <<< "$curl_metadata" + if [[ "$http_status" == "200" ]]; then + found=true + else + rm -f "$body_path" + fi + jq -n \ + --arg key "$key" \ + --arg url "$url" \ + --arg http_status "${http_status:-000}" \ + --arg content_type "$content_type" \ + --arg body_path "$body_path" \ + --argjson found "$found" \ + '{ + key: $key, + url: $url, + http_status: ($http_status | tonumber), + content_type: $content_type, + found: $found, + body_path: (if $found then $body_path else null end) + }' >> "$CHECKS_JSONL" + } + + check_url "llms_txt" "llms.txt" "/tmp/gh-aw/agent/geo-optimizer/docs-llms.txt" + # Verify the project-scoped AI discovery signal that the docs site serves below /gh-aw. + check_url "ai_txt" ".well-known/ai.txt" "/tmp/gh-aw/agent/geo-optimizer/docs-ai.txt" + check_url "ai_summary_json" "ai/summary.json" "/tmp/gh-aw/agent/geo-optimizer/docs-ai-summary.json" + check_url "ai_faq_json" "ai/faq.json" "/tmp/gh-aw/agent/geo-optimizer/docs-ai-faq.json" + check_url "ai_service_json" "ai/service.json" "/tmp/gh-aw/agent/geo-optimizer/docs-ai-service.json" + + jq -s \ + --arg base_url "$PROJECT_SITE_BASE_URL" \ + '{ + base_url: $base_url, + checks: ((map(. as $check | {($check.key): ($check | del(.key))}) | add) // {}), + summary: { + llms_txt_found: ( + map(select(.key == "llms_txt")) | + if length > 0 then .[0].found else false end + ), + ai_discovery_found_count: (map(select(.key | startswith("ai_"))) | map(select(.found)) | length), + ai_discovery_total: (map(select(.key | startswith("ai_"))) | length), + ai_discovery_all_found: ( + map(select(.key | startswith("ai_"))) as $ai_checks | + (($ai_checks | length) > 0 and ($ai_checks | all(.found))) + ) + } + }' "$CHECKS_JSONL" \ + > /tmp/gh-aw/agent/geo-optimizer/docs-ai-discovery-verification.json + - name: Write audit metadata run: | python3 - <<'EOF' @@ -188,6 +259,8 @@ ls /tmp/gh-aw/agent/geo-optimizer/ - `docs-sitemap-audit.json` — sitemap-wide audit of up to 20 documentation pages - `docs-robots-verification.json` — authoritative HTTP check of the GitHub Pages project-site robots.txt - `docs-robots.txt` — robots.txt response body when the authoritative check succeeds +- `docs-ai-discovery-verification.json` — authoritative HTTP checks for the GitHub Pages project-site llms.txt and AI discovery files +- `docs-llms.txt`, `docs-ai.txt`, `docs-ai-summary.json`, `docs-ai-faq.json`, `docs-ai-service.json` — response bodies when the authoritative checks succeed - `readme-audit.json` — GEO audit of the GitHub repository homepage (README) - `metadata.json` — run metadata (timestamp, URLs) @@ -204,6 +277,14 @@ which does not preserve the `/gh-aw/` base path of this GitHub Pages project sit verification reports `found: true`, do not report robots.txt as missing; inspect the response body before recommending changes to crawler permissions. +For llms.txt and AI discovery findings on the documentation site, treat +`docs-ai-discovery-verification.json` as authoritative. The GEO package may probe the domain +root and miss the `/gh-aw/` base path of this GitHub Pages project site. When the authoritative +verification reports `summary.llms_txt_found: true`, do not report llms.txt as missing. When it +reports `summary.ai_discovery_all_found: true`, do not report AI discovery files as missing. +Inspect the downloaded response bodies before recommending changes to llms.txt or AI discovery +content. + ## Phase 2: Analyze and Summarize Based on the audit results, identify: @@ -286,6 +367,11 @@ Pick the recommendation that: 2. Is concrete and actionable (not "improve content quality" in general) 3. Covers a gap not already tracked by an open issue +If the audit's highest-impact recommendation says documentation-site robots.txt, llms.txt, or AI +discovery files are missing, first compare it against the authoritative verification files above. +If the project-site verification found the file or files, treat that recommendation as a scanner +false negative and select the next actionable recommendation instead. + If **all scores are already Excellent (90+/100)** and there are no actionable recommendations, use `noop` and skip issue creation. ### Issue title