Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
45 changes: 45 additions & 0 deletions .github/scripts/link-r-dev.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,45 @@
#!/usr/bin/env bash
# Point <docs>/r/ at the development site when that is the only site built.
#
# pkgdown's automatic development mode sends a released version's site to the
# root of its destination and a development version's to dev/ below it. Every
# build from a branch is a development version, so r/ is empty and the landing
# page's link to it lands on nothing.
#
# Moving dev/ up is the obvious fix and the wrong one: the build writes its own
# absolute address into search.json, sitemap.xml, and the pages themselves, so
# a moved tree keeps telling the browser to go to r/dev/, and site search
# breaks. A redirect leaves the build untouched.
#
# Only for local previews and pull request previews. On the published site r/
# is the released documentation, and overwriting its index with a redirect to
# the development docs would be a regression for every reader.
set -euo pipefail

docs="${1:?usage: link-r-dev.sh <docs directory>}"

if [ -f "$docs/r/index.html" ]; then
echo "r/ already has an index, leaving it alone."
exit 0
fi
if [ ! -f "$docs/r/dev/index.html" ]; then
echo "Neither r/ nor r/dev/ has an index, nothing to point at." >&2
exit 0
fi

cat > "$docs/r/index.html" <<'HTML'
<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="robots" content="noindex, nofollow">
<meta http-equiv="refresh" content="0; url=dev/">
<title>Redirecting to the development documentation</title>
</head>
<body>
<p>This preview contains the development documentation. Redirecting to <a href="dev/">dev/</a>.</p>
</body>
</html>
HTML

echo "Wrote $docs/r/index.html, redirecting to dev/."
73 changes: 73 additions & 0 deletions .github/scripts/noindex-preview.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,73 @@
"""Mark every page of a built site as noindex, in place.

Pull request previews live under posit-dev.github.io/commons/pr-<number>/.
Crawlers read robots.txt only at the origin root, and this repository does not
own posit-dev.github.io, so a robots.txt shipped with the site is never
consulted. A meta tag travels with the page instead.

Usage: noindex-preview.py <directory>
"""

from __future__ import annotations

import pathlib
import re
import sys

TAG = '<meta name="robots" content="noindex, nofollow">'
HEAD = re.compile(r"<head[^>]*>", re.IGNORECASE)
HTML = re.compile(r"<html[^>]*>", re.IGNORECASE)
# Anchored, and tolerant of a leading byte order mark, because a doctype
# counts as one only when nothing but that precedes it. The anchor is also
# what keeps a fragment that merely quotes a doctype from passing as a
# document. Written as an escape: a literal mark here would be invisible.
DOCTYPE = re.compile("\\A\\ufeff?\\s*<!doctype[^>]*>", re.IGNORECASE)


def mark(text: str) -> str | None:
"""Return `text` with the tag added, or None if it needs no change.

A file with no head, no html tag, and no leading doctype is a fragment
rather than a page. Nothing indexes it on its own, and giving it a head
would only corrupt whatever embeds it.
"""
if 'name="robots"' in text:
return None

patched, count = HEAD.subn(lambda m: m.group(0) + TAG, text, count=1)
if count:
return patched

# The html tag is tried before the doctype because a head belongs inside
# html. Taking whichever appears first in the text would put the head
# between the two on every page that has both.
head = f"<head>{TAG}</head>"
for pattern in (HTML, DOCTYPE):
patched, count = pattern.subn(lambda m: m.group(0) + head, text, count=1)
if count:
return patched
return None


def main(root: str) -> int:
directory = pathlib.Path(root)
if not directory.is_dir():
print(f"{root} is not a directory", file=sys.stderr)
return 1

marked = skipped = 0
for path in sorted(directory.rglob("*.html")):
text = path.read_text(encoding="utf-8", errors="surrogateescape")
patched = mark(text)
if patched is None:
skipped += 1
continue
path.write_text(patched, encoding="utf-8", errors="surrogateescape")
marked += 1

print(f"Marked {marked} page(s) noindex, skipped {skipped}.")
return 0


if __name__ == "__main__":
sys.exit(main(sys.argv[1] if len(sys.argv) > 1 else "."))
82 changes: 82 additions & 0 deletions .github/scripts/test_noindex_preview.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,82 @@
"""Tests for noindex-preview.py. Run: python3 test_noindex_preview.py"""

import importlib.util
import pathlib
import unittest

spec = importlib.util.spec_from_file_location(
"noindex_preview", pathlib.Path(__file__).parent / "noindex-preview.py"
)
assert spec and spec.loader
noindex = importlib.util.module_from_spec(spec)
spec.loader.exec_module(noindex)

mark = noindex.mark
TAG = noindex.TAG


class MarkTest(unittest.TestCase):
def test_adds_the_tag_inside_an_existing_head(self):
out = mark("<!DOCTYPE html>\n<html><head><title>a</title></head></html>")
assert out is not None
self.assertIn(f"<head>{TAG}", out)

def test_matches_head_case_insensitively_and_with_attributes(self):
out = mark('<html><HEAD lang="en"><title>b</title></HEAD></html>')
assert out is not None
self.assertIn(f'<HEAD lang="en">{TAG}', out)

def test_leaves_a_page_that_already_declares_robots(self):
page = '<html><head><meta name="robots" content="noindex"></head></html>'
self.assertIsNone(mark(page))

def test_synthesizes_a_head_after_the_html_tag(self):
out = mark('<html lang="en"><body>no head</body></html>')
assert out is not None
self.assertTrue(out.startswith(f'<html lang="en"><head>{TAG}</head>'))

# The next three are the regressions this script actually shipped: a
# doctype has to stay first, or the browser renders the preview in quirks
# mode and it no longer resembles the page it previews.
def test_keeps_a_doctype_first_when_there_is_no_html_tag(self):
out = mark("<!DOCTYPE html>\n<body>x</body>")
assert out is not None
self.assertTrue(out.startswith("<!DOCTYPE html>"))
self.assertIn(TAG, out)

def test_keeps_a_lowercase_doctype_first(self):
out = mark("<!doctype HTML>\n<body>x</body>")
assert out is not None
self.assertTrue(out.startswith("<!doctype HTML>"))

def test_keeps_a_byte_order_mark_and_doctype_first(self):
out = mark("<!DOCTYPE html>\n<body>x</body>")
assert out is not None
self.assertTrue(out.startswith("<!DOCTYPE html>"))
self.assertIn(TAG, out)

def test_leaves_a_fragment_alone(self):
self.assertIsNone(mark("<div>bare fragment</div>\n"))

def test_leaves_a_fragment_that_merely_quotes_a_doctype(self):
# Only a doctype at the very start makes the file a document. An
# unanchored match treated this snippet as one and rewrote it.
self.assertIsNone(mark("<p>Start a page with &lt;!DOCTYPE html&gt;</p>"))
self.assertIsNone(mark("<pre>example: <!DOCTYPE html></pre>"))

def test_puts_the_head_inside_html_when_a_doctype_precedes_it(self):
# The head belongs inside html. Matching whichever token came first
# put it between the doctype and the html tag.
out = mark("<!DOCTYPE html>\n<html lang=\"en\"><body>x</body></html>")
assert out is not None
self.assertIn(f'<html lang="en"><head>{TAG}</head>', out)
self.assertTrue(out.startswith("<!DOCTYPE html>"))

def test_is_idempotent(self):
once = mark("<html><head></head></html>")
assert once is not None
self.assertIsNone(mark(once))


if __name__ == "__main__":
unittest.main(verbosity=2)
81 changes: 78 additions & 3 deletions .github/workflows/pkgdown.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -20,14 +20,19 @@ jobs:
defaults:
run:
working-directory: pkg-r
# Shared with the sibling site's workflow: both deploy to gh-pages, and
# two pushes to one branch race.
# Per workflow and per pull request, so a run supersedes an older pending
# run of the same site. Serializing every gh-pages push in one group would
# be wrong: Actions keeps only one pending run per group, so a third
# arrival cancels the queued one outright and its preview never publishes.
# Races between the two sites are handled at the push instead, by the
# deploy action's rebase.
concurrency:
group: gh-pages-${{ github.event_name != 'pull_request' || github.run_id }}
group: pkgdown-${{ github.event.number || github.ref }}
env:
GITHUB_PAT: ${{ secrets.GITHUB_TOKEN }}
permissions:
contents: write
pull-requests: write
steps:
- uses: actions/checkout@v6

Expand All @@ -52,3 +57,73 @@ jobs:
clean: false
branch: gh-pages
folder: docs
# Both sites and the preview deploys push to this one branch, so a
# deploy has to rebase onto whatever landed while it was building.
# Forcing would silently drop that other deploy.
force: false
attempt-limit: 10

# A fork's GITHUB_TOKEN is read-only, so the deploy and the comment both
# fail there. Fork previews are not worth the alternative: publishing
# fork-authored HTML to posit-dev.github.io means serving unreviewed
# content from our own origin.
# Preview only. The production deploy above publishes r/ as the released
# documentation, and this would replace its index with a redirect.
- name: Point r/ at the development site
if: >
github.event_name == 'pull_request' &&
github.event.pull_request.head.repo.fork != true
working-directory: .
run: .github/scripts/link-r-dev.sh docs

# pkgdown's automatic development mode puts a released version's site at
# the root of destination and a development version's in dev/ below it,
# so the entry point moves with the version in DESCRIPTION. Linking to
# r/ unconditionally hands the reviewer a directory listing.
- name: Find the built site's entry point
id: entry
if: >
github.event_name == 'pull_request' &&
github.event.pull_request.head.repo.fork != true
working-directory: .
run: |
if [ -f docs/r/index.html ]; then
echo "path=r/" >> "$GITHUB_OUTPUT"
elif [ -f docs/r/dev/index.html ]; then
echo "path=r/dev/" >> "$GITHUB_OUTPUT"
else
echo "The R site has no index.html to link to." >&2
exit 1
fi

- name: Mark the preview noindex
if: >
github.event_name == 'pull_request' &&
github.event.pull_request.head.repo.fork != true
working-directory: .
run: python3 .github/scripts/noindex-preview.py docs

- name: Deploy the pull request preview 🔎
if: >
github.event_name == 'pull_request' &&
github.event.pull_request.head.repo.fork != true
uses: JamesIves/github-pages-deploy-action@d92aa235d04922e8f08b40ce78cc5442fcfbfa2f # v4.8.0
with:
clean: false
branch: gh-pages
folder: docs
target-folder: pr-${{ github.event.number }}
force: false
attempt-limit: 10

- name: Comment the preview link
if: >
github.event_name == 'pull_request' &&
github.event.pull_request.head.repo.fork != true
uses: marocchino/sticky-pull-request-comment@5770ad5eb8f42dd2c4f34da00c94c5381e49af88 # v3.0.5
with:
header: preview-r
message: |
**R site preview:** [https://posit-dev.github.io/commons/pr-${{ github.event.number }}/${{ steps.entry.outputs.path }}](https://posit-dev.github.io/commons/pr-${{ github.event.number }}/${{ steps.entry.outputs.path }})

Built from the latest commit on this branch. Use the link above rather than the landing page's, which points at the released site. The landing page's Python link resolves only if this pull request also rebuilt the Python site.
Loading
Loading