diff options
| author | Christophe Besson <cbesson@gmail.com> | 2026-09-30 10:25:28 +0200 |
|---|---|---|
| committer | Christophe Besson <cbesson@gmail.com> | 2026-09-30 10:25:28 +0200 |
| commit | e7859865e9519aa529f15c1e7a852f3841c44986 (patch) | |
| tree | 8ee654a1e6b8bafebccfbefe1d9e7b39a724938e /packages/meshbay-hub | |
| parent | 609003e907e66e951db840e800fca77322bbde99 (diff) | |
| download | meshbay-e7859865e9519aa529f15c1e7a852f3841c44986.tar.gz | |
feat(hub): sitemap.xml, named by robots.txt
Home, downloads, and the repository's about page and docs on
git.meshbay.org, spelled as the welcome page links them.
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Diffstat (limited to 'packages/meshbay-hub')
| -rw-r--r-- | packages/meshbay-hub/src/meshbay_hub/static/robots.txt | 2 | ||||
| -rw-r--r-- | packages/meshbay-hub/src/meshbay_hub/static/sitemap.xml | 20 | ||||
| -rw-r--r-- | packages/meshbay-hub/tests/test_site_basics.py | 16 |
3 files changed, 37 insertions, 1 deletions
diff --git a/packages/meshbay-hub/src/meshbay_hub/static/robots.txt b/packages/meshbay-hub/src/meshbay_hub/static/robots.txt index a2ef43d..9b89f0f 100644 --- a/packages/meshbay-hub/src/meshbay_hub/static/robots.txt +++ b/packages/meshbay-hub/src/meshbay_hub/static/robots.txt @@ -7,3 +7,5 @@ User-agent: * Disallow: /app/ Disallow: /v1/ + +Sitemap: https://meshbay.org/sitemap.xml diff --git a/packages/meshbay-hub/src/meshbay_hub/static/sitemap.xml b/packages/meshbay-hub/src/meshbay_hub/static/sitemap.xml new file mode 100644 index 0000000..0b40829 --- /dev/null +++ b/packages/meshbay-hub/src/meshbay_hub/static/sitemap.xml @@ -0,0 +1,20 @@ +<?xml version="1.0" encoding="UTF-8"?> +<!-- + Served by the hub at the root of its origin, like robots.txt, which names it. + The public pages of meshbay.org: the downloads listing, and the repository's + about page and documentation on git.meshbay.org, spelled exactly as the + welcome page links them (auth-page.js, REPO) so both count as one URL. + + The git.meshbay.org entries are on another host: a crawler accepts them only + if that host is verified by the same owner (a Search Console domain property + covers every subdomain) or its own robots.txt names this sitemap. +--> +<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"> + <url><loc>https://meshbay.org/</loc></url> + <url><loc>https://meshbay.org/downloads/</loc></url> + <url><loc>https://git.meshbay.org/meshbay.git/about/</loc></url> + <url><loc>https://git.meshbay.org/meshbay.git/about/docs/QUICKSTART.md</loc></url> + <url><loc>https://git.meshbay.org/meshbay.git/about/docs/USERGUIDE.md</loc></url> + <url><loc>https://git.meshbay.org/meshbay.git/about/docs/MESHBAY_DESIGN.md</loc></url> + <url><loc>https://git.meshbay.org/meshbay.git/about/docs/MESHBAY_NODE_PROTOCOL.md</loc></url> +</urlset> diff --git a/packages/meshbay-hub/tests/test_site_basics.py b/packages/meshbay-hub/tests/test_site_basics.py index 78cad83..5c19aa6 100644 --- a/packages/meshbay-hub/tests/test_site_basics.py +++ b/packages/meshbay-hub/tests/test_site_basics.py @@ -31,6 +31,20 @@ def test_robots_keeps_crawlers_out_of_the_signed_in_views(client): assert not any(rule in ("/", "/a/", "/app") for rule in rules), rules +def test_the_sitemap_is_valid_and_named_by_robots(client): + import xml.etree.ElementTree as ET + r = client.get("/sitemap.xml") + assert r.status_code == 200 + assert "xml" in r.headers["content-type"] + ns = {"s": "http://www.sitemaps.org/schemas/sitemap/0.9"} + locs = [e.text for e in ET.fromstring(r.content).findall("s:url/s:loc", ns)] + assert "https://meshbay.org/downloads/" in locs + assert all(loc.startswith("https://") for loc in locs), locs + # Nothing robots.txt forbids: a sitemap listing it is a contradiction. + assert not any("/app/" in loc or "/v1/" in loc for loc in locs), locs + assert "Sitemap: https://meshbay.org/sitemap.xml" in client.get("/robots.txt").text + + @pytest.mark.parametrize("path, magic", [ ("/favicon.ico", b"\x00\x00\x01\x00"), ("/apple-touch-icon.png", b"\x89PNG"), @@ -43,7 +57,7 @@ def test_the_icons_are_served_from_the_root(client, path, magic): @pytest.mark.skipif(not CADDYFILE.exists(), reason="no Caddyfile in this tree") -@pytest.mark.parametrize("path", ["/robots.txt", "/favicon.ico", "/apple-touch-icon.png"]) +@pytest.mark.parametrize("path", ["/robots.txt", "/sitemap.xml", "/favicon.ico", "/apple-touch-icon.png"]) def test_meshbay_org_sends_them_to_the_hub(path): """The public site owns only the paths its matcher names; a file claimed there would be looked for in /srv/meshbay/site and 404.""" |