aboutsummaryrefslogtreecommitdiffstats
path: root/packages/meshbay-hub/tests/test_site_basics.py
blob: 40508a1c601fe4960836008a12720227d2658d2a (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
"""
The files every website is expected to have at the root of its origin.

They live in the hub's static directory, which is mounted at "/", so every hub
serves them — meshbay.org included, because its Caddyfile hands the hub every
path the public site does not name.
"""

import re
from pathlib import Path

import pytest
from fastapi.testclient import TestClient
from meshbay_hub.app import create_app

CADDYFILE = Path(__file__).resolve().parents[3] / "packaging" / "caddy" / "meshbay.org.Caddyfile"


@pytest.fixture(scope="module")
def client():
    return TestClient(create_app())


def test_robots_keeps_crawlers_out_of_the_signed_in_views(client):
    r = client.get("/robots.txt")
    assert r.status_code == 200
    assert r.headers["content-type"].startswith("text/plain")
    rules = re.findall(r"^Disallow:\s*(\S+)", r.text, re.M)
    assert "/app/" in rules and "/v1/" in rules
    # A crawler rendering the sign-in page needs its scripts and stylesheet.
    assert not any(rule in ("/", "/a/", "/app") for rule in rules), rules


def test_the_sitemap_is_valid_and_named_by_robots(client):
    import xml.etree.ElementTree as ET
    r = client.get("/sitemap.xml")
    assert r.status_code == 200
    assert "xml" in r.headers["content-type"]
    ns = {"s": "http://www.sitemaps.org/schemas/sitemap/0.9"}
    locs = [e.text for e in ET.fromstring(r.content).findall("s:url/s:loc", ns)]
    assert "https://meshbay.org/downloads/" in locs
    assert all(loc.startswith("https://") for loc in locs), locs
    # Nothing robots.txt forbids: a sitemap listing it is a contradiction.
    assert not any("/app/" in loc or "/v1/" in loc for loc in locs), locs
    assert "Sitemap: https://meshbay.org/sitemap.xml" in client.get("/robots.txt").text


@pytest.mark.parametrize("path, magic", [
    ("/favicon.ico", b"\x00\x00\x01\x00"),
    ("/apple-touch-icon.png", b"\x89PNG"),
])
def test_the_icons_are_served_from_the_root(client, path, magic):
    """Browsers ask for both at the root whether or not a page links them."""
    r = client.get(path)
    assert r.status_code == 200
    assert r.content.startswith(magic), f"{path} is not the image it claims to be"


@pytest.mark.skipif(not CADDYFILE.exists(), reason="no Caddyfile in this tree")
@pytest.mark.parametrize("path", ["/robots.txt", "/sitemap.xml", "/favicon.ico", "/apple-touch-icon.png"])
def test_meshbay_org_sends_them_to_the_hub(path):
    """The public site owns only the paths its matcher names; a file claimed
    there would be looked for in /srv/meshbay/site and 404."""
    m = re.search(r"^\s*@site path (.+)$", CADDYFILE.read_text(encoding="utf-8"), re.M)
    assert m, "the @site matcher is gone"
    assert path not in m.group(1).split()


def _og(html: str, prop: str) -> str | None:
    m = re.search(rf'<meta property="og:{prop}" content="([^"]*)">', html)
    return m and m.group(1)


def test_a_link_to_the_hub_previews_with_the_logo():
    """Messengers draw a link from og:title and og:image, and resolve only an
    absolute image URL — so it names the hub's public name, and the file is
    one the hub serves."""
    from meshbay_hub.config import load_config
    cfg = load_config()
    cfg.identity.id = "hub.example.org"
    client = TestClient(create_app(cfg))
    for path in ("/", "/app"):
        html = client.get(path).text
        assert _og(html, "title") == "MeshBay"
        assert _og(html, "image") == "https://hub.example.org/og-image.jpg", path
    image = client.get("/og-image.jpg")
    assert image.status_code == 200 and image.content.startswith(b"\xff\xd8")
    assert len(image.content) < 300_000, "too heavy for some messengers to fetch"


def test_the_preview_description_fits_in_a_preview():
    """A preview shows a line or two and cuts the rest; a cut sentence says
    nothing."""
    from meshbay_hub.api.webapp import PREVIEW_DESCRIPTION
    assert len(PREVIEW_DESCRIPTION) <= 60


def test_search_engines_get_a_title_and_description_that_say_what_it_is(client):
    """The preview description is cut for messengers; a search result has
    room for two lines, and a bare "MeshBay" title matches no search."""
    from meshbay_hub.api.webapp import SEARCH_DESCRIPTION, SEARCH_TITLE
    html = client.get("/").text
    assert f"<title>{SEARCH_TITLE}</title>" in html
    assert 50 <= len(SEARCH_TITLE) <= 60
    assert re.search(r'<meta name="description" content="([^"]*)">', html).group(1) == SEARCH_DESCRIPTION
    assert 110 <= len(SEARCH_DESCRIPTION) <= 160


def test_every_path_to_the_shell_names_the_root_as_canonical():
    from meshbay_hub.config import load_config
    cfg = load_config()
    cfg.identity.id = "hub.example.org"
    client = TestClient(create_app(cfg))
    for path in ("/", "/app", "/app/groups"):
        html = client.get(path).text
        assert '<link rel="canonical" href="https://hub.example.org/">' in html, path


def test_the_shell_has_text_and_links_without_scripts(client):
    """A crawler that does not run app.js reads the <noscript> block and
    nothing else, so it must carry the pitch and the links to follow."""
    html = client.get("/").text
    m = re.search(r"<noscript>(.*?)</noscript>", html, re.S)
    assert m, "no <noscript> block"
    links = re.findall(r'href="([^"]+)"', m.group(1))
    assert "https://meshbay.org/downloads/" in links
    assert any(link.endswith("docs/QUICKSTART.md") for link in links)
    assert "<h1>" in m.group(1)