1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
|
"""Scheduled cleanup tasks — IP log purge, and unhosted group collection."""
import asyncio
import logging
from datetime import UTC, datetime, timedelta
from sqlalchemy import delete, select, update
from sqlalchemy.ext.asyncio import AsyncSession
from meshbay_hub.db.models import EmailVerification, Group, GroupInviteLink, IPLog, User
log = logging.getLogger(__name__)
RETENTION_DAYS = 365
CLEANUP_INTERVAL_HOURS = 24
async def purge_old_ip_logs(db: AsyncSession, retention_days: int = RETENTION_DAYS) -> int:
cutoff = datetime.now(UTC) - timedelta(days=retention_days)
result = await db.execute(delete(IPLog).where(IPLog.timestamp < cutoff))
await db.commit()
return result.rowcount
PENDING_USER_EXPIRY_DAYS = 7
async def purge_expired_verifications(db: AsyncSession) -> int:
now = datetime.now(UTC)
result = await db.execute(
delete(EmailVerification).where(EmailVerification.expires_at < now))
await db.commit()
return result.rowcount
async def purge_stale_pending_users(db: AsyncSession,
expiry_days: int = PENDING_USER_EXPIRY_DAYS) -> int:
"""Delete accounts that never verified their address, and what points at them.
A pending account is not childless: registering writes an `account_create`
IP-log row, a failed sign-in another, and a group owner may have added it to
a group. A bare `DELETE FROM users` therefore violates those foreign keys —
which PostgreSQL enforces and the SQLite the suite runs on does not — so on
the real hub it raised, the stale account stayed, and every later run
raised again.
The IP log is kept and detached, keeping the name, exactly as
`erase_account` does for a deleted account: it is the legal record. Every
other nullable reference is cleared and every other row deleted, found from
the schema so a table added later is covered. An account that owns a group
is left alone — a pending account cannot create one, so that would be a
state this function did not expect and should not guess about.
"""
from meshbay_hub.db.models import Base, Group
cutoff = datetime.now(UTC) - timedelta(days=expiry_days)
stale = (await db.execute(
select(User.id, User.username).where(
User.status == "pending", User.created_at < cutoff,
~User.id.in_(select(Group.admin_id))))).all()
if not stale:
return 0
ids = [uid for uid, _ in stale]
for uid, name in stale:
await db.execute(update(IPLog).where(IPLog.user_id == uid)
.values(username=name, user_id=None))
for table in Base.metadata.sorted_tables:
if table.name in (User.__tablename__, IPLog.__tablename__):
continue
for fk in table.foreign_keys:
if fk.column.table.name != User.__tablename__:
continue
column = fk.parent
if column.nullable:
await db.execute(update(table).where(column.in_(ids))
.values({column.name: None}))
else:
await db.execute(delete(table).where(column.in_(ids)))
result = await db.execute(delete(User).where(User.id.in_(ids)))
await db.commit()
return result.rowcount
async def purge_invite_links(db: AsyncSession) -> int:
"""Unused invitation links once expired; used ones after a month."""
from meshbay_hub.api.invite_links import KEEP_REDEEMED
now = datetime.now(UTC)
result = await db.execute(delete(GroupInviteLink).where(
(GroupInviteLink.redeemed_by.is_(None) & (GroupInviteLink.expires_at <= now))
| (GroupInviteLink.redeemed_at < now - KEEP_REDEEMED)))
await db.commit()
return result.rowcount
async def _purge_login_throttle(db: AsyncSession) -> int:
from meshbay_hub import login_throttle
return await login_throttle.purge_expired(db)
async def _purge_mail_quota(db: AsyncSession) -> int:
from meshbay_hub import mail
return await mail.purge_expired_quota(db)
# In order, each in its own session and its own try: one step that raises must
# not cost the others their run. They used to share one `try`, so the day
# `purge_stale_pending_users` first hit a foreign key on PostgreSQL, the mail
# counters, the sign-in counters and the invitation links behind it stopped
# being purged at all — silently, since the loop logged one line and slept.
_STEPS = (
("IP log entries older than the retention period", purge_old_ip_logs),
("expired email verifications", purge_expired_verifications),
("stale pending users", purge_stale_pending_users),
# One row per recipient the hub has written to, and the window is a day:
# without this the table grows for the life of the instance.
("expired mail counters", _purge_mail_quota),
# Every name anybody types at the sign-in form is a row, real or not.
("expired sign-in counters", _purge_login_throttle),
("spent or expired invitation links", purge_invite_links),
)
async def run_cleanup(get_session) -> dict[str, int | None]:
"""One pass of every step. A step that failed reports None."""
done: dict[str, int | None] = {}
for what, step in _STEPS:
try:
async with get_session() as db:
n = await step(db)
done[what] = n
if n:
log.info("Purged %d %s", n, what)
except asyncio.CancelledError:
raise
except Exception as e:
done[what] = None
log.error("Cleanup of %s failed: %s", what, e)
return done
async def cleanup_loop(get_session):
"""Run cleanup once at startup, then every 24 hours."""
try:
while True:
await run_cleanup(get_session)
await asyncio.sleep(CLEANUP_INTERVAL_HOURS * 3600)
except asyncio.CancelledError:
return
# ── Groups that never got a node ──────────────────────────────────────────────
UNHOSTED_GRACE_DAYS = 7
async def find_unhosted_groups(db: AsyncSession, grace_days: int = UNHOSTED_GRACE_DAYS):
"""Groups created more than `grace_days` ago that no node has ever announced.
`hosted_at` is set the first time a node registers claiming the group and is
never cleared, so this finds groups that were created and then abandoned —
not ones whose node happens to be offline today. That distinction is the
whole reason the column exists rather than a check against the live socket
registry, which would delete every group during a hub restart.
"""
cutoff = datetime.now(UTC) - timedelta(days=grace_days)
result = await db.execute(
select(Group).where(Group.hosted_at.is_(None), Group.created_at < cutoff))
return list(result.scalars().all())
async def prune_unhosted_groups(db: AsyncSession, grace_days: int = UNHOSTED_GRACE_DAYS,
dry_run: bool = False) -> list[tuple[str, str]]:
"""Delete abandoned groups. Returns [(id, name)] of what was (or would be) removed.
Everything on the hub that points at the group goes with it (`purge_groups`):
there is no cascade configured, an orphan membership would keep the group in
everyone's /mine query through the join, and on PostgreSQL any remaining
reference refuses the deletion outright. Nothing on a node is touched: the
hub does not command those machines, and by definition no node ever claimed
this group anyway.
"""
doomed = await find_unhosted_groups(db, grace_days)
if not doomed or dry_run:
return [(g.id, g.name) for g in doomed]
from meshbay_hub.db.purge import purge_groups
await purge_groups(db, [g.id for g in doomed])
await db.commit()
log.info("Pruned %d group(s) that no node ever hosted", len(doomed))
return [(g.id, g.name) for g in doomed]
|