Repository navigation
Expand file tree
/
Copy pathdatabase.py
More file actions
433 lines (412 loc) · 21.3 KB
/
Copy pathdatabase.py
File metadata and controls
433 lines (412 loc) · 21.3 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
import sqlite3
import asyncio
import logging
import time
import os
import json
import config
logger = logging.getLogger("MacbLogger")
class DatabaseManager:
def __init__(self, bot, metricsTracker=None):
self.bot = bot
self.metricsTracker = metricsTracker
self.writeConn = None
self.sharedReadConn = None
self.dbQueue = asyncio.Queue(maxsize=config.maxDbQueueSize)
self.retryQueue = asyncio.Queue(maxsize=config.maxDbQueueSize)
self.writeLock = asyncio.Lock()
self.workerTask = None
self.isReady = False
self.isBatchActive = False
self.lastBatchProgressAt = 0.0
def getReadConnection(self):
conn = sqlite3.connect(config.dbPath, check_same_thread=False)
conn.execute("PRAGMA journal_mode=WAL;")
conn.execute("PRAGMA busy_timeout=5000;")
conn.execute("PRAGMA synchronous=NORMAL;")
return conn
def _isConnectionHealthy(self, conn):
if not conn:
return False
try:
conn.execute("SELECT 1")
return True
except Exception:
return False
def initDatabase(self):
try:
if not self._isConnectionHealthy(self.writeConn):
if self.writeConn:
try:
self.writeConn.close()
except Exception:
pass
self.writeConn = sqlite3.connect(config.dbPath, check_same_thread=False)
self.writeConn.execute("PRAGMA journal_mode=WAL;")
self.writeConn.execute("PRAGMA busy_timeout=5000;")
self.writeConn.execute("PRAGMA synchronous=NORMAL;")
self.writeConn.execute("PRAGMA mmap_size=268435456;")
self.writeConn.execute("PRAGMA cache_size=-64000;")
self.writeConn.execute("PRAGMA temp_store=MEMORY;")
if not self._isConnectionHealthy(self.sharedReadConn):
if self.sharedReadConn:
try:
self.sharedReadConn.close()
except Exception:
pass
self.sharedReadConn = sqlite3.connect(config.dbPath, isolation_level=None, check_same_thread=False)
self.sharedReadConn.execute("PRAGMA journal_mode=WAL;")
self.sharedReadConn.execute("PRAGMA busy_timeout=5000;")
self.sharedReadConn.execute("PRAGMA synchronous=NORMAL;")
self.sharedReadConn.execute("PRAGMA mmap_size=268435456;")
self.sharedReadConn.execute("PRAGMA cache_size=-64000;")
self.sharedReadConn.execute("PRAGMA temp_store=MEMORY;")
cursor = self.writeConn.cursor()
cursor.execute("""
CREATE TABLE IF NOT EXISTS cachedMessages (
messageId INTEGER PRIMARY KEY,
authorId INTEGER,
authorName TEXT,
authorDisplayName TEXT,
authorGlobalName TEXT,
authorAvatar TEXT,
channelId INTEGER,
channelName TEXT,
parentChannelName TEXT,
content TEXT,
attachments TEXT,
stickers TEXT,
embeds TEXT,
replyReference TEXT,
messageType INTEGER,
contentTypes TEXT,
updatedAt REAL DEFAULT 0.0
)
""")
cursor.execute("CREATE INDEX IF NOT EXISTS idx_channel_msg ON cachedMessages(channelId, messageId DESC);")
cursor.execute("""
CREATE TABLE IF NOT EXISTS dbDeadLetters (
id INTEGER PRIMARY KEY AUTOINCREMENT,
actionType TEXT,
actionData TEXT,
attempts INTEGER,
lastError TEXT,
createdAt REAL
)
""")
cursor.execute("""
CREATE TABLE IF NOT EXISTS logDeadLetters (
id INTEGER PRIMARY KEY AUTOINCREMENT,
logType TEXT,
payload TEXT,
attempts INTEGER,
lastError TEXT,
createdAt REAL,
claimedAt REAL DEFAULT 0.0
)
""")
try:
cursor.execute("ALTER TABLE cachedMessages ADD COLUMN updatedAt REAL DEFAULT 0.0")
except sqlite3.OperationalError:
pass
try:
cursor.execute("ALTER TABLE logDeadLetters ADD COLUMN claimedAt REAL DEFAULT 0.0")
except sqlite3.OperationalError:
pass
cursor.execute("SELECT messageId, updatedAt FROM cachedMessages LIMIT 1")
cursor.execute("INSERT INTO dbDeadLetters (actionType, actionData, attempts, lastError, createdAt) VALUES ('__health_check__', '{}', 0, '', ?)", (time.time(),))
self.writeConn.commit()
cursor.execute("DELETE FROM dbDeadLetters WHERE actionType = '__health_check__'")
self.writeConn.commit()
cursor.close()
self.isReady = True
return True
except Exception as ex:
logger.critical(f"Database initialization failed: {str(ex)}")
self.isReady = False
return False
def startWorker(self, loop):
if self.isReady and (self.workerTask is None or self.workerTask.done()):
self.workerTask = loop.create_task(self.dbWorker())
async def enqueueAction(self, actionType, actionData):
if not self.isReady:
logger.warning(f"DB not ready; dropping action {actionType}")
return False
try:
await asyncio.wait_for(self.dbQueue.put((actionType, actionData)), timeout=5.0)
if self.metricsTracker:
self.metricsTracker.updateDbQueue(self.dbQueue.qsize())
return True
except asyncio.TimeoutError:
logger.error(f"DB queue full; dropping action {actionType}")
if self.metricsTracker:
self.metricsTracker.incrementQueueDropped()
return False
except Exception as ex:
logger.error(f"Error enqueuing database action: {str(ex)}")
return False
async def _persistDeadLetter(self, actionType, actionData, attempts, errorMsg):
persisted = False
try:
if self.writeConn:
cursor = self.writeConn.cursor()
dataStr = json.dumps(actionData) if not isinstance(actionData, str) else actionData
cursor.execute(
"INSERT INTO dbDeadLetters (actionType, actionData, attempts, lastError, createdAt) VALUES (?, ?, ?, ?, ?)",
(actionType, dataStr, attempts, str(errorMsg), time.time())
)
self.writeConn.commit()
cursor.close()
persisted = True
if self.metricsTracker:
self.metricsTracker.incrementDeadLettered()
logger.critical(f"Persisted dead-letter DB action: {actionType} (attempts: {attempts})")
except Exception as ex:
logger.critical(f"Failed to persist dead-letter action to SQLite: {str(ex)}")
if not persisted:
try:
spoolEntry = {
"actionType": actionType,
"actionData": actionData,
"attempts": attempts,
"lastError": str(errorMsg),
"createdAt": time.time()
}
with open("db_dead_letters_spool.jsonl", "a", encoding="utf-8") as f:
f.write(json.dumps(spoolEntry, ensure_ascii=False) + "\n")
if self.metricsTracker:
self.metricsTracker.incrementDeadLettered()
logger.critical(f"Spolled dead-letter DB action to disk: {actionType}")
except Exception as spoolEx:
if self.metricsTracker:
self.metricsTracker.incrementDeadLetterPersistFailed()
logger.critical(f"Disk spool fallback failed: {str(spoolEx)}")
async def dbWorker(self):
while True:
trackedItems = []
try:
try:
actionItem = self.retryQueue.get_nowait()
trackedItems.append(("retry", actionItem))
except asyncio.QueueEmpty:
try:
actionItem = await asyncio.wait_for(self.dbQueue.get(), timeout=1.0)
trackedItems.append(("db", actionItem))
except asyncio.TimeoutError:
if getattr(self.bot, "watchdog", None):
self.bot.watchdog.feedHeartbeat("DatabaseWorker")
continue
while len(trackedItems) < 100:
try:
trackedItems.append(("retry", self.retryQueue.get_nowait()))
except asyncio.QueueEmpty:
try:
trackedItems.append(("db", self.dbQueue.get_nowait()))
except asyncio.QueueEmpty:
break
normalizedActions = []
for qName, item in trackedItems:
if len(item) == 3:
normalizedActions.append((qName, item[0], item[1], item[2]))
else:
normalizedActions.append((qName, item[0], item[1], 0))
self.isBatchActive = True
self.lastBatchProgressAt = time.perf_counter()
cursor = None
async with self.writeLock:
try:
cursor = self.writeConn.cursor()
for index, (qName, actType, actData, attempts) in enumerate(normalizedActions, start=1):
if actType == "save":
cursor.execute("""
INSERT INTO cachedMessages (
messageId, authorId, authorName, authorDisplayName, authorGlobalName, authorAvatar,
channelId, channelName, parentChannelName, content, attachments, stickers, embeds,
replyReference, messageType, contentTypes, updatedAt
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
ON CONFLICT(messageId) DO UPDATE SET
authorName=excluded.authorName,
authorDisplayName=excluded.authorDisplayName,
authorGlobalName=excluded.authorGlobalName,
authorAvatar=excluded.authorAvatar,
content=excluded.content,
attachments=excluded.attachments,
contentTypes=excluded.contentTypes,
updatedAt=excluded.updatedAt
WHERE excluded.updatedAt >= cachedMessages.updatedAt
""", actData)
elif actType == "bulkSave":
cursor.executemany("""
INSERT INTO cachedMessages (
messageId, authorId, authorName, authorDisplayName, authorGlobalName, authorAvatar,
channelId, channelName, parentChannelName, content, attachments, stickers, embeds,
replyReference, messageType, contentTypes, updatedAt
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
ON CONFLICT(messageId) DO UPDATE SET
authorName=excluded.authorName,
authorDisplayName=excluded.authorDisplayName,
authorGlobalName=excluded.authorGlobalName,
authorAvatar=excluded.authorAvatar,
content=excluded.content,
attachments=excluded.attachments,
contentTypes=excluded.contentTypes,
updatedAt=excluded.updatedAt
WHERE excluded.updatedAt >= cachedMessages.updatedAt
""", actData)
elif actType == "bulkUpdateOffline":
cursor.executemany("UPDATE cachedMessages SET content=?, attachments=?, contentTypes=?, updatedAt=? WHERE messageId=? AND updatedAt <= ?", actData)
elif actType == "bulkDelete":
cursor.executemany("DELETE FROM cachedMessages WHERE messageId = ?", [(mId,) for mId in actData])
elif actType == "delete":
cursor.execute("DELETE FROM cachedMessages WHERE messageId = ?", (actData,))
elif actType == "updateFields":
mId, fields = actData
setClauses = ", ".join([f"{k} = ?" for k in fields.keys()])
values = list(fields.values()) + [mId]
if "updatedAt" in fields:
values.append(fields["updatedAt"])
cursor.execute(f"UPDATE cachedMessages SET {setClauses} WHERE messageId = ? AND updatedAt <= ?", values)
else:
cursor.execute(f"UPDATE cachedMessages SET {setClauses} WHERE messageId = ?", values)
if index % 20 == 0:
self.lastBatchProgressAt = time.perf_counter()
if getattr(self.bot, "watchdog", None):
self.bot.watchdog.feedHeartbeat("DatabaseWorker")
self.writeConn.commit()
self.lastBatchProgressAt = time.perf_counter()
if getattr(self.bot, "watchdog", None):
self.bot.watchdog.feedHeartbeat("DatabaseWorker")
except Exception as batchError:
logger.error(f"Error processing database batch: {str(batchError)}")
try:
self.writeConn.rollback()
except Exception:
pass
try:
if not self._isConnectionHealthy(self.writeConn):
self.writeConn.close()
self.writeConn = None
self.initDatabase()
except Exception:
pass
for qName, actType, actData, attempts in normalizedActions:
nextAttempts = attempts + 1
if nextAttempts <= 3:
try:
await asyncio.wait_for(self.retryQueue.put((actType, actData, nextAttempts)), timeout=2.0)
if self.metricsTracker:
self.metricsTracker.incrementRetryEnqueued()
except Exception:
await self._persistDeadLetter(actType, actData, nextAttempts, batchError)
else:
await self._persistDeadLetter(actType, actData, nextAttempts, batchError)
finally:
self.isBatchActive = False
if cursor:
try:
cursor.close()
except Exception:
pass
for qName, _ in trackedItems:
if qName == "retry":
self.retryQueue.task_done()
else:
self.dbQueue.task_done()
except asyncio.CancelledError:
break
except Exception as workerError:
logger.error(f"Error in database worker loop: {str(workerError)}")
await asyncio.sleep(1)
def getStartupMetadata(self):
try:
cursor = self.sharedReadConn.cursor()
cursor.execute("SELECT MIN(messageId) FROM cachedMessages")
oldestRow = cursor.fetchone()
oldestId = oldestRow[0] if oldestRow else None
cursor.execute("SELECT channelId, MAX(messageId) FROM cachedMessages GROUP BY channelId")
rows = cursor.fetchall()
cursor.close()
return oldestId, {row[0]: row[1] for row in rows}
except Exception as ex:
logger.error(f"Error fetching startup metadata combo: {str(ex)}")
return None, {}
def getMessagesFromId(self, channelId, minId):
try:
cursor = self.sharedReadConn.cursor()
query = "SELECT authorId, authorName, authorAvatar, content, attachments, replyReference, contentTypes, messageId FROM cachedMessages WHERE channelId = ? AND messageId >= ?"
cursor.execute(query, (channelId, minId))
rows = cursor.fetchall()
cursor.close()
cacheMap = {}
for row in rows:
if row[7] is not None:
cacheMap[row[7]] = row[:7]
return cacheMap
except Exception as ex:
logger.error(f"Error reading messages from ID {minId} for channel {channelId}: {str(ex)}")
return {}
def getMessage(self, messageId):
conn = None
try:
conn = self.getReadConnection()
dbCursor = conn.cursor()
dbCursor.execute("SELECT messageId, authorId, authorName, authorDisplayName, authorGlobalName, authorAvatar, channelId, channelName, parentChannelName, content, attachments, stickers, embeds, replyReference, messageType, contentTypes FROM cachedMessages WHERE messageId = ?", (messageId,))
dataRow = dbCursor.fetchone()
dbCursor.close()
return dataRow
except Exception as executionException:
logger.error(f"Error reading message {messageId}: {str(executionException)}")
return None
finally:
if conn:
conn.close()
def getMessagesBulk(self, messageIds):
if not messageIds:
return {}
conn = None
try:
conn = self.getReadConnection()
dbCursor = conn.cursor()
result = {}
chunkSize = 900
idList = list(messageIds)
for i in range(0, len(idList), chunkSize):
chunk = idList[i:i + chunkSize]
sqlPlaceholders = ", ".join(["?"] * len(chunk))
sqlQuery = f"SELECT messageId, authorId, authorName, authorDisplayName, authorGlobalName, authorAvatar, channelId, channelName, parentChannelName, content, attachments, stickers, embeds, replyReference, messageType, contentTypes FROM cachedMessages WHERE messageId IN ({sqlPlaceholders})"
dbCursor.execute(sqlQuery, chunk)
for row in dbCursor.fetchall():
result[row[0]] = row
dbCursor.close()
return result
except Exception as executionException:
logger.error(f"Error bulk reading messages: {str(executionException)}")
return {}
finally:
if conn:
conn.close()
def getTotalMessageCount(self):
try:
cursor = self.sharedReadConn.cursor()
cursor.execute("SELECT COUNT(*) FROM cachedMessages")
row = cursor.fetchone()
cursor.close()
return row[0] if row else 0
except Exception as ex:
logger.error(f"Error counting total messages: {str(ex)}")
return 0
async def flushAndClose(self):
try:
await asyncio.wait_for(self.dbQueue.join(), timeout=30.0)
except asyncio.TimeoutError:
logger.warning(f"DB queue drain timed out; {self.dbQueue.qsize()} actions unprocessed")
if self.workerTask:
self.workerTask.cancel()
try:
await self.workerTask
except asyncio.CancelledError:
pass
if self.sharedReadConn:
self.sharedReadConn.close()
if self.writeConn:
self.writeConn.close()