fix(api): harvest every authored message, not the first 100000 (#2631)

This commit is contained in:
Hampus
2026-09-09 11:52:55 +02:00
committed by GitHub
parent 75be6aa492
commit 48b569b9d4
2 changed files with 19 additions and 2 deletions
@@ -171,6 +171,10 @@ interface ArchiveResult {
downloadUrl: string; downloadUrl: string;
} }
// A harvest is every message the account wrote, so the read pages to the end of
// the account rather than stopping at a count. The page size is what bounds one
// query, not what bounds the archive.
const HARVEST_MESSAGE_CHUNK_SIZE = 1000;
const CONCURRENT_MESSAGE_LIMIT = 10; const CONCURRENT_MESSAGE_LIMIT = 10;
const INITIAL_PROGRESS = 5; const INITIAL_PROGRESS = 5;
const MESSAGES_PROGRESS_MAX = 55; const MESSAGES_PROGRESS_MAX = 55;
@@ -231,6 +235,7 @@ async function harvestMessages(
listMessagesByAuthor: ( listMessagesByAuthor: (
userId: UserID, userId: UserID,
limit: number, limit: number,
lastMessageId?: MessageID,
) => Promise< ) => Promise<
Array<{ Array<{
channelId: ChannelID; channelId: ChannelID;
@@ -252,7 +257,19 @@ async function harvestMessages(
const channelMessagesMap = new Map<string, Array<HarvestedMessage>>(); const channelMessagesMap = new Map<string, Array<HarvestedMessage>>();
Logger.debug('Fetching all user messages'); Logger.debug('Fetching all user messages');
const startFetchTime = Date.now(); const startFetchTime = Date.now();
const messageRefs = await channelRepository.listMessagesByAuthor(userId, 100000); const messageRefs: Array<{channelId: ChannelID; messageId: MessageID}> = [];
let lastMessageId: MessageID | undefined;
while (true) {
const page = await channelRepository.listMessagesByAuthor(userId, HARVEST_MESSAGE_CHUNK_SIZE, lastMessageId);
if (page.length === 0) {
break;
}
messageRefs.push(...page);
lastMessageId = page[page.length - 1].messageId;
if (page.length < HARVEST_MESSAGE_CHUNK_SIZE) {
break;
}
}
Logger.debug( Logger.debug(
{ {
totalMessages: messageRefs.length, totalMessages: messageRefs.length,
@@ -191,7 +191,7 @@ Fluxer prepares the archive in the background. It contains:
Attachment metadata appears with its message and has the attachment ID, filename, size, content type, CDN URL, and pixel dimensions. The attachment files themselves are never included. Attachment metadata appears with its message and has the attachment ID, filename, size, content type, CDN URL, and pixel dimensions. The attachment files themselves are never included.
At most 100,000 authored messages are collected. Fluxer skips a message it cannot read, and the harvest still succeeds. When the archive completes, Fluxer sends one email containing a download URL if the account has an email address and the instance has email delivery enabled. That URL expires seven days after it was issued. Every authored message is collected, with no ceiling on the count. Fluxer skips a message it cannot read, and the harvest still succeeds. When the archive completes, Fluxer sends one email containing a download URL if the account has an email address and the instance has email delivery enabled. That URL expires seven days after it was issued.
### Rate limit ### Rate limit