Merge dev: Node.js scraper migration + CI fix (#32)
All checks were successful
CI/CD Pipeline - Apartment API / Scan Dependencies (push) Successful in 13s
CI/CD Pipeline - Apartment API / Lint & Test (push) Successful in 44s
CI/CD Pipeline - Apartment API / Send Webhook Notification (push) Successful in 2s
CI/CD Pipeline - Apartment API / Build & Push Image (push) Successful in 1m41s
CI/CD Pipeline - Apartment API / Deploy to Production (push) Successful in 14s

Co-authored-by: Stephen Minakian <stephenminakian@gmail.com>
Co-committed-by: Stephen Minakian <stephenminakian@gmail.com>
This commit is contained in:
2026-02-08 20:35:29 -07:00
committed by stephen
parent 8dfdfdc8bb
commit 58593111da
27 changed files with 12227 additions and 30 deletions

View File

@ -1142,5 +1142,236 @@ router.patch('/settings', async (req, res) => {
}
});
// ============================================================
// Scraper Endpoints
// ============================================================
const crypto = require('crypto');
const {
isScraperRunning,
acquireLock,
releaseLock,
getCurrentJobId,
getScheduleExpression,
getNextScheduledRun
} = require('../jobs/scraperJob');
const { runScrape } = require('../services/scraperService');
const { createLogger } = require('../services/scraperLogger');
const scraperConfig = require('../config/scraper');
// Logger for scraper admin routes
const scraperRouteLogger = createLogger('scraper-admin');
// Rate limiting for manual scrape trigger (per-user, in-memory)
const RATE_LIMIT_MAX_REQUESTS = 5;
const RATE_LIMIT_WINDOW_MS = 60 * 60 * 1000; // 1 hour in milliseconds
const rateLimitStore = new Map();
/**
* Check rate limit for a given user ID.
* Returns an object indicating whether the request is allowed.
*
* @param {string} userId - The user ID to check
* @returns {{ allowed: boolean, retryAfterSeconds: number|null }}
*/
function checkRateLimit(userId) {
const now = Date.now();
const userKey = userId.toString();
if (!rateLimitStore.has(userKey)) {
rateLimitStore.set(userKey, []);
}
const timestamps = rateLimitStore.get(userKey);
// Remove timestamps outside the current window
const windowStart = now - RATE_LIMIT_WINDOW_MS;
const validTimestamps = timestamps.filter(ts => ts > windowStart);
rateLimitStore.set(userKey, validTimestamps);
if (validTimestamps.length >= RATE_LIMIT_MAX_REQUESTS) {
// Calculate when the oldest request in the window will expire
const oldestTimestamp = validTimestamps[0];
const retryAfterMs = (oldestTimestamp + RATE_LIMIT_WINDOW_MS) - now;
const retryAfterSeconds = Math.ceil(retryAfterMs / 1000);
return { allowed: false, retryAfterSeconds };
}
// Record this request
validTimestamps.push(now);
return { allowed: true, retryAfterSeconds: null };
}
/**
* Reset the rate limiter (for testing)
*/
function resetRateLimiter() {
rateLimitStore.clear();
}
/**
* POST /api/admin/scraper/run
* Trigger a manual scrape
*
* Request body (optional):
* - dryRun: boolean - Skip database writes for safe testing
* - htmlContent: string - Use provided HTML instead of fetching (for testing/debugging)
*/
router.post('/scraper/run', async (req, res) => {
try {
const db = req.app.locals.db;
const { dryRun = false, htmlContent = null } = req.body || {};
// Check rate limit (per-user, before mutex check)
const rateLimitResult = checkRateLimit(req.user._id);
if (!rateLimitResult.allowed) {
res.setHeader('Retry-After', rateLimitResult.retryAfterSeconds.toString());
return res.status(429).json({
error: 'Rate limit exceeded. Maximum 5 trigger requests per hour.'
});
}
// Check if scraper is already running
if (isScraperRunning()) {
return res.status(409).json({ error: 'Scrape already in progress' });
}
// Generate job ID and acquire lock
const jobId = crypto.randomUUID();
acquireLock(jobId);
// Log admin action
await logActivity(db, {
userId: req.user._id.toString(),
action: 'ADMIN_TRIGGER_SCRAPE',
metadata: { jobId, dryRun, usingProvidedHtml: !!htmlContent }
});
// Start scrape asynchronously (do not await - return 202 immediately)
runScrape(db, { trigger: 'manual', jobId, dryRun, htmlContent })
.catch((error) => {
scraperRouteLogger.error('Async scrape failed', {
errorType: error.name,
errorMessage: error.message
});
})
.finally(() => releaseLock());
// Return immediately with job ID
res.status(202).json({
data: {
jobId,
status: 'started',
dryRun,
message: dryRun ? 'Scrape job initiated (dry run - no DB writes)' : 'Scrape job initiated'
}
});
} catch (error) {
scraperRouteLogger.error('Error triggering scrape', {
errorType: error.name,
errorMessage: error.message
});
res.status(500).json({ error: 'Failed to start scrape job' });
}
});
/**
* GET /api/admin/scraper/status
* Get current scraper status including running state, last run details,
* next scheduled run, and schedule expression.
*/
router.get('/scraper/status', async (req, res) => {
try {
const db = req.app.locals.db;
// Get last run from history
const lastRunDoc = await db.collection(scraperConfig.COLLECTIONS.SCRAPER_RUNS)
.findOne({}, { sort: { startedAt: -1 } });
const lastRun = lastRunDoc ? {
jobId: lastRunDoc.jobId,
timestamp: lastRunDoc.startedAt,
status: lastRunDoc.status,
duration: lastRunDoc.duration,
trigger: lastRunDoc.trigger,
unitsProcessed: lastRunDoc.unitsProcessed,
pricesInserted: lastRunDoc.pricesInserted,
errors: lastRunDoc.errors?.length > 0 ? lastRunDoc.errors : null
} : null;
res.json({
data: {
currentStatus: isScraperRunning() ? 'running' : 'idle',
runningJobId: isScraperRunning() ? getCurrentJobId() : null,
lastRun,
nextScheduledRun: getNextScheduledRun(),
schedule: getScheduleExpression()
}
});
} catch (error) {
scraperRouteLogger.error('Error fetching scraper status', {
errorType: error.name,
errorMessage: error.message
});
res.status(503).json({ error: 'Service temporarily unavailable' });
}
});
/**
* GET /api/admin/scraper/history
* Get scraper run history with pagination
*
* Query params:
* - limit: Number of records (1-100, default 30)
* - offset: Number of records to skip (default 0)
*/
router.get('/scraper/history', async (req, res) => {
try {
const db = req.app.locals.db;
// Parse and validate pagination parameters
let limit = parseInt(req.query.limit);
let offset = parseInt(req.query.offset);
// Validate limit
if (req.query.limit !== undefined) {
if (isNaN(limit) || limit < 1) {
return res.status(400).json({ error: 'Limit must be between 1 and 100' });
}
limit = Math.min(limit, 100);
} else {
limit = 30;
}
// Validate offset
if (req.query.offset !== undefined) {
if (isNaN(offset) || offset < 0) {
return res.status(400).json({ error: 'Invalid offset parameter' });
}
} else {
offset = 0;
}
const history = await db.collection(scraperConfig.COLLECTIONS.SCRAPER_RUNS)
.find({})
.sort({ startedAt: -1 })
.skip(offset)
.limit(limit)
.toArray();
res.json({ data: history });
} catch (error) {
scraperRouteLogger.error('Error fetching scraper history', {
errorType: error.name,
errorMessage: error.message
});
res.status(503).json({ error: 'Service temporarily unavailable' });
}
});
module.exports = router;
module.exports.clearStatsCache = clearStatsCache;
module.exports.resetRateLimiter = resetRateLimiter;