Merge dev: Node.js scraper migration + CI fix (#32)
All checks were successful
CI/CD Pipeline - Apartment API / Scan Dependencies (push) Successful in 13s
CI/CD Pipeline - Apartment API / Lint & Test (push) Successful in 44s
CI/CD Pipeline - Apartment API / Send Webhook Notification (push) Successful in 2s
CI/CD Pipeline - Apartment API / Build & Push Image (push) Successful in 1m41s
CI/CD Pipeline - Apartment API / Deploy to Production (push) Successful in 14s
All checks were successful
CI/CD Pipeline - Apartment API / Scan Dependencies (push) Successful in 13s
CI/CD Pipeline - Apartment API / Lint & Test (push) Successful in 44s
CI/CD Pipeline - Apartment API / Send Webhook Notification (push) Successful in 2s
CI/CD Pipeline - Apartment API / Build & Push Image (push) Successful in 1m41s
CI/CD Pipeline - Apartment API / Deploy to Production (push) Successful in 14s
Co-authored-by: Stephen Minakian <stephenminakian@gmail.com> Co-committed-by: Stephen Minakian <stephenminakian@gmail.com>
This commit is contained in:
231
routes/admin.js
231
routes/admin.js
@ -1142,5 +1142,236 @@ router.patch('/settings', async (req, res) => {
|
||||
}
|
||||
});
|
||||
|
||||
// ============================================================
|
||||
// Scraper Endpoints
|
||||
// ============================================================
|
||||
|
||||
const crypto = require('crypto');
|
||||
const {
|
||||
isScraperRunning,
|
||||
acquireLock,
|
||||
releaseLock,
|
||||
getCurrentJobId,
|
||||
getScheduleExpression,
|
||||
getNextScheduledRun
|
||||
} = require('../jobs/scraperJob');
|
||||
const { runScrape } = require('../services/scraperService');
|
||||
const { createLogger } = require('../services/scraperLogger');
|
||||
const scraperConfig = require('../config/scraper');
|
||||
|
||||
// Logger for scraper admin routes
|
||||
const scraperRouteLogger = createLogger('scraper-admin');
|
||||
|
||||
// Rate limiting for manual scrape trigger (per-user, in-memory)
|
||||
const RATE_LIMIT_MAX_REQUESTS = 5;
|
||||
const RATE_LIMIT_WINDOW_MS = 60 * 60 * 1000; // 1 hour in milliseconds
|
||||
const rateLimitStore = new Map();
|
||||
|
||||
/**
|
||||
* Check rate limit for a given user ID.
|
||||
* Returns an object indicating whether the request is allowed.
|
||||
*
|
||||
* @param {string} userId - The user ID to check
|
||||
* @returns {{ allowed: boolean, retryAfterSeconds: number|null }}
|
||||
*/
|
||||
function checkRateLimit(userId) {
|
||||
const now = Date.now();
|
||||
const userKey = userId.toString();
|
||||
|
||||
if (!rateLimitStore.has(userKey)) {
|
||||
rateLimitStore.set(userKey, []);
|
||||
}
|
||||
|
||||
const timestamps = rateLimitStore.get(userKey);
|
||||
|
||||
// Remove timestamps outside the current window
|
||||
const windowStart = now - RATE_LIMIT_WINDOW_MS;
|
||||
const validTimestamps = timestamps.filter(ts => ts > windowStart);
|
||||
rateLimitStore.set(userKey, validTimestamps);
|
||||
|
||||
if (validTimestamps.length >= RATE_LIMIT_MAX_REQUESTS) {
|
||||
// Calculate when the oldest request in the window will expire
|
||||
const oldestTimestamp = validTimestamps[0];
|
||||
const retryAfterMs = (oldestTimestamp + RATE_LIMIT_WINDOW_MS) - now;
|
||||
const retryAfterSeconds = Math.ceil(retryAfterMs / 1000);
|
||||
return { allowed: false, retryAfterSeconds };
|
||||
}
|
||||
|
||||
// Record this request
|
||||
validTimestamps.push(now);
|
||||
return { allowed: true, retryAfterSeconds: null };
|
||||
}
|
||||
|
||||
/**
|
||||
* Reset the rate limiter (for testing)
|
||||
*/
|
||||
function resetRateLimiter() {
|
||||
rateLimitStore.clear();
|
||||
}
|
||||
|
||||
/**
|
||||
* POST /api/admin/scraper/run
|
||||
* Trigger a manual scrape
|
||||
*
|
||||
* Request body (optional):
|
||||
* - dryRun: boolean - Skip database writes for safe testing
|
||||
* - htmlContent: string - Use provided HTML instead of fetching (for testing/debugging)
|
||||
*/
|
||||
router.post('/scraper/run', async (req, res) => {
|
||||
try {
|
||||
const db = req.app.locals.db;
|
||||
const { dryRun = false, htmlContent = null } = req.body || {};
|
||||
|
||||
// Check rate limit (per-user, before mutex check)
|
||||
const rateLimitResult = checkRateLimit(req.user._id);
|
||||
if (!rateLimitResult.allowed) {
|
||||
res.setHeader('Retry-After', rateLimitResult.retryAfterSeconds.toString());
|
||||
return res.status(429).json({
|
||||
error: 'Rate limit exceeded. Maximum 5 trigger requests per hour.'
|
||||
});
|
||||
}
|
||||
|
||||
// Check if scraper is already running
|
||||
if (isScraperRunning()) {
|
||||
return res.status(409).json({ error: 'Scrape already in progress' });
|
||||
}
|
||||
|
||||
// Generate job ID and acquire lock
|
||||
const jobId = crypto.randomUUID();
|
||||
acquireLock(jobId);
|
||||
|
||||
// Log admin action
|
||||
await logActivity(db, {
|
||||
userId: req.user._id.toString(),
|
||||
action: 'ADMIN_TRIGGER_SCRAPE',
|
||||
metadata: { jobId, dryRun, usingProvidedHtml: !!htmlContent }
|
||||
});
|
||||
|
||||
// Start scrape asynchronously (do not await - return 202 immediately)
|
||||
runScrape(db, { trigger: 'manual', jobId, dryRun, htmlContent })
|
||||
.catch((error) => {
|
||||
scraperRouteLogger.error('Async scrape failed', {
|
||||
errorType: error.name,
|
||||
errorMessage: error.message
|
||||
});
|
||||
})
|
||||
.finally(() => releaseLock());
|
||||
|
||||
// Return immediately with job ID
|
||||
res.status(202).json({
|
||||
data: {
|
||||
jobId,
|
||||
status: 'started',
|
||||
dryRun,
|
||||
message: dryRun ? 'Scrape job initiated (dry run - no DB writes)' : 'Scrape job initiated'
|
||||
}
|
||||
});
|
||||
|
||||
} catch (error) {
|
||||
scraperRouteLogger.error('Error triggering scrape', {
|
||||
errorType: error.name,
|
||||
errorMessage: error.message
|
||||
});
|
||||
res.status(500).json({ error: 'Failed to start scrape job' });
|
||||
}
|
||||
});
|
||||
|
||||
/**
|
||||
* GET /api/admin/scraper/status
|
||||
* Get current scraper status including running state, last run details,
|
||||
* next scheduled run, and schedule expression.
|
||||
*/
|
||||
router.get('/scraper/status', async (req, res) => {
|
||||
try {
|
||||
const db = req.app.locals.db;
|
||||
|
||||
// Get last run from history
|
||||
const lastRunDoc = await db.collection(scraperConfig.COLLECTIONS.SCRAPER_RUNS)
|
||||
.findOne({}, { sort: { startedAt: -1 } });
|
||||
|
||||
const lastRun = lastRunDoc ? {
|
||||
jobId: lastRunDoc.jobId,
|
||||
timestamp: lastRunDoc.startedAt,
|
||||
status: lastRunDoc.status,
|
||||
duration: lastRunDoc.duration,
|
||||
trigger: lastRunDoc.trigger,
|
||||
unitsProcessed: lastRunDoc.unitsProcessed,
|
||||
pricesInserted: lastRunDoc.pricesInserted,
|
||||
errors: lastRunDoc.errors?.length > 0 ? lastRunDoc.errors : null
|
||||
} : null;
|
||||
|
||||
res.json({
|
||||
data: {
|
||||
currentStatus: isScraperRunning() ? 'running' : 'idle',
|
||||
runningJobId: isScraperRunning() ? getCurrentJobId() : null,
|
||||
lastRun,
|
||||
nextScheduledRun: getNextScheduledRun(),
|
||||
schedule: getScheduleExpression()
|
||||
}
|
||||
});
|
||||
|
||||
} catch (error) {
|
||||
scraperRouteLogger.error('Error fetching scraper status', {
|
||||
errorType: error.name,
|
||||
errorMessage: error.message
|
||||
});
|
||||
res.status(503).json({ error: 'Service temporarily unavailable' });
|
||||
}
|
||||
});
|
||||
|
||||
/**
|
||||
* GET /api/admin/scraper/history
|
||||
* Get scraper run history with pagination
|
||||
*
|
||||
* Query params:
|
||||
* - limit: Number of records (1-100, default 30)
|
||||
* - offset: Number of records to skip (default 0)
|
||||
*/
|
||||
router.get('/scraper/history', async (req, res) => {
|
||||
try {
|
||||
const db = req.app.locals.db;
|
||||
|
||||
// Parse and validate pagination parameters
|
||||
let limit = parseInt(req.query.limit);
|
||||
let offset = parseInt(req.query.offset);
|
||||
|
||||
// Validate limit
|
||||
if (req.query.limit !== undefined) {
|
||||
if (isNaN(limit) || limit < 1) {
|
||||
return res.status(400).json({ error: 'Limit must be between 1 and 100' });
|
||||
}
|
||||
limit = Math.min(limit, 100);
|
||||
} else {
|
||||
limit = 30;
|
||||
}
|
||||
|
||||
// Validate offset
|
||||
if (req.query.offset !== undefined) {
|
||||
if (isNaN(offset) || offset < 0) {
|
||||
return res.status(400).json({ error: 'Invalid offset parameter' });
|
||||
}
|
||||
} else {
|
||||
offset = 0;
|
||||
}
|
||||
|
||||
const history = await db.collection(scraperConfig.COLLECTIONS.SCRAPER_RUNS)
|
||||
.find({})
|
||||
.sort({ startedAt: -1 })
|
||||
.skip(offset)
|
||||
.limit(limit)
|
||||
.toArray();
|
||||
|
||||
res.json({ data: history });
|
||||
|
||||
} catch (error) {
|
||||
scraperRouteLogger.error('Error fetching scraper history', {
|
||||
errorType: error.name,
|
||||
errorMessage: error.message
|
||||
});
|
||||
res.status(503).json({ error: 'Service temporarily unavailable' });
|
||||
}
|
||||
});
|
||||
|
||||
module.exports = router;
|
||||
module.exports.clearStatsCache = clearStatsCache;
|
||||
module.exports.resetRateLimiter = resetRateLimiter;
|
||||
|
||||
Reference in New Issue
Block a user