Merge dev: Node.js scraper migration + CI fix (#32)
All checks were successful
CI/CD Pipeline - Apartment API / Scan Dependencies (push) Successful in 13s
CI/CD Pipeline - Apartment API / Lint & Test (push) Successful in 44s
CI/CD Pipeline - Apartment API / Send Webhook Notification (push) Successful in 2s
CI/CD Pipeline - Apartment API / Build & Push Image (push) Successful in 1m41s
CI/CD Pipeline - Apartment API / Deploy to Production (push) Successful in 14s
All checks were successful
CI/CD Pipeline - Apartment API / Scan Dependencies (push) Successful in 13s
CI/CD Pipeline - Apartment API / Lint & Test (push) Successful in 44s
CI/CD Pipeline - Apartment API / Send Webhook Notification (push) Successful in 2s
CI/CD Pipeline - Apartment API / Build & Push Image (push) Successful in 1m41s
CI/CD Pipeline - Apartment API / Deploy to Production (push) Successful in 14s
Co-authored-by: Stephen Minakian <stephenminakian@gmail.com> Co-committed-by: Stephen Minakian <stephenminakian@gmail.com>
This commit is contained in:
275
jobs/scraperJob.js
Normal file
275
jobs/scraperJob.js
Normal file
@ -0,0 +1,275 @@
|
||||
const cron = require('node-cron');
|
||||
const crypto = require('crypto');
|
||||
const config = require('../config/scraper');
|
||||
const { runScrape } = require('../services/scraperService');
|
||||
const { createLogger } = require('../services/scraperLogger');
|
||||
|
||||
// In-process mutex state
|
||||
let isRunning = false;
|
||||
let currentJobId = null;
|
||||
|
||||
// Scheduler state
|
||||
let scheduledJob = null;
|
||||
|
||||
// Shutdown state
|
||||
let shuttingDown = false;
|
||||
let runningJobPromise = null;
|
||||
|
||||
/**
|
||||
* Check if scraper is currently running
|
||||
* @returns {boolean}
|
||||
*/
|
||||
function isScraperRunning() {
|
||||
return isRunning;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get current job ID if running
|
||||
* @returns {string|null}
|
||||
*/
|
||||
function getCurrentJobId() {
|
||||
return currentJobId;
|
||||
}
|
||||
|
||||
/**
|
||||
* Acquire the scraper lock
|
||||
* @param {string} jobId - Job ID to set
|
||||
* @returns {boolean} True if lock acquired
|
||||
*/
|
||||
function acquireLock(jobId) {
|
||||
if (isRunning || shuttingDown) {
|
||||
return false;
|
||||
}
|
||||
isRunning = true;
|
||||
currentJobId = jobId;
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Release the scraper lock
|
||||
*/
|
||||
function releaseLock() {
|
||||
isRunning = false;
|
||||
currentJobId = null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get the configured schedule expression
|
||||
* @returns {string} Cron expression or 'disabled'
|
||||
*/
|
||||
function getScheduleExpression() {
|
||||
if (!config.SCRAPER_ENABLED) {
|
||||
return 'disabled';
|
||||
}
|
||||
return config.SCRAPER_SCHEDULE;
|
||||
}
|
||||
|
||||
/**
|
||||
* Calculate next scheduled run time
|
||||
* @returns {string|null} ISO timestamp or null if disabled
|
||||
*/
|
||||
function getNextScheduledRun() {
|
||||
if (!config.SCRAPER_ENABLED || !scheduledJob) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const { CronExpressionParser } = require('cron-parser');
|
||||
try {
|
||||
const interval = CronExpressionParser.parse(config.SCRAPER_SCHEDULE, {
|
||||
tz: config.SCRAPER_TIMEZONE
|
||||
});
|
||||
return interval.next().toISOString();
|
||||
} catch (error) {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if schedule runs more frequently than 1 hour
|
||||
* @param {string} schedule - Cron expression
|
||||
* @returns {boolean} True if schedule is too frequent
|
||||
*/
|
||||
function isScheduleTooFrequent(schedule) {
|
||||
const parts = schedule.trim().split(/\s+/);
|
||||
|
||||
if (parts.length < 5) return false;
|
||||
|
||||
const minuteField = parts[0];
|
||||
|
||||
// If minute field is */N with N < 60, it runs more than once per hour
|
||||
if (/^\*\/\d+$/.test(minuteField)) {
|
||||
const interval = parseInt(minuteField.substring(2), 10);
|
||||
if (interval < 60) return true;
|
||||
}
|
||||
|
||||
// If minute field is *, it runs every minute
|
||||
if (minuteField === '*') return true;
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Initialize the scraper scheduler
|
||||
* @param {Db} db - MongoDB database instance
|
||||
*/
|
||||
function initializeScheduler(db) {
|
||||
const logger = createLogger('scheduler');
|
||||
|
||||
// Check if scheduling is enabled
|
||||
if (!config.SCRAPER_ENABLED) {
|
||||
logger.info('Scraper scheduling is disabled');
|
||||
return;
|
||||
}
|
||||
|
||||
// Validate cron expression
|
||||
if (!cron.validate(config.SCRAPER_SCHEDULE)) {
|
||||
logger.error('Invalid cron schedule expression', {
|
||||
schedule: config.SCRAPER_SCHEDULE
|
||||
});
|
||||
logger.warn('Falling back to default schedule: 0 6 * * *');
|
||||
config.SCRAPER_SCHEDULE = '0 6 * * *';
|
||||
}
|
||||
|
||||
// Validate minimum interval (1 hour)
|
||||
if (isScheduleTooFrequent(config.SCRAPER_SCHEDULE)) {
|
||||
logger.warn('Schedule interval less than 1 hour - adjusting to hourly', {
|
||||
originalSchedule: config.SCRAPER_SCHEDULE
|
||||
});
|
||||
config.SCRAPER_SCHEDULE = '0 * * * *';
|
||||
}
|
||||
|
||||
// Create the scheduled job
|
||||
scheduledJob = cron.schedule(config.SCRAPER_SCHEDULE, async () => {
|
||||
const jobId = crypto.randomUUID();
|
||||
const jobLogger = createLogger(jobId);
|
||||
|
||||
jobLogger.info('Scheduled scrape triggered');
|
||||
|
||||
// Check if already running or shutting down
|
||||
if (!acquireLock(jobId)) {
|
||||
jobLogger.warn('Skipped - scrape already in progress');
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
const jobExecution = runScrape(db, { trigger: 'scheduled', jobId });
|
||||
runningJobPromise = jobExecution;
|
||||
await jobExecution;
|
||||
} catch (error) {
|
||||
jobLogger.error('Scheduled scrape failed', {
|
||||
errorType: error.name,
|
||||
errorMessage: error.message
|
||||
});
|
||||
} finally {
|
||||
runningJobPromise = null;
|
||||
releaseLock();
|
||||
}
|
||||
}, {
|
||||
timezone: config.SCRAPER_TIMEZONE,
|
||||
scheduled: true
|
||||
});
|
||||
|
||||
logger.info('Scraper scheduler initialized', {
|
||||
schedule: config.SCRAPER_SCHEDULE,
|
||||
timezone: config.SCRAPER_TIMEZONE,
|
||||
nextRun: getNextScheduledRun()
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Stop the scheduler (for graceful shutdown)
|
||||
*/
|
||||
function stopScheduler() {
|
||||
if (scheduledJob) {
|
||||
scheduledJob.stop();
|
||||
scheduledJob = null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if the scraper is in the process of shutting down
|
||||
* @returns {boolean}
|
||||
*/
|
||||
function isShuttingDown() {
|
||||
return shuttingDown;
|
||||
}
|
||||
|
||||
/**
|
||||
* Perform a graceful shutdown of the scraper
|
||||
* - Stops the cron scheduler to prevent new jobs
|
||||
* - Waits for any running job to complete (with timeout)
|
||||
* - Releases the mutex lock
|
||||
* - Logs shutdown progress
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
async function gracefulShutdown() {
|
||||
const logger = createLogger('shutdown');
|
||||
|
||||
logger.info('Shutdown initiated');
|
||||
|
||||
// Mark as shutting down to prevent new jobs
|
||||
shuttingDown = true;
|
||||
|
||||
// Stop the cron scheduler
|
||||
stopScheduler();
|
||||
|
||||
// Wait for running job to complete (with timeout)
|
||||
if (isRunning && runningJobPromise) {
|
||||
logger.info('Waiting for running job to complete', {
|
||||
jobId: currentJobId,
|
||||
timeout: config.SHUTDOWN_TIMEOUT
|
||||
});
|
||||
|
||||
let timeoutHandle;
|
||||
const timeoutPromise = new Promise((resolve) => {
|
||||
timeoutHandle = setTimeout(() => resolve('timeout'), config.SHUTDOWN_TIMEOUT);
|
||||
});
|
||||
|
||||
const result = await Promise.race([
|
||||
runningJobPromise.then(() => 'completed').catch(() => 'completed'),
|
||||
timeoutPromise
|
||||
]);
|
||||
|
||||
clearTimeout(timeoutHandle);
|
||||
|
||||
if (result === 'timeout') {
|
||||
logger.warn('Shutdown wait for running job timed out', {
|
||||
jobId: currentJobId,
|
||||
timeout: config.SHUTDOWN_TIMEOUT
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Force release the lock
|
||||
releaseLock();
|
||||
|
||||
logger.info('Shutdown complete');
|
||||
}
|
||||
|
||||
/**
|
||||
* Register process signal handlers for graceful shutdown
|
||||
* Listens for SIGTERM and SIGINT signals
|
||||
*/
|
||||
function registerSignalHandlers() {
|
||||
process.on('SIGTERM', () => {
|
||||
gracefulShutdown();
|
||||
});
|
||||
|
||||
process.on('SIGINT', () => {
|
||||
gracefulShutdown();
|
||||
});
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
isScraperRunning,
|
||||
getCurrentJobId,
|
||||
acquireLock,
|
||||
releaseLock,
|
||||
getScheduleExpression,
|
||||
getNextScheduledRun,
|
||||
initializeScheduler,
|
||||
stopScheduler,
|
||||
isShuttingDown,
|
||||
gracefulShutdown,
|
||||
registerSignalHandlers
|
||||
};
|
||||
Reference in New Issue
Block a user