feat: add cron scheduler initialization with validation and mutex integration
All checks were successful
CI/CD Pipeline - Apartment API / Send Webhook Notification (pull_request) Successful in 2s
CI/CD Pipeline - Apartment API / Build & Push Image (pull_request) Has been skipped
CI/CD Pipeline - Apartment API / Deploy to Production (pull_request) Has been skipped
CI/CD Pipeline - Apartment API / Scan Dependencies (pull_request) Successful in 13s
CI/CD Pipeline - Apartment API / Lint & Test (pull_request) Successful in 38s
All checks were successful
CI/CD Pipeline - Apartment API / Send Webhook Notification (pull_request) Successful in 2s
CI/CD Pipeline - Apartment API / Build & Push Image (pull_request) Has been skipped
CI/CD Pipeline - Apartment API / Deploy to Production (pull_request) Has been skipped
CI/CD Pipeline - Apartment API / Scan Dependencies (pull_request) Successful in 13s
CI/CD Pipeline - Apartment API / Lint & Test (pull_request) Successful in 38s
Implement initializeScheduler() and stopScheduler() functions for the scraper cron job module. Key features: - Config-driven enable/disable via SCRAPER_ENABLED environment variable - Cron expression validation with fallback to default schedule (0 6 * * *) - Minimum interval enforcement (1-hour floor) to prevent excessive runs - Timezone support via SCRAPER_TIMEZONE configuration - Mutex lock integration to prevent concurrent scraper executions - Automatic lock release in finally block for error resilience - getScheduleExpression() and getNextScheduledRun() status helpers Add comprehensive test suite with 42 tests covering mutex locking, scheduler lifecycle, cron validation, minimum interval enforcement, concurrent execution prevention, and error handling.
This commit is contained in:
@ -1,7 +1,16 @@
|
||||
const cron = require('node-cron');
|
||||
const config = require('../config/scraper');
|
||||
const { runScrape } = require('../services/scraperService');
|
||||
const { createLogger } = require('../services/scraperLogger');
|
||||
const { v4: uuidv4 } = require('uuid');
|
||||
|
||||
// In-process mutex state
|
||||
let isRunning = false;
|
||||
let currentJobId = null;
|
||||
|
||||
// Scheduler state
|
||||
let scheduledJob = null;
|
||||
|
||||
/**
|
||||
* Check if scraper is currently running
|
||||
* @returns {boolean}
|
||||
@ -40,9 +49,143 @@ function releaseLock() {
|
||||
currentJobId = null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get the configured schedule expression
|
||||
* @returns {string} Cron expression or 'disabled'
|
||||
*/
|
||||
function getScheduleExpression() {
|
||||
if (!config.SCRAPER_ENABLED) {
|
||||
return 'disabled';
|
||||
}
|
||||
return config.SCRAPER_SCHEDULE;
|
||||
}
|
||||
|
||||
/**
|
||||
* Calculate next scheduled run time
|
||||
* @returns {string|null} ISO timestamp or null if disabled
|
||||
*/
|
||||
function getNextScheduledRun() {
|
||||
if (!config.SCRAPER_ENABLED || !scheduledJob) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const { CronExpressionParser } = require('cron-parser');
|
||||
try {
|
||||
const interval = CronExpressionParser.parse(config.SCRAPER_SCHEDULE, {
|
||||
tz: config.SCRAPER_TIMEZONE
|
||||
});
|
||||
return interval.next().toISOString();
|
||||
} catch (error) {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if schedule runs more frequently than 1 hour
|
||||
* @param {string} schedule - Cron expression
|
||||
* @returns {boolean} True if schedule is too frequent
|
||||
*/
|
||||
function isScheduleTooFrequent(schedule) {
|
||||
const parts = schedule.trim().split(/\s+/);
|
||||
|
||||
if (parts.length < 5) return false;
|
||||
|
||||
const minuteField = parts[0];
|
||||
|
||||
// If minute field is */N with N < 60, it runs more than once per hour
|
||||
if (/^\*\/\d+$/.test(minuteField)) {
|
||||
const interval = parseInt(minuteField.substring(2), 10);
|
||||
if (interval < 60) return true;
|
||||
}
|
||||
|
||||
// If minute field is *, it runs every minute
|
||||
if (minuteField === '*') return true;
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Initialize the scraper scheduler
|
||||
* @param {Db} db - MongoDB database instance
|
||||
*/
|
||||
function initializeScheduler(db) {
|
||||
const logger = createLogger('scheduler');
|
||||
|
||||
// Check if scheduling is enabled
|
||||
if (!config.SCRAPER_ENABLED) {
|
||||
logger.info('Scraper scheduling is disabled');
|
||||
return;
|
||||
}
|
||||
|
||||
// Validate cron expression
|
||||
if (!cron.validate(config.SCRAPER_SCHEDULE)) {
|
||||
logger.error('Invalid cron schedule expression', {
|
||||
schedule: config.SCRAPER_SCHEDULE
|
||||
});
|
||||
logger.warn('Falling back to default schedule: 0 6 * * *');
|
||||
config.SCRAPER_SCHEDULE = '0 6 * * *';
|
||||
}
|
||||
|
||||
// Validate minimum interval (1 hour)
|
||||
if (isScheduleTooFrequent(config.SCRAPER_SCHEDULE)) {
|
||||
logger.warn('Schedule interval less than 1 hour - adjusting to hourly', {
|
||||
originalSchedule: config.SCRAPER_SCHEDULE
|
||||
});
|
||||
config.SCRAPER_SCHEDULE = '0 * * * *';
|
||||
}
|
||||
|
||||
// Create the scheduled job
|
||||
scheduledJob = cron.schedule(config.SCRAPER_SCHEDULE, async () => {
|
||||
const jobId = uuidv4();
|
||||
const jobLogger = createLogger(jobId);
|
||||
|
||||
jobLogger.info('Scheduled scrape triggered');
|
||||
|
||||
// Check if already running
|
||||
if (!acquireLock(jobId)) {
|
||||
jobLogger.warn('Skipped - scrape already in progress');
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
await runScrape(db, { trigger: 'scheduled', jobId });
|
||||
} catch (error) {
|
||||
jobLogger.error('Scheduled scrape failed', {
|
||||
errorType: error.name,
|
||||
errorMessage: error.message
|
||||
});
|
||||
} finally {
|
||||
releaseLock();
|
||||
}
|
||||
}, {
|
||||
timezone: config.SCRAPER_TIMEZONE,
|
||||
scheduled: true
|
||||
});
|
||||
|
||||
logger.info('Scraper scheduler initialized', {
|
||||
schedule: config.SCRAPER_SCHEDULE,
|
||||
timezone: config.SCRAPER_TIMEZONE,
|
||||
nextRun: getNextScheduledRun()
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Stop the scheduler (for graceful shutdown)
|
||||
*/
|
||||
function stopScheduler() {
|
||||
if (scheduledJob) {
|
||||
scheduledJob.stop();
|
||||
scheduledJob = null;
|
||||
}
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
isScraperRunning,
|
||||
getCurrentJobId,
|
||||
acquireLock,
|
||||
releaseLock
|
||||
releaseLock,
|
||||
getScheduleExpression,
|
||||
getNextScheduledRun,
|
||||
initializeScheduler,
|
||||
stopScheduler
|
||||
};
|
||||
|
||||
Reference in New Issue
Block a user