feat: add POST /api/admin/scraper/run endpoint
Some checks failed
CI/CD Pipeline - Apartment API / Scan Dependencies (pull_request) Successful in 13s
CI/CD Pipeline - Apartment API / Lint & Test (pull_request) Successful in 41s
CI/CD Pipeline - Apartment API / Send Webhook Notification (pull_request) Failing after 1s
CI/CD Pipeline - Apartment API / Build & Push Image (pull_request) Has been skipped
CI/CD Pipeline - Apartment API / Deploy to Production (pull_request) Has been skipped

Implement the manual scraper trigger endpoint with the following:

- Protected by requireAuth and requireAdmin middleware (401/403)
- Mutex lock check via isScraperRunning to prevent concurrent runs (409)
- Generates UUID jobId for tracking async scrape execution
- Returns 202 Accepted immediately without blocking on scrape completion
- Releases mutex lock in .finally() to ensure cleanup on success or failure
- Logs ADMIN_TRIGGER_SCRAPE activity with jobId, dryRun, and
  usingProvidedHtml metadata via activityLogger
- Supports dryRun option (defaults to false) and htmlContent for
  testing with pre-fetched HTML
- Passes trigger: 'manual', jobId, dryRun, and htmlContent to runScrape

Includes 20 tests covering auth, mutex, async execution, activity
logging, dryRun/htmlContent options, and response structure.
This commit is contained in:
2026-02-06 21:20:42 -07:00
parent c6d480a870
commit c30b02681e
2 changed files with 592 additions and 0 deletions

View File

@ -0,0 +1,528 @@
/**
* Tests for POST /api/admin/scraper/run endpoint
*
* Covers:
* - Authentication and authorization (401, 403)
* - Successful scrape trigger (202 with jobId)
* - Conflict when scraper already running (409)
* - Mutex lock acquisition and release
* - Activity logging for admin trigger
* - dryRun option support
* - htmlContent option support
* - Response structure validation
*/
const request = require('supertest');
const { MongoClient } = require('mongodb');
const {
createTestApp,
generateTestToken,
createTestAdmin,
createTestUser,
insertTestUser,
cleanupTestData
} = require('../helpers/testHelpers');
// We need to mock scraperJob functions to control mutex behavior
jest.mock('../../jobs/scraperJob', () => {
const actual = {
_isRunning: false,
_currentJobId: null
};
return {
isScraperRunning: jest.fn(() => actual._isRunning),
acquireLock: jest.fn((jobId) => {
if (actual._isRunning) return false;
actual._isRunning = true;
actual._currentJobId = jobId;
return true;
}),
releaseLock: jest.fn(() => {
actual._isRunning = false;
actual._currentJobId = null;
}),
getCurrentJobId: jest.fn(() => actual._currentJobId),
getScheduleExpression: jest.fn(() => '0 6 * * *'),
getNextScheduledRun: jest.fn(() => null),
initializeScheduler: jest.fn(),
stopScheduler: jest.fn(),
isShuttingDown: jest.fn(() => false),
gracefulShutdown: jest.fn(),
registerSignalHandlers: jest.fn(),
// Expose internal state for test manipulation
_setState: (running, jobId) => {
actual._isRunning = running;
actual._currentJobId = jobId;
},
_getState: () => actual
};
});
// Mock scraperService to prevent actual scraping
jest.mock('../../services/scraperService', () => ({
runScrape: jest.fn(() => Promise.resolve({ status: 'success', jobId: 'test-job-id' }))
}));
// Mock activityLogger to track calls
jest.mock('../../services/activityLogger', () => {
const originalModule = jest.requireActual('../../services/activityLogger');
return {
...originalModule,
logActivity: jest.fn(() => Promise.resolve())
};
});
const scraperJob = require('../../jobs/scraperJob');
const { runScrape } = require('../../services/scraperService');
const { logActivity } = require('../../services/activityLogger');
describe('POST /api/admin/scraper/run', () => {
let connection;
let db;
let app;
beforeAll(async () => {
const uri = process.env.MONGO_URI;
connection = await MongoClient.connect(uri);
db = connection.db('apartments_scraper_routes_test');
});
afterAll(async () => {
// Allow pending async operations (e.g., .finally() callbacks) to complete
await new Promise(resolve => setTimeout(resolve, 200));
if (connection) {
await connection.close();
}
});
beforeEach(async () => {
await cleanupTestData(db);
app = await createTestApp(db);
// Reset scraper state to not running
scraperJob._setState(false, null);
// Clear mock call history (but keep implementations intact)
scraperJob.isScraperRunning.mockClear();
scraperJob.acquireLock.mockClear();
scraperJob.releaseLock.mockClear();
runScrape.mockClear();
logActivity.mockClear();
// Restore default mock implementations
runScrape.mockImplementation(() => Promise.resolve({ status: 'success', jobId: 'test-job-id' }));
});
// ============================================================
// Authentication Tests
// ============================================================
describe('authentication and authorization', () => {
it('should return 401 without auth token', async () => {
const res = await request(app)
.post('/api/admin/scraper/run')
.send({});
expect(res.status).toBe(401);
expect(res.body).toHaveProperty('error');
});
it('should return 403 for non-admin user', async () => {
const regularUser = createTestUser({ role: 'user' });
await insertTestUser(db, regularUser);
const token = generateTestToken(regularUser._id);
const res = await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${token}`])
.send({});
expect(res.status).toBe(403);
expect(res.body).toHaveProperty('error');
expect(res.body.error).toMatch(/admin/i);
});
});
// ============================================================
// Successful Trigger Tests
// ============================================================
describe('successful scrape trigger', () => {
let adminUser;
let adminToken;
beforeEach(async () => {
adminUser = createTestAdmin();
await insertTestUser(db, adminUser);
adminToken = generateTestToken(adminUser._id);
});
it('should return 202 with jobId for admin user', async () => {
const res = await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({});
expect(res.status).toBe(202);
expect(res.body).toHaveProperty('data');
expect(res.body.data).toHaveProperty('jobId');
expect(res.body.data).toHaveProperty('status', 'started');
expect(res.body.data).toHaveProperty('message');
});
it('should return response with correct structure (jobId, status, message, dryRun)', async () => {
const res = await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({});
expect(res.status).toBe(202);
const { data } = res.body;
expect(data.jobId).toBeDefined();
expect(typeof data.jobId).toBe('string');
expect(data.jobId.length).toBeGreaterThan(0);
expect(data.status).toBe('started');
expect(typeof data.message).toBe('string');
expect(data.dryRun).toBe(false);
});
it('should return a UUID-formatted jobId', async () => {
const res = await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({});
expect(res.status).toBe(202);
// UUID v4 format: xxxxxxxx-xxxx-4xxx-yxxx-xxxxxxxxxxxx
const uuidRegex = /^[0-9a-f]{8}-[0-9a-f]{4}-4[0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/i;
expect(res.body.data.jobId).toMatch(uuidRegex);
});
});
// ============================================================
// Conflict Tests (409)
// ============================================================
describe('scraper already running', () => {
let adminUser;
let adminToken;
beforeEach(async () => {
adminUser = createTestAdmin();
await insertTestUser(db, adminUser);
adminToken = generateTestToken(adminUser._id);
});
it('should return 409 when scraper is already running', async () => {
// Set scraper as running (the mock reads from actual._isRunning)
scraperJob._setState(true, 'existing-job-id');
const res = await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({});
expect(res.status).toBe(409);
expect(res.body).toHaveProperty('error');
expect(res.body.error).toMatch(/already in progress/i);
});
it('should check isScraperRunning before starting', async () => {
await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({});
expect(scraperJob.isScraperRunning).toHaveBeenCalled();
});
});
// ============================================================
// Lock Acquisition Tests
// ============================================================
describe('mutex lock behavior', () => {
let adminUser;
let adminToken;
beforeEach(async () => {
adminUser = createTestAdmin();
await insertTestUser(db, adminUser);
adminToken = generateTestToken(adminUser._id);
});
it('should acquire lock before starting scrape', async () => {
const res = await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({});
expect(res.status).toBe(202);
expect(scraperJob.acquireLock).toHaveBeenCalledWith(expect.any(String));
});
it('should release lock after scrape completes', async () => {
// Make runScrape resolve quickly
runScrape.mockImplementation(() => Promise.resolve({ status: 'success' }));
const res = await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({});
expect(res.status).toBe(202);
// Wait for the async scrape to complete
await new Promise(resolve => setTimeout(resolve, 100));
// The lock should be released via the .finally() callback
expect(scraperJob.releaseLock).toHaveBeenCalled();
});
it('should release lock even when scrape fails', async () => {
// Make runScrape reject
runScrape.mockImplementation(() => Promise.reject(new Error('Scrape failed')));
const res = await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({});
expect(res.status).toBe(202);
// Wait for the async scrape to complete
await new Promise(resolve => setTimeout(resolve, 100));
// Lock should still be released in finally block
expect(scraperJob.releaseLock).toHaveBeenCalled();
});
});
// ============================================================
// Activity Logging Tests
// ============================================================
describe('activity logging', () => {
let adminUser;
let adminToken;
beforeEach(async () => {
adminUser = createTestAdmin();
await insertTestUser(db, adminUser);
adminToken = generateTestToken(adminUser._id);
});
it('should log ADMIN_TRIGGER_SCRAPE activity', async () => {
const res = await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({});
expect(res.status).toBe(202);
expect(logActivity).toHaveBeenCalledWith(
expect.anything(), // db
expect.objectContaining({
userId: adminUser._id.toString(),
action: 'ADMIN_TRIGGER_SCRAPE',
metadata: expect.objectContaining({
jobId: expect.any(String),
dryRun: false
})
})
);
});
it('should log dryRun flag in activity metadata', async () => {
const res = await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({ dryRun: true });
expect(res.status).toBe(202);
expect(logActivity).toHaveBeenCalledWith(
expect.anything(),
expect.objectContaining({
action: 'ADMIN_TRIGGER_SCRAPE',
metadata: expect.objectContaining({
dryRun: true
})
})
);
});
});
// ============================================================
// dryRun Option Tests
// ============================================================
describe('dryRun option', () => {
let adminUser;
let adminToken;
beforeEach(async () => {
adminUser = createTestAdmin();
await insertTestUser(db, adminUser);
adminToken = generateTestToken(adminUser._id);
});
it('should handle dryRun option in request body', async () => {
const res = await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({ dryRun: true });
expect(res.status).toBe(202);
expect(res.body.data.dryRun).toBe(true);
expect(res.body.data.message).toMatch(/dry run/i);
});
it('should pass dryRun option to runScrape', async () => {
await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({ dryRun: true });
// Wait for async execution
await new Promise(resolve => setTimeout(resolve, 100));
expect(runScrape).toHaveBeenCalledWith(
expect.anything(), // db
expect.objectContaining({
dryRun: true
})
);
});
it('should default dryRun to false when not provided', async () => {
const res = await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({});
expect(res.status).toBe(202);
expect(res.body.data.dryRun).toBe(false);
});
});
// ============================================================
// htmlContent Option Tests
// ============================================================
describe('htmlContent option', () => {
let adminUser;
let adminToken;
beforeEach(async () => {
adminUser = createTestAdmin();
await insertTestUser(db, adminUser);
adminToken = generateTestToken(adminUser._id);
});
it('should pass htmlContent option to runScrape', async () => {
const testHtml = '<html><body>Test content</body></html>';
await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({ htmlContent: testHtml });
// Wait for async execution
await new Promise(resolve => setTimeout(resolve, 100));
expect(runScrape).toHaveBeenCalledWith(
expect.anything(),
expect.objectContaining({
htmlContent: testHtml
})
);
});
it('should log usingProvidedHtml in activity metadata when htmlContent provided', async () => {
const testHtml = '<html><body>Test</body></html>';
await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({ htmlContent: testHtml });
expect(logActivity).toHaveBeenCalledWith(
expect.anything(),
expect.objectContaining({
metadata: expect.objectContaining({
usingProvidedHtml: true
})
})
);
});
});
// ============================================================
// runScrape Integration Tests
// ============================================================
describe('scrape execution', () => {
let adminUser;
let adminToken;
beforeEach(async () => {
adminUser = createTestAdmin();
await insertTestUser(db, adminUser);
adminToken = generateTestToken(adminUser._id);
});
it('should call runScrape with trigger "manual"', async () => {
await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({});
// Wait for async execution
await new Promise(resolve => setTimeout(resolve, 100));
expect(runScrape).toHaveBeenCalledWith(
expect.anything(),
expect.objectContaining({
trigger: 'manual'
})
);
});
it('should call runScrape with the generated jobId', async () => {
const res = await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({});
const { jobId } = res.body.data;
// Wait for async execution
await new Promise(resolve => setTimeout(resolve, 100));
expect(runScrape).toHaveBeenCalledWith(
expect.anything(),
expect.objectContaining({
jobId
})
);
});
it('should return 202 immediately without waiting for scrape to finish', async () => {
// Make runScrape take a "long time" (500ms is enough to prove async behavior)
let scrapeResolve;
runScrape.mockImplementation(() => new Promise(resolve => {
scrapeResolve = resolve;
setTimeout(() => resolve({ status: 'success' }), 500);
}));
const startTime = Date.now();
const res = await request(app)
.post('/api/admin/scraper/run')
.set('Cookie', [`auth_token=${adminToken}`])
.send({});
const elapsed = Date.now() - startTime;
// Should respond in well under 200ms despite scrape taking 500ms
expect(res.status).toBe(202);
expect(elapsed).toBeLessThan(200);
// Resolve the scrape to clean up
if (scrapeResolve) scrapeResolve({ status: 'success' });
await new Promise(resolve => setTimeout(resolve, 100));
});
});
});

View File

@ -1142,5 +1142,69 @@ router.patch('/settings', async (req, res) => {
}
});
// ============================================================
// Scraper Endpoints
// ============================================================
const crypto = require('crypto');
const {
isScraperRunning,
acquireLock,
releaseLock
} = require('../jobs/scraperJob');
const { runScrape } = require('../services/scraperService');
/**
* POST /api/admin/scraper/run
* Trigger a manual scrape
*
* Request body (optional):
* - dryRun: boolean - Skip database writes for safe testing
* - htmlContent: string - Use provided HTML instead of fetching (for testing/debugging)
*/
router.post('/scraper/run', async (req, res) => {
try {
const db = req.app.locals.db;
const { dryRun = false, htmlContent = null } = req.body || {};
// Check if scraper is already running
if (isScraperRunning()) {
return res.status(409).json({ error: 'Scrape already in progress' });
}
// Generate job ID and acquire lock
const jobId = crypto.randomUUID();
acquireLock(jobId);
// Log admin action
await logActivity(db, {
userId: req.user._id.toString(),
action: 'ADMIN_TRIGGER_SCRAPE',
metadata: { jobId, dryRun, usingProvidedHtml: !!htmlContent }
});
// Start scrape asynchronously (do not await - return 202 immediately)
runScrape(db, { trigger: 'manual', jobId, dryRun, htmlContent })
.catch((error) => {
console.error('Async scrape failed:', error.message);
})
.finally(() => releaseLock());
// Return immediately with job ID
res.status(202).json({
data: {
jobId,
status: 'started',
dryRun,
message: dryRun ? 'Scrape job initiated (dry run - no DB writes)' : 'Scrape job initiated'
}
});
} catch (error) {
console.error('Error triggering scrape:', error);
res.status(500).json({ error: 'Failed to start scrape job' });
}
});
module.exports = router;
module.exports.clearStatsCache = clearStatsCache;