SCRAPE-17: Implement GET /admin/scraper/status endpoint #22

Merged
stephen merged 1 commits from scraper/api-status into dev 2026-02-06 22:20:06 -07:00
2 changed files with 748 additions and 386 deletions
Showing only changes of commit 1c89181d87 - Show all commits

View File

@ -1,7 +1,8 @@
/** /**
* Tests for POST /api/admin/scraper/run endpoint * Tests for scraper admin route endpoints
* *
* Covers: * Covers:
* - POST /api/admin/scraper/run
* - Authentication and authorization (401, 403) * - Authentication and authorization (401, 403)
* - Successful scrape trigger (202 with jobId) * - Successful scrape trigger (202 with jobId)
* - Conflict when scraper already running (409) * - Conflict when scraper already running (409)
@ -10,10 +11,17 @@
* - dryRun option support * - dryRun option support
* - htmlContent option support * - htmlContent option support
* - Response structure validation * - Response structure validation
* - GET /api/admin/scraper/status
* - Authentication and authorization (401/403)
* - Correct status response structure
* - Idle vs running state
* - lastRun details from database
* - nextScheduledRun and schedule fields
* - 503 on database errors
*/ */
const request = require('supertest'); const request = require('supertest');
const { MongoClient } = require('mongodb'); const { MongoClient, ObjectId } = require('mongodb');
const { const {
createTestApp, createTestApp,
generateTestToken, generateTestToken,
@ -23,38 +31,52 @@ const {
cleanupTestData cleanupTestData
} = require('../helpers/testHelpers'); } = require('../helpers/testHelpers');
// We need to mock scraperJob functions to control mutex behavior // Mock the scraperJob module to control mutex state and schedule functions
jest.mock('../../jobs/scraperJob', () => { jest.mock('../../jobs/scraperJob', () => {
const actual = { let _isRunning = false;
_isRunning: false, let _currentJobId = null;
_currentJobId: null let _scheduleExpression = '0 6 * * *';
}; let _nextScheduledRun = '2026-02-07T06:00:00.000Z';
return { return {
isScraperRunning: jest.fn(() => actual._isRunning), isScraperRunning: jest.fn(() => _isRunning),
acquireLock: jest.fn((jobId) => { acquireLock: jest.fn((jobId) => {
if (actual._isRunning) return false; if (_isRunning) return false;
actual._isRunning = true; _isRunning = true;
actual._currentJobId = jobId; _currentJobId = jobId;
return true; return true;
}), }),
releaseLock: jest.fn(() => { releaseLock: jest.fn(() => {
actual._isRunning = false; _isRunning = false;
actual._currentJobId = null; _currentJobId = null;
}), }),
getCurrentJobId: jest.fn(() => actual._currentJobId), getCurrentJobId: jest.fn(() => _currentJobId),
getScheduleExpression: jest.fn(() => '0 6 * * *'), getScheduleExpression: jest.fn(() => _scheduleExpression),
getNextScheduledRun: jest.fn(() => null), getNextScheduledRun: jest.fn(() => _nextScheduledRun),
initializeScheduler: jest.fn(), initializeScheduler: jest.fn(),
stopScheduler: jest.fn(), stopScheduler: jest.fn(),
isShuttingDown: jest.fn(() => false), isShuttingDown: jest.fn(() => false),
gracefulShutdown: jest.fn(), gracefulShutdown: jest.fn(),
registerSignalHandlers: jest.fn(), registerSignalHandlers: jest.fn(),
// Expose internal state for test manipulation // Test helpers to control mock state
_setState: (running, jobId) => { _setState: (running, jobId) => {
actual._isRunning = running; _isRunning = running;
actual._currentJobId = jobId; _currentJobId = jobId || null;
}, },
_getState: () => actual __setRunning: (running, jobId) => {
_isRunning = running;
_currentJobId = jobId || null;
},
__setSchedule: (expression, nextRun) => {
_scheduleExpression = expression;
_nextScheduledRun = nextRun;
},
__reset: () => {
_isRunning = false;
_currentJobId = null;
_scheduleExpression = '0 6 * * *';
_nextScheduledRun = '2026-02-07T06:00:00.000Z';
}
}; };
}); });
@ -72,11 +94,29 @@ jest.mock('../../services/activityLogger', () => {
}; };
}); });
// Mock the scraper config for collection names
jest.mock('../../config/scraper', () => ({
TARGET_URL: 'https://example.com/test',
SCRAPER_SCHEDULE: '0 6 * * *',
SCRAPER_TIMEZONE: 'UTC',
SCRAPER_ENABLED: true,
SCRAPER_TIMEOUT: 30000,
USER_AGENT: 'TestBot/1.0',
RETRY_CONFIG: { maxRetries: 3, baseDelay: 1000, timeout: 30000 },
SHUTDOWN_TIMEOUT: 30000,
COLLECTIONS: {
UNITS: 'units_migration_test',
PRICES: 'unit_prices_migration_test',
DAILY_SUMMARIES: 'daily_summaries',
SCRAPER_RUNS: 'scraper_runs'
}
}));
const scraperJob = require('../../jobs/scraperJob'); const scraperJob = require('../../jobs/scraperJob');
const { runScrape } = require('../../services/scraperService'); const { runScrape } = require('../../services/scraperService');
const { logActivity } = require('../../services/activityLogger'); const { logActivity } = require('../../services/activityLogger');
describe('POST /api/admin/scraper/run', () => { describe('Scraper Routes', () => {
let connection; let connection;
let db; let db;
let app; let app;
@ -97,10 +137,12 @@ describe('POST /api/admin/scraper/run', () => {
beforeEach(async () => { beforeEach(async () => {
await cleanupTestData(db); await cleanupTestData(db);
// Also clean scraper_runs collection
await db.collection('scraper_runs').deleteMany({});
app = await createTestApp(db); app = await createTestApp(db);
// Reset scraper state to not running // Reset scraper state
scraperJob._setState(false, null); scraperJob.__reset();
// Clear mock call history (but keep implementations intact) // Clear mock call history (but keep implementations intact)
scraperJob.isScraperRunning.mockClear(); scraperJob.isScraperRunning.mockClear();
@ -113,6 +155,11 @@ describe('POST /api/admin/scraper/run', () => {
runScrape.mockImplementation(() => Promise.resolve({ status: 'success', jobId: 'test-job-id' })); runScrape.mockImplementation(() => Promise.resolve({ status: 'success', jobId: 'test-job-id' }));
}); });
// ============================================================
// POST /api/admin/scraper/run
// ============================================================
describe('POST /api/admin/scraper/run', () => {
// ============================================================ // ============================================================
// Authentication Tests // Authentication Tests
// ============================================================ // ============================================================
@ -213,7 +260,7 @@ describe('POST /api/admin/scraper/run', () => {
}); });
it('should return 409 when scraper is already running', async () => { it('should return 409 when scraper is already running', async () => {
// Set scraper as running (the mock reads from actual._isRunning) // Set scraper as running
scraperJob._setState(true, 'existing-job-id'); scraperJob._setState(true, 'existing-job-id');
const res = await request(app) const res = await request(app)
@ -526,3 +573,274 @@ describe('POST /api/admin/scraper/run', () => {
}); });
}); });
}); });
// ============================================================
// GET /api/admin/scraper/status
// ============================================================
describe('GET /api/admin/scraper/status', () => {
describe('Authentication and Authorization', () => {
it('should return 401 without authentication', async () => {
const res = await request(app)
.get('/api/admin/scraper/status')
.expect(401);
expect(res.body).toHaveProperty('error');
});
it('should return 403 for non-admin user', async () => {
const regularUser = createTestUser({ role: 'user' });
await insertTestUser(db, regularUser);
const token = generateTestToken(regularUser._id);
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(403);
expect(res.body).toHaveProperty('error');
expect(res.body.error).toBe('Admin access required');
});
});
describe('Successful status response (admin)', () => {
let admin;
let token;
beforeEach(async () => {
admin = createTestAdmin();
await insertTestUser(db, admin);
token = generateTestToken(admin._id);
});
it('should return 200 with status for admin', async () => {
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body).toHaveProperty('data');
expect(res.body.data).toHaveProperty('currentStatus');
expect(res.body.data).toHaveProperty('runningJobId');
expect(res.body.data).toHaveProperty('lastRun');
expect(res.body.data).toHaveProperty('nextScheduledRun');
expect(res.body.data).toHaveProperty('schedule');
});
it('should return currentStatus "idle" when scraper is not running', async () => {
scraperJob.__setRunning(false);
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.currentStatus).toBe('idle');
expect(res.body.data.runningJobId).toBeNull();
});
it('should return currentStatus "running" with runningJobId when scraper is running', async () => {
const jobId = 'test-job-id-abc-123';
scraperJob.__setRunning(true, jobId);
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.currentStatus).toBe('running');
expect(res.body.data.runningJobId).toBe(jobId);
});
it('should return lastRun as null when no runs exist', async () => {
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.lastRun).toBeNull();
});
it('should return lastRun object with correct fields when a run exists', async () => {
// Insert a scraper run record
await db.collection('scraper_runs').insertOne({
jobId: 'previous-job-001',
trigger: 'scheduled',
status: 'success',
startedAt: '2026-02-05T06:00:00.000Z',
completedAt: '2026-02-05T06:00:12.345Z',
duration: 12345,
unitsProcessed: 50,
pricesInserted: 48,
newUnitsCount: 2,
rentedUnitsCount: 1,
staleUnitsCount: 0,
errors: [],
recordedAt: new Date('2026-02-05T06:00:12.345Z')
});
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
const lastRun = res.body.data.lastRun;
expect(lastRun).not.toBeNull();
expect(lastRun.jobId).toBe('previous-job-001');
expect(lastRun.timestamp).toBe('2026-02-05T06:00:00.000Z');
expect(lastRun.status).toBe('success');
expect(lastRun.duration).toBe(12345);
expect(lastRun.trigger).toBe('scheduled');
expect(lastRun.unitsProcessed).toBe(50);
expect(lastRun.pricesInserted).toBe(48);
expect(lastRun.errors).toBeNull();
});
it('should return lastRun with errors array when last run had errors', async () => {
await db.collection('scraper_runs').insertOne({
jobId: 'failed-job-002',
trigger: 'manual',
status: 'failed',
startedAt: '2026-02-05T10:00:00.000Z',
completedAt: '2026-02-05T10:00:32.000Z',
duration: 32000,
unitsProcessed: 0,
pricesInserted: 0,
newUnitsCount: 0,
rentedUnitsCount: 0,
staleUnitsCount: 0,
errors: ['HTTP request failed after 3 retries'],
recordedAt: new Date('2026-02-05T10:00:32.000Z')
});
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
const lastRun = res.body.data.lastRun;
expect(lastRun.status).toBe('failed');
expect(lastRun.errors).toEqual(['HTTP request failed after 3 retries']);
});
it('should return the most recent run when multiple runs exist', async () => {
// Insert older run
await db.collection('scraper_runs').insertOne({
jobId: 'older-job-001',
trigger: 'scheduled',
status: 'success',
startedAt: '2026-02-04T06:00:00.000Z',
completedAt: '2026-02-04T06:00:10.000Z',
duration: 10000,
unitsProcessed: 45,
pricesInserted: 45,
errors: [],
recordedAt: new Date('2026-02-04T06:00:10.000Z')
});
// Insert newer run
await db.collection('scraper_runs').insertOne({
jobId: 'newer-job-002',
trigger: 'manual',
status: 'success',
startedAt: '2026-02-05T14:00:00.000Z',
completedAt: '2026-02-05T14:00:08.000Z',
duration: 8000,
unitsProcessed: 52,
pricesInserted: 50,
errors: [],
recordedAt: new Date('2026-02-05T14:00:08.000Z')
});
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.lastRun.jobId).toBe('newer-job-002');
expect(res.body.data.lastRun.trigger).toBe('manual');
});
it('should return nextScheduledRun as ISO timestamp', async () => {
scraperJob.__setSchedule('0 6 * * *', '2026-02-07T06:00:00.000Z');
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.nextScheduledRun).toBe('2026-02-07T06:00:00.000Z');
});
it('should return nextScheduledRun as null when scheduler is disabled', async () => {
scraperJob.__setSchedule('disabled', null);
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.nextScheduledRun).toBeNull();
expect(res.body.data.schedule).toBe('disabled');
});
it('should return schedule cron expression', async () => {
scraperJob.__setSchedule('0 6 * * *', '2026-02-07T06:00:00.000Z');
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.schedule).toBe('0 6 * * *');
});
});
describe('Error handling', () => {
let admin;
let token;
beforeEach(async () => {
admin = createTestAdmin();
await insertTestUser(db, admin);
token = generateTestToken(admin._id);
});
it('should return 503 on database error', async () => {
// Create a proxy db that works for auth (users collection)
// but throws errors for scraper_runs collection
const express = require('express');
const cookieParser = require('cookie-parser');
const brokenApp = express();
brokenApp.use(express.json());
brokenApp.use(cookieParser());
// Proxy db: real db for users, broken for scraper_runs
const proxyDb = {
collection: jest.fn((name) => {
if (name === 'scraper_runs') {
return {
findOne: jest.fn().mockRejectedValue(new Error('Database connection lost'))
};
}
// Delegate to real db for all other collections (users, etc.)
return db.collection(name);
})
};
brokenApp.locals.db = proxyDb;
const adminRoutes = require('../../routes/admin');
brokenApp.use('/api/admin', adminRoutes);
const res = await request(brokenApp)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(503);
expect(res.body).toHaveProperty('error');
expect(res.body.error).toBe('Service temporarily unavailable');
});
});
});
});

View File

@ -1150,9 +1150,13 @@ const crypto = require('crypto');
const { const {
isScraperRunning, isScraperRunning,
acquireLock, acquireLock,
releaseLock releaseLock,
getCurrentJobId,
getScheduleExpression,
getNextScheduledRun
} = require('../jobs/scraperJob'); } = require('../jobs/scraperJob');
const { runScrape } = require('../services/scraperService'); const { runScrape } = require('../services/scraperService');
const scraperConfig = require('../config/scraper');
/** /**
* POST /api/admin/scraper/run * POST /api/admin/scraper/run
@ -1206,5 +1210,45 @@ router.post('/scraper/run', async (req, res) => {
} }
}); });
/**
* GET /api/admin/scraper/status
* Get current scraper status including running state, last run details,
* next scheduled run, and schedule expression.
*/
router.get('/scraper/status', async (req, res) => {
try {
const db = req.app.locals.db;
// Get last run from history
const lastRunDoc = await db.collection(scraperConfig.COLLECTIONS.SCRAPER_RUNS)
.findOne({}, { sort: { startedAt: -1 } });
const lastRun = lastRunDoc ? {
jobId: lastRunDoc.jobId,
timestamp: lastRunDoc.startedAt,
status: lastRunDoc.status,
duration: lastRunDoc.duration,
trigger: lastRunDoc.trigger,
unitsProcessed: lastRunDoc.unitsProcessed,
pricesInserted: lastRunDoc.pricesInserted,
errors: lastRunDoc.errors?.length > 0 ? lastRunDoc.errors : null
} : null;
res.json({
data: {
currentStatus: isScraperRunning() ? 'running' : 'idle',
runningJobId: isScraperRunning() ? getCurrentJobId() : null,
lastRun,
nextScheduledRun: getNextScheduledRun(),
schedule: getScheduleExpression()
}
});
} catch (error) {
console.error('Error fetching scraper status:', error);
res.status(503).json({ error: 'Service temporarily unavailable' });
}
});
module.exports = router; module.exports = router;
module.exports.clearStatsCache = clearStatsCache; module.exports.clearStatsCache = clearStatsCache;