SCRAPE-17: Implement GET /admin/scraper/status endpoint #22

Merged
stephen merged 1 commits from scraper/api-status into dev 2026-02-06 22:20:06 -07:00
2 changed files with 748 additions and 386 deletions
Showing only changes of commit 1c89181d87 - Show all commits

View File

@ -1,7 +1,8 @@
/**
* Tests for POST /api/admin/scraper/run endpoint
* Tests for scraper admin route endpoints
*
* Covers:
* - POST /api/admin/scraper/run
* - Authentication and authorization (401, 403)
* - Successful scrape trigger (202 with jobId)
* - Conflict when scraper already running (409)
@ -10,10 +11,17 @@
* - dryRun option support
* - htmlContent option support
* - Response structure validation
* - GET /api/admin/scraper/status
* - Authentication and authorization (401/403)
* - Correct status response structure
* - Idle vs running state
* - lastRun details from database
* - nextScheduledRun and schedule fields
* - 503 on database errors
*/
const request = require('supertest');
const { MongoClient } = require('mongodb');
const { MongoClient, ObjectId } = require('mongodb');
const {
createTestApp,
generateTestToken,
@ -23,38 +31,52 @@ const {
cleanupTestData
} = require('../helpers/testHelpers');
// We need to mock scraperJob functions to control mutex behavior
// Mock the scraperJob module to control mutex state and schedule functions
jest.mock('../../jobs/scraperJob', () => {
const actual = {
_isRunning: false,
_currentJobId: null
};
let _isRunning = false;
let _currentJobId = null;
let _scheduleExpression = '0 6 * * *';
let _nextScheduledRun = '2026-02-07T06:00:00.000Z';
return {
isScraperRunning: jest.fn(() => actual._isRunning),
isScraperRunning: jest.fn(() => _isRunning),
acquireLock: jest.fn((jobId) => {
if (actual._isRunning) return false;
actual._isRunning = true;
actual._currentJobId = jobId;
if (_isRunning) return false;
_isRunning = true;
_currentJobId = jobId;
return true;
}),
releaseLock: jest.fn(() => {
actual._isRunning = false;
actual._currentJobId = null;
_isRunning = false;
_currentJobId = null;
}),
getCurrentJobId: jest.fn(() => actual._currentJobId),
getScheduleExpression: jest.fn(() => '0 6 * * *'),
getNextScheduledRun: jest.fn(() => null),
getCurrentJobId: jest.fn(() => _currentJobId),
getScheduleExpression: jest.fn(() => _scheduleExpression),
getNextScheduledRun: jest.fn(() => _nextScheduledRun),
initializeScheduler: jest.fn(),
stopScheduler: jest.fn(),
isShuttingDown: jest.fn(() => false),
gracefulShutdown: jest.fn(),
registerSignalHandlers: jest.fn(),
// Expose internal state for test manipulation
// Test helpers to control mock state
_setState: (running, jobId) => {
actual._isRunning = running;
actual._currentJobId = jobId;
_isRunning = running;
_currentJobId = jobId || null;
},
_getState: () => actual
__setRunning: (running, jobId) => {
_isRunning = running;
_currentJobId = jobId || null;
},
__setSchedule: (expression, nextRun) => {
_scheduleExpression = expression;
_nextScheduledRun = nextRun;
},
__reset: () => {
_isRunning = false;
_currentJobId = null;
_scheduleExpression = '0 6 * * *';
_nextScheduledRun = '2026-02-07T06:00:00.000Z';
}
};
});
@ -72,11 +94,29 @@ jest.mock('../../services/activityLogger', () => {
};
});
// Mock the scraper config for collection names
jest.mock('../../config/scraper', () => ({
TARGET_URL: 'https://example.com/test',
SCRAPER_SCHEDULE: '0 6 * * *',
SCRAPER_TIMEZONE: 'UTC',
SCRAPER_ENABLED: true,
SCRAPER_TIMEOUT: 30000,
USER_AGENT: 'TestBot/1.0',
RETRY_CONFIG: { maxRetries: 3, baseDelay: 1000, timeout: 30000 },
SHUTDOWN_TIMEOUT: 30000,
COLLECTIONS: {
UNITS: 'units_migration_test',
PRICES: 'unit_prices_migration_test',
DAILY_SUMMARIES: 'daily_summaries',
SCRAPER_RUNS: 'scraper_runs'
}
}));
const scraperJob = require('../../jobs/scraperJob');
const { runScrape } = require('../../services/scraperService');
const { logActivity } = require('../../services/activityLogger');
describe('POST /api/admin/scraper/run', () => {
describe('Scraper Routes', () => {
let connection;
let db;
let app;
@ -97,10 +137,12 @@ describe('POST /api/admin/scraper/run', () => {
beforeEach(async () => {
await cleanupTestData(db);
// Also clean scraper_runs collection
await db.collection('scraper_runs').deleteMany({});
app = await createTestApp(db);
// Reset scraper state to not running
scraperJob._setState(false, null);
// Reset scraper state
scraperJob.__reset();
// Clear mock call history (but keep implementations intact)
scraperJob.isScraperRunning.mockClear();
@ -113,6 +155,11 @@ describe('POST /api/admin/scraper/run', () => {
runScrape.mockImplementation(() => Promise.resolve({ status: 'success', jobId: 'test-job-id' }));
});
// ============================================================
// POST /api/admin/scraper/run
// ============================================================
describe('POST /api/admin/scraper/run', () => {
// ============================================================
// Authentication Tests
// ============================================================
@ -213,7 +260,7 @@ describe('POST /api/admin/scraper/run', () => {
});
it('should return 409 when scraper is already running', async () => {
// Set scraper as running (the mock reads from actual._isRunning)
// Set scraper as running
scraperJob._setState(true, 'existing-job-id');
const res = await request(app)
@ -525,4 +572,275 @@ describe('POST /api/admin/scraper/run', () => {
await new Promise(resolve => setTimeout(resolve, 100));
});
});
});
// ============================================================
// GET /api/admin/scraper/status
// ============================================================
describe('GET /api/admin/scraper/status', () => {
describe('Authentication and Authorization', () => {
it('should return 401 without authentication', async () => {
const res = await request(app)
.get('/api/admin/scraper/status')
.expect(401);
expect(res.body).toHaveProperty('error');
});
it('should return 403 for non-admin user', async () => {
const regularUser = createTestUser({ role: 'user' });
await insertTestUser(db, regularUser);
const token = generateTestToken(regularUser._id);
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(403);
expect(res.body).toHaveProperty('error');
expect(res.body.error).toBe('Admin access required');
});
});
describe('Successful status response (admin)', () => {
let admin;
let token;
beforeEach(async () => {
admin = createTestAdmin();
await insertTestUser(db, admin);
token = generateTestToken(admin._id);
});
it('should return 200 with status for admin', async () => {
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body).toHaveProperty('data');
expect(res.body.data).toHaveProperty('currentStatus');
expect(res.body.data).toHaveProperty('runningJobId');
expect(res.body.data).toHaveProperty('lastRun');
expect(res.body.data).toHaveProperty('nextScheduledRun');
expect(res.body.data).toHaveProperty('schedule');
});
it('should return currentStatus "idle" when scraper is not running', async () => {
scraperJob.__setRunning(false);
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.currentStatus).toBe('idle');
expect(res.body.data.runningJobId).toBeNull();
});
it('should return currentStatus "running" with runningJobId when scraper is running', async () => {
const jobId = 'test-job-id-abc-123';
scraperJob.__setRunning(true, jobId);
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.currentStatus).toBe('running');
expect(res.body.data.runningJobId).toBe(jobId);
});
it('should return lastRun as null when no runs exist', async () => {
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.lastRun).toBeNull();
});
it('should return lastRun object with correct fields when a run exists', async () => {
// Insert a scraper run record
await db.collection('scraper_runs').insertOne({
jobId: 'previous-job-001',
trigger: 'scheduled',
status: 'success',
startedAt: '2026-02-05T06:00:00.000Z',
completedAt: '2026-02-05T06:00:12.345Z',
duration: 12345,
unitsProcessed: 50,
pricesInserted: 48,
newUnitsCount: 2,
rentedUnitsCount: 1,
staleUnitsCount: 0,
errors: [],
recordedAt: new Date('2026-02-05T06:00:12.345Z')
});
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
const lastRun = res.body.data.lastRun;
expect(lastRun).not.toBeNull();
expect(lastRun.jobId).toBe('previous-job-001');
expect(lastRun.timestamp).toBe('2026-02-05T06:00:00.000Z');
expect(lastRun.status).toBe('success');
expect(lastRun.duration).toBe(12345);
expect(lastRun.trigger).toBe('scheduled');
expect(lastRun.unitsProcessed).toBe(50);
expect(lastRun.pricesInserted).toBe(48);
expect(lastRun.errors).toBeNull();
});
it('should return lastRun with errors array when last run had errors', async () => {
await db.collection('scraper_runs').insertOne({
jobId: 'failed-job-002',
trigger: 'manual',
status: 'failed',
startedAt: '2026-02-05T10:00:00.000Z',
completedAt: '2026-02-05T10:00:32.000Z',
duration: 32000,
unitsProcessed: 0,
pricesInserted: 0,
newUnitsCount: 0,
rentedUnitsCount: 0,
staleUnitsCount: 0,
errors: ['HTTP request failed after 3 retries'],
recordedAt: new Date('2026-02-05T10:00:32.000Z')
});
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
const lastRun = res.body.data.lastRun;
expect(lastRun.status).toBe('failed');
expect(lastRun.errors).toEqual(['HTTP request failed after 3 retries']);
});
it('should return the most recent run when multiple runs exist', async () => {
// Insert older run
await db.collection('scraper_runs').insertOne({
jobId: 'older-job-001',
trigger: 'scheduled',
status: 'success',
startedAt: '2026-02-04T06:00:00.000Z',
completedAt: '2026-02-04T06:00:10.000Z',
duration: 10000,
unitsProcessed: 45,
pricesInserted: 45,
errors: [],
recordedAt: new Date('2026-02-04T06:00:10.000Z')
});
// Insert newer run
await db.collection('scraper_runs').insertOne({
jobId: 'newer-job-002',
trigger: 'manual',
status: 'success',
startedAt: '2026-02-05T14:00:00.000Z',
completedAt: '2026-02-05T14:00:08.000Z',
duration: 8000,
unitsProcessed: 52,
pricesInserted: 50,
errors: [],
recordedAt: new Date('2026-02-05T14:00:08.000Z')
});
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.lastRun.jobId).toBe('newer-job-002');
expect(res.body.data.lastRun.trigger).toBe('manual');
});
it('should return nextScheduledRun as ISO timestamp', async () => {
scraperJob.__setSchedule('0 6 * * *', '2026-02-07T06:00:00.000Z');
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.nextScheduledRun).toBe('2026-02-07T06:00:00.000Z');
});
it('should return nextScheduledRun as null when scheduler is disabled', async () => {
scraperJob.__setSchedule('disabled', null);
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.nextScheduledRun).toBeNull();
expect(res.body.data.schedule).toBe('disabled');
});
it('should return schedule cron expression', async () => {
scraperJob.__setSchedule('0 6 * * *', '2026-02-07T06:00:00.000Z');
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.schedule).toBe('0 6 * * *');
});
});
describe('Error handling', () => {
let admin;
let token;
beforeEach(async () => {
admin = createTestAdmin();
await insertTestUser(db, admin);
token = generateTestToken(admin._id);
});
it('should return 503 on database error', async () => {
// Create a proxy db that works for auth (users collection)
// but throws errors for scraper_runs collection
const express = require('express');
const cookieParser = require('cookie-parser');
const brokenApp = express();
brokenApp.use(express.json());
brokenApp.use(cookieParser());
// Proxy db: real db for users, broken for scraper_runs
const proxyDb = {
collection: jest.fn((name) => {
if (name === 'scraper_runs') {
return {
findOne: jest.fn().mockRejectedValue(new Error('Database connection lost'))
};
}
// Delegate to real db for all other collections (users, etc.)
return db.collection(name);
})
};
brokenApp.locals.db = proxyDb;
const adminRoutes = require('../../routes/admin');
brokenApp.use('/api/admin', adminRoutes);
const res = await request(brokenApp)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(503);
expect(res.body).toHaveProperty('error');
expect(res.body.error).toBe('Service temporarily unavailable');
});
});
});
});

View File

@ -1150,9 +1150,13 @@ const crypto = require('crypto');
const {
isScraperRunning,
acquireLock,
releaseLock
releaseLock,
getCurrentJobId,
getScheduleExpression,
getNextScheduledRun
} = require('../jobs/scraperJob');
const { runScrape } = require('../services/scraperService');
const scraperConfig = require('../config/scraper');
/**
* POST /api/admin/scraper/run
@ -1206,5 +1210,45 @@ router.post('/scraper/run', async (req, res) => {
}
});
/**
* GET /api/admin/scraper/status
* Get current scraper status including running state, last run details,
* next scheduled run, and schedule expression.
*/
router.get('/scraper/status', async (req, res) => {
try {
const db = req.app.locals.db;
// Get last run from history
const lastRunDoc = await db.collection(scraperConfig.COLLECTIONS.SCRAPER_RUNS)
.findOne({}, { sort: { startedAt: -1 } });
const lastRun = lastRunDoc ? {
jobId: lastRunDoc.jobId,
timestamp: lastRunDoc.startedAt,
status: lastRunDoc.status,
duration: lastRunDoc.duration,
trigger: lastRunDoc.trigger,
unitsProcessed: lastRunDoc.unitsProcessed,
pricesInserted: lastRunDoc.pricesInserted,
errors: lastRunDoc.errors?.length > 0 ? lastRunDoc.errors : null
} : null;
res.json({
data: {
currentStatus: isScraperRunning() ? 'running' : 'idle',
runningJobId: isScraperRunning() ? getCurrentJobId() : null,
lastRun,
nextScheduledRun: getNextScheduledRun(),
schedule: getScheduleExpression()
}
});
} catch (error) {
console.error('Error fetching scraper status:', error);
res.status(503).json({ error: 'Service temporarily unavailable' });
}
});
module.exports = router;
module.exports.clearStatsCache = clearStatsCache;