feat: add GET /api/admin/scraper/status endpoint
Some checks failed
CI/CD Pipeline - Apartment API / Scan Dependencies (pull_request) Successful in 12s
CI/CD Pipeline - Apartment API / Lint & Test (pull_request) Successful in 40s
CI/CD Pipeline - Apartment API / Send Webhook Notification (pull_request) Failing after 1s
CI/CD Pipeline - Apartment API / Build & Push Image (pull_request) Has been skipped
CI/CD Pipeline - Apartment API / Deploy to Production (pull_request) Has been skipped

Implement a status endpoint that returns the current state of the
scraper system. The response includes mutex state (idle/running with
job ID), last run history from scraper_runs collection (status,
timing, unit/error counts), next scheduled run timestamp, and
cron schedule expression.

Protected by requireAuth + requireAdmin middleware. Returns 503
on database errors for graceful degradation. Includes 13 tests
covering auth, response structure, edge cases, and error handling.
This commit is contained in:
2026-02-06 21:30:40 -07:00
parent c6d480a870
commit 2b36ef4e23
2 changed files with 424 additions and 0 deletions

View File

@ -0,0 +1,372 @@
/**
* Tests for scraper admin route endpoints
*
* Tests GET /api/admin/scraper/status endpoint for:
* - Authentication and authorization (401/403)
* - Correct status response structure
* - Idle vs running state
* - lastRun details from database
* - nextScheduledRun and schedule fields
* - 503 on database errors
*/
const request = require('supertest');
const { MongoClient, ObjectId } = require('mongodb');
const {
createTestApp,
generateTestToken,
createTestAdmin,
createTestUser,
insertTestUser,
cleanupTestData
} = require('../helpers/testHelpers');
// Mock the scraperJob module to control mutex state and schedule functions
jest.mock('../../jobs/scraperJob', () => {
let _isRunning = false;
let _currentJobId = null;
let _scheduleExpression = '0 6 * * *';
let _nextScheduledRun = '2026-02-07T06:00:00.000Z';
return {
isScraperRunning: jest.fn(() => _isRunning),
getCurrentJobId: jest.fn(() => _currentJobId),
getScheduleExpression: jest.fn(() => _scheduleExpression),
getNextScheduledRun: jest.fn(() => _nextScheduledRun),
acquireLock: jest.fn(),
releaseLock: jest.fn(),
initializeScheduler: jest.fn(),
stopScheduler: jest.fn(),
// Test helpers to control mock state
__setRunning: (running, jobId) => {
_isRunning = running;
_currentJobId = jobId || null;
},
__setSchedule: (expression, nextRun) => {
_scheduleExpression = expression;
_nextScheduledRun = nextRun;
},
__reset: () => {
_isRunning = false;
_currentJobId = null;
_scheduleExpression = '0 6 * * *';
_nextScheduledRun = '2026-02-07T06:00:00.000Z';
}
};
});
// Mock the scraper config for collection names
jest.mock('../../config/scraper', () => ({
TARGET_URL: 'https://example.com/test',
SCRAPER_SCHEDULE: '0 6 * * *',
SCRAPER_TIMEZONE: 'UTC',
SCRAPER_ENABLED: true,
SCRAPER_TIMEOUT: 30000,
USER_AGENT: 'TestBot/1.0',
RETRY_CONFIG: { maxRetries: 3, baseDelay: 1000, timeout: 30000 },
SHUTDOWN_TIMEOUT: 30000,
COLLECTIONS: {
UNITS: 'units_migration_test',
PRICES: 'unit_prices_migration_test',
DAILY_SUMMARIES: 'daily_summaries',
SCRAPER_RUNS: 'scraper_runs'
}
}));
const scraperJob = require('../../jobs/scraperJob');
describe('Scraper Routes', () => {
let connection;
let db;
let app;
beforeAll(async () => {
const uri = process.env.MONGO_URI;
connection = await MongoClient.connect(uri);
db = connection.db('apartments_test');
});
afterAll(async () => {
if (connection) {
await connection.close();
}
});
beforeEach(async () => {
await cleanupTestData(db);
// Also clean scraper_runs collection
await db.collection('scraper_runs').deleteMany({});
app = await createTestApp(db);
scraperJob.__reset();
});
// ============================================================
// GET /api/admin/scraper/status
// ============================================================
describe('GET /api/admin/scraper/status', () => {
describe('Authentication and Authorization', () => {
it('should return 401 without authentication', async () => {
const res = await request(app)
.get('/api/admin/scraper/status')
.expect(401);
expect(res.body).toHaveProperty('error');
});
it('should return 403 for non-admin user', async () => {
const regularUser = createTestUser({ role: 'user' });
await insertTestUser(db, regularUser);
const token = generateTestToken(regularUser._id);
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(403);
expect(res.body).toHaveProperty('error');
expect(res.body.error).toBe('Admin access required');
});
});
describe('Successful status response (admin)', () => {
let admin;
let token;
beforeEach(async () => {
admin = createTestAdmin();
await insertTestUser(db, admin);
token = generateTestToken(admin._id);
});
it('should return 200 with status for admin', async () => {
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body).toHaveProperty('data');
expect(res.body.data).toHaveProperty('currentStatus');
expect(res.body.data).toHaveProperty('runningJobId');
expect(res.body.data).toHaveProperty('lastRun');
expect(res.body.data).toHaveProperty('nextScheduledRun');
expect(res.body.data).toHaveProperty('schedule');
});
it('should return currentStatus "idle" when scraper is not running', async () => {
scraperJob.__setRunning(false);
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.currentStatus).toBe('idle');
expect(res.body.data.runningJobId).toBeNull();
});
it('should return currentStatus "running" with runningJobId when scraper is running', async () => {
const jobId = 'test-job-id-abc-123';
scraperJob.__setRunning(true, jobId);
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.currentStatus).toBe('running');
expect(res.body.data.runningJobId).toBe(jobId);
});
it('should return lastRun as null when no runs exist', async () => {
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.lastRun).toBeNull();
});
it('should return lastRun object with correct fields when a run exists', async () => {
// Insert a scraper run record
await db.collection('scraper_runs').insertOne({
jobId: 'previous-job-001',
trigger: 'scheduled',
status: 'success',
startedAt: '2026-02-05T06:00:00.000Z',
completedAt: '2026-02-05T06:00:12.345Z',
duration: 12345,
unitsProcessed: 50,
pricesInserted: 48,
newUnitsCount: 2,
rentedUnitsCount: 1,
staleUnitsCount: 0,
errors: [],
recordedAt: new Date('2026-02-05T06:00:12.345Z')
});
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
const lastRun = res.body.data.lastRun;
expect(lastRun).not.toBeNull();
expect(lastRun.jobId).toBe('previous-job-001');
expect(lastRun.timestamp).toBe('2026-02-05T06:00:00.000Z');
expect(lastRun.status).toBe('success');
expect(lastRun.duration).toBe(12345);
expect(lastRun.trigger).toBe('scheduled');
expect(lastRun.unitsProcessed).toBe(50);
expect(lastRun.pricesInserted).toBe(48);
expect(lastRun.errors).toBeNull();
});
it('should return lastRun with errors array when last run had errors', async () => {
await db.collection('scraper_runs').insertOne({
jobId: 'failed-job-002',
trigger: 'manual',
status: 'failed',
startedAt: '2026-02-05T10:00:00.000Z',
completedAt: '2026-02-05T10:00:32.000Z',
duration: 32000,
unitsProcessed: 0,
pricesInserted: 0,
newUnitsCount: 0,
rentedUnitsCount: 0,
staleUnitsCount: 0,
errors: ['HTTP request failed after 3 retries'],
recordedAt: new Date('2026-02-05T10:00:32.000Z')
});
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
const lastRun = res.body.data.lastRun;
expect(lastRun.status).toBe('failed');
expect(lastRun.errors).toEqual(['HTTP request failed after 3 retries']);
});
it('should return the most recent run when multiple runs exist', async () => {
// Insert older run
await db.collection('scraper_runs').insertOne({
jobId: 'older-job-001',
trigger: 'scheduled',
status: 'success',
startedAt: '2026-02-04T06:00:00.000Z',
completedAt: '2026-02-04T06:00:10.000Z',
duration: 10000,
unitsProcessed: 45,
pricesInserted: 45,
errors: [],
recordedAt: new Date('2026-02-04T06:00:10.000Z')
});
// Insert newer run
await db.collection('scraper_runs').insertOne({
jobId: 'newer-job-002',
trigger: 'manual',
status: 'success',
startedAt: '2026-02-05T14:00:00.000Z',
completedAt: '2026-02-05T14:00:08.000Z',
duration: 8000,
unitsProcessed: 52,
pricesInserted: 50,
errors: [],
recordedAt: new Date('2026-02-05T14:00:08.000Z')
});
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.lastRun.jobId).toBe('newer-job-002');
expect(res.body.data.lastRun.trigger).toBe('manual');
});
it('should return nextScheduledRun as ISO timestamp', async () => {
scraperJob.__setSchedule('0 6 * * *', '2026-02-07T06:00:00.000Z');
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.nextScheduledRun).toBe('2026-02-07T06:00:00.000Z');
});
it('should return nextScheduledRun as null when scheduler is disabled', async () => {
scraperJob.__setSchedule('disabled', null);
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.nextScheduledRun).toBeNull();
expect(res.body.data.schedule).toBe('disabled');
});
it('should return schedule cron expression', async () => {
scraperJob.__setSchedule('0 6 * * *', '2026-02-07T06:00:00.000Z');
const res = await request(app)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(200);
expect(res.body.data.schedule).toBe('0 6 * * *');
});
});
describe('Error handling', () => {
let admin;
let token;
beforeEach(async () => {
admin = createTestAdmin();
await insertTestUser(db, admin);
token = generateTestToken(admin._id);
});
it('should return 503 on database error', async () => {
// Create a proxy db that works for auth (users collection)
// but throws errors for scraper_runs collection
const express = require('express');
const cookieParser = require('cookie-parser');
const brokenApp = express();
brokenApp.use(express.json());
brokenApp.use(cookieParser());
// Proxy db: real db for users, broken for scraper_runs
const proxyDb = {
collection: jest.fn((name) => {
if (name === 'scraper_runs') {
return {
findOne: jest.fn().mockRejectedValue(new Error('Database connection lost'))
};
}
// Delegate to real db for all other collections (users, etc.)
return db.collection(name);
})
};
brokenApp.locals.db = proxyDb;
const adminRoutes = require('../../routes/admin');
brokenApp.use('/api/admin', adminRoutes);
const res = await request(brokenApp)
.get('/api/admin/scraper/status')
.set('Cookie', [`auth_token=${token}`])
.expect(503);
expect(res.body).toHaveProperty('error');
expect(res.body.error).toBe('Service temporarily unavailable');
});
});
});
});

View File

@ -1142,5 +1142,57 @@ router.patch('/settings', async (req, res) => {
} }
}); });
// ============================================================
// Scraper Endpoints
// ============================================================
const {
isScraperRunning,
getCurrentJobId,
getScheduleExpression,
getNextScheduledRun
} = require('../jobs/scraperJob');
const scraperConfig = require('../config/scraper');
/**
* GET /api/admin/scraper/status
* Get current scraper status including running state, last run details,
* next scheduled run, and schedule expression.
*/
router.get('/scraper/status', async (req, res) => {
try {
const db = req.app.locals.db;
// Get last run from history
const lastRunDoc = await db.collection(scraperConfig.COLLECTIONS.SCRAPER_RUNS)
.findOne({}, { sort: { startedAt: -1 } });
const lastRun = lastRunDoc ? {
jobId: lastRunDoc.jobId,
timestamp: lastRunDoc.startedAt,
status: lastRunDoc.status,
duration: lastRunDoc.duration,
trigger: lastRunDoc.trigger,
unitsProcessed: lastRunDoc.unitsProcessed,
pricesInserted: lastRunDoc.pricesInserted,
errors: lastRunDoc.errors?.length > 0 ? lastRunDoc.errors : null
} : null;
res.json({
data: {
currentStatus: isScraperRunning() ? 'running' : 'idle',
runningJobId: isScraperRunning() ? getCurrentJobId() : null,
lastRun,
nextScheduledRun: getNextScheduledRun(),
schedule: getScheduleExpression()
}
});
} catch (error) {
console.error('Error fetching scraper status:', error);
res.status(503).json({ error: 'Service temporarily unavailable' });
}
});
module.exports = router; module.exports = router;
module.exports.clearStatsCache = clearStatsCache; module.exports.clearStatsCache = clearStatsCache;