feat: add POST /api/admin/scraper/run endpoint
Some checks failed
CI/CD Pipeline - Apartment API / Scan Dependencies (pull_request) Successful in 13s
CI/CD Pipeline - Apartment API / Lint & Test (pull_request) Successful in 41s
CI/CD Pipeline - Apartment API / Send Webhook Notification (pull_request) Failing after 1s
CI/CD Pipeline - Apartment API / Build & Push Image (pull_request) Has been skipped
CI/CD Pipeline - Apartment API / Deploy to Production (pull_request) Has been skipped

Implement the manual scraper trigger endpoint with the following:

- Protected by requireAuth and requireAdmin middleware (401/403)
- Mutex lock check via isScraperRunning to prevent concurrent runs (409)
- Generates UUID jobId for tracking async scrape execution
- Returns 202 Accepted immediately without blocking on scrape completion
- Releases mutex lock in .finally() to ensure cleanup on success or failure
- Logs ADMIN_TRIGGER_SCRAPE activity with jobId, dryRun, and
  usingProvidedHtml metadata via activityLogger
- Supports dryRun option (defaults to false) and htmlContent for
  testing with pre-fetched HTML
- Passes trigger: 'manual', jobId, dryRun, and htmlContent to runScrape

Includes 20 tests covering auth, mutex, async execution, activity
logging, dryRun/htmlContent options, and response structure.
This commit is contained in:
2026-02-06 21:20:42 -07:00
parent c6d480a870
commit c30b02681e
2 changed files with 592 additions and 0 deletions

View File

@ -1142,5 +1142,69 @@ router.patch('/settings', async (req, res) => {
}
});
// ============================================================
// Scraper Endpoints
// ============================================================
const crypto = require('crypto');
const {
isScraperRunning,
acquireLock,
releaseLock
} = require('../jobs/scraperJob');
const { runScrape } = require('../services/scraperService');
/**
* POST /api/admin/scraper/run
* Trigger a manual scrape
*
* Request body (optional):
* - dryRun: boolean - Skip database writes for safe testing
* - htmlContent: string - Use provided HTML instead of fetching (for testing/debugging)
*/
router.post('/scraper/run', async (req, res) => {
try {
const db = req.app.locals.db;
const { dryRun = false, htmlContent = null } = req.body || {};
// Check if scraper is already running
if (isScraperRunning()) {
return res.status(409).json({ error: 'Scrape already in progress' });
}
// Generate job ID and acquire lock
const jobId = crypto.randomUUID();
acquireLock(jobId);
// Log admin action
await logActivity(db, {
userId: req.user._id.toString(),
action: 'ADMIN_TRIGGER_SCRAPE',
metadata: { jobId, dryRun, usingProvidedHtml: !!htmlContent }
});
// Start scrape asynchronously (do not await - return 202 immediately)
runScrape(db, { trigger: 'manual', jobId, dryRun, htmlContent })
.catch((error) => {
console.error('Async scrape failed:', error.message);
})
.finally(() => releaseLock());
// Return immediately with job ID
res.status(202).json({
data: {
jobId,
status: 'started',
dryRun,
message: dryRun ? 'Scrape job initiated (dry run - no DB writes)' : 'Scrape job initiated'
}
});
} catch (error) {
console.error('Error triggering scrape:', error);
res.status(500).json({ error: 'Failed to start scrape job' });
}
});
module.exports = router;
module.exports.clearStatsCache = clearStatsCache;