SCRAPE-16: Implement POST /admin/scraper/run endpoint (#21)

Co-authored-by: Stephen Minakian <stephenminakian@gmail.com>
Co-committed-by: Stephen Minakian <stephenminakian@gmail.com>
This commit is contained in:
2026-02-06 22:07:10 -07:00
committed by stephen
parent c6d480a870
commit daa234428e
4 changed files with 689 additions and 69 deletions

View File

@ -1142,5 +1142,69 @@ router.patch('/settings', async (req, res) => {
}
});
// ============================================================
// Scraper Endpoints
// ============================================================
const crypto = require('crypto');
const {
isScraperRunning,
acquireLock,
releaseLock
} = require('../jobs/scraperJob');
const { runScrape } = require('../services/scraperService');
/**
* POST /api/admin/scraper/run
* Trigger a manual scrape
*
* Request body (optional):
* - dryRun: boolean - Skip database writes for safe testing
* - htmlContent: string - Use provided HTML instead of fetching (for testing/debugging)
*/
router.post('/scraper/run', async (req, res) => {
try {
const db = req.app.locals.db;
const { dryRun = false, htmlContent = null } = req.body || {};
// Check if scraper is already running
if (isScraperRunning()) {
return res.status(409).json({ error: 'Scrape already in progress' });
}
// Generate job ID and acquire lock
const jobId = crypto.randomUUID();
acquireLock(jobId);
// Log admin action
await logActivity(db, {
userId: req.user._id.toString(),
action: 'ADMIN_TRIGGER_SCRAPE',
metadata: { jobId, dryRun, usingProvidedHtml: !!htmlContent }
});
// Start scrape asynchronously (do not await - return 202 immediately)
runScrape(db, { trigger: 'manual', jobId, dryRun, htmlContent })
.catch((error) => {
console.error('Async scrape failed:', error.message);
})
.finally(() => releaseLock());
// Return immediately with job ID
res.status(202).json({
data: {
jobId,
status: 'started',
dryRun,
message: dryRun ? 'Scrape job initiated (dry run - no DB writes)' : 'Scrape job initiated'
}
});
} catch (error) {
console.error('Error triggering scrape:', error);
res.status(500).json({ error: 'Failed to start scrape job' });
}
});
module.exports = router;
module.exports.clearStatsCache = clearStatsCache;