100 lines
3.1 KiB
JavaScript
100 lines
3.1 KiB
JavaScript
const { spawn } = require('child_process');
|
|
const path = require('path');
|
|
const logger = require('../utils/logger');
|
|
|
|
class BygmaScraperService {
|
|
/**
|
|
* Trigger Bygma PDF scraper to download and analyze installation manual
|
|
* @param {string} productUrl - Bygma product URL
|
|
* @returns {Promise<Object>} - Analysis result from Python script
|
|
*/
|
|
async scrapeAndAnalyzeProduct(productUrl) {
|
|
return new Promise((resolve, reject) => {
|
|
const scriptPath = path.join(__dirname, '../../../scrape_bygma_api.py');
|
|
|
|
logger.info(`Triggering Bygma scraper for: ${productUrl}`);
|
|
|
|
// Spawn Python process
|
|
const pythonProcess = spawn('python3', [scriptPath, productUrl], {
|
|
cwd: path.join(__dirname, '../../..'),
|
|
env: { ...process.env }
|
|
});
|
|
|
|
let outputData = '';
|
|
let errorData = '';
|
|
|
|
// Collect stdout
|
|
pythonProcess.stdout.on('data', (data) => {
|
|
const output = data.toString();
|
|
outputData += output;
|
|
logger.info(`Bygma scraper: ${output.trim()}`);
|
|
});
|
|
|
|
// Collect stderr
|
|
pythonProcess.stderr.on('data', (data) => {
|
|
const error = data.toString();
|
|
errorData += error;
|
|
logger.error(`Bygma scraper error: ${error.trim()}`);
|
|
});
|
|
|
|
// Handle process completion
|
|
pythonProcess.on('close', (code) => {
|
|
if (code === 0) {
|
|
logger.info('Bygma scraper completed successfully');
|
|
|
|
// Try to extract JSON result from output
|
|
try {
|
|
const jsonMatch = outputData.match(/\{[\s\S]*"success":\s*true[\s\S]*\}/);
|
|
if (jsonMatch) {
|
|
const result = JSON.parse(jsonMatch[0]);
|
|
resolve(result);
|
|
} else {
|
|
resolve({
|
|
success: true,
|
|
message: 'Scraping completed but no structured result found',
|
|
output: outputData
|
|
});
|
|
}
|
|
} catch (parseError) {
|
|
resolve({
|
|
success: true,
|
|
message: 'Scraping completed',
|
|
output: outputData
|
|
});
|
|
}
|
|
} else {
|
|
reject({
|
|
success: false,
|
|
error: `Scraper failed with code ${code}`,
|
|
stderr: errorData,
|
|
stdout: outputData
|
|
});
|
|
}
|
|
});
|
|
|
|
// Handle process errors
|
|
pythonProcess.on('error', (error) => {
|
|
logger.error('Failed to start Bygma scraper:', error);
|
|
reject({
|
|
success: false,
|
|
error: 'Failed to start scraper process',
|
|
details: error.message
|
|
});
|
|
});
|
|
});
|
|
}
|
|
|
|
/**
|
|
* Scrape installation manual by product name search
|
|
* @param {string} productName - Product name to search for on Bygma
|
|
* @returns {Promise<Object>} - Analysis result
|
|
*/
|
|
async scrapeByProductName(productName) {
|
|
// For now, we'll need to implement a search function
|
|
// This would search Bygma for the product and get the URL
|
|
throw new Error('Search by product name not yet implemented. Please provide product URL.');
|
|
}
|
|
}
|
|
|
|
module.exports = new BygmaScraperService();
|