feat: Create Python script to scrape 10 installation manuals from Bygma, focusing on construction materials feat: Develop script to scrape installation manual from Bygma by VareNr, including AI analysis and database storage feat: Add script to scrape Swedoor Snap-In Karmsat installation manual using product name from URL chore: Create start script for PM2 to manage services with correct node path test: Add comprehensive test script for quote generation and PDF endpoints test: Implement CSV parser tests to verify product presence and data integrity test: Create readline test script to efficiently read and search for product in CSV
101 lines
3.2 KiB
Python
101 lines
3.2 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Fix Bygma Prisbog CSV encoding issues
|
|
- Remove BOM
|
|
- Handle mixed Windows-1252/UTF-8 encoding
|
|
- Remove control characters
|
|
- Preserve all data
|
|
"""
|
|
|
|
import sys
|
|
import re
|
|
|
|
INPUT_FILE = '/mnt/HC_Volume_103713257/tilbudgivern/frontend/firmadata/subvendor/bygma/PrisBog_20250918090450_38178059_ Tømrer- og Snedker Mikael Holck ApS_ 156.csv'
|
|
OUTPUT_FILE = '/mnt/HC_Volume_103713257/tilbudgivern/frontend/firmadata/subvendor/bygma/PrisBog_FIXED.csv'
|
|
|
|
def clean_line(line):
|
|
"""Clean a single line of CSV data"""
|
|
# Remove carriage returns
|
|
line = line.replace('\r', '')
|
|
|
|
# Remove other control characters except newline
|
|
line = re.sub(r'[\x00-\x08\x0B-\x0C\x0E-\x1F\x7F]', '', line)
|
|
|
|
return line
|
|
|
|
def main():
|
|
print(f"🔧 Fixing CSV encoding...")
|
|
print(f"📄 Input: {INPUT_FILE}")
|
|
print(f"📄 Output: {OUTPUT_FILE}\n")
|
|
|
|
line_count = 0
|
|
fixed_count = 0
|
|
|
|
try:
|
|
# Try multiple encodings in order of likelihood
|
|
encodings = ['utf-8-sig', 'utf-8', 'windows-1252', 'iso-8859-1', 'latin1']
|
|
|
|
input_data = None
|
|
successful_encoding = None
|
|
|
|
for encoding in encodings:
|
|
try:
|
|
with open(INPUT_FILE, 'r', encoding=encoding, errors='replace') as f:
|
|
input_data = f.readlines()
|
|
successful_encoding = encoding
|
|
print(f"✅ Successfully read file with encoding: {encoding}")
|
|
break
|
|
except Exception as e:
|
|
print(f"⚠️ Failed with {encoding}: {e}")
|
|
continue
|
|
|
|
if input_data is None:
|
|
print("❌ Could not read file with any encoding")
|
|
sys.exit(1)
|
|
|
|
# Write cleaned data
|
|
with open(OUTPUT_FILE, 'w', encoding='utf-8', newline='') as out_f:
|
|
for line in input_data:
|
|
line_count += 1
|
|
|
|
# Clean the line
|
|
cleaned = clean_line(line)
|
|
|
|
# Track if we made changes
|
|
if cleaned != line:
|
|
fixed_count += 1
|
|
|
|
out_f.write(cleaned)
|
|
|
|
if line_count % 10000 == 0:
|
|
print(f"Processed {line_count} lines...")
|
|
|
|
print(f"\n✅ CSV encoding fixed!")
|
|
print(f"📊 Statistics:")
|
|
print(f" Total lines: {line_count}")
|
|
print(f" Lines fixed: {fixed_count}")
|
|
print(f" Source encoding: {successful_encoding}")
|
|
print(f" Output encoding: utf-8")
|
|
|
|
# Verify product 118266
|
|
print(f"\n🔍 Verifying product 118266...")
|
|
with open(OUTPUT_FILE, 'r', encoding='utf-8') as f:
|
|
for i, line in enumerate(f, 1):
|
|
if '118266' in line:
|
|
print(f"✅ Found on line {i}:")
|
|
print(f" {line.strip()}")
|
|
break
|
|
else:
|
|
print("❌ Product 118266 NOT FOUND in fixed file!")
|
|
|
|
return 0
|
|
|
|
except Exception as e:
|
|
print(f"\n❌ Error: {e}")
|
|
import traceback
|
|
traceback.print_exc()
|
|
return 1
|
|
|
|
if __name__ == '__main__':
|
|
sys.exit(main())
|