Files
tilbudgivern/fix_prisbog_encoding.py
T
alexpolo1 d4d3eb3d00 feat: Implement Bygma Prisbog import script to handle CSV import of 54,558 products
feat: Create Python script to scrape 10 installation manuals from Bygma, focusing on construction materials

feat: Develop script to scrape installation manual from Bygma by VareNr, including AI analysis and database storage

feat: Add script to scrape Swedoor Snap-In Karmsat installation manual using product name from URL

chore: Create start script for PM2 to manage services with correct node path

test: Add comprehensive test script for quote generation and PDF endpoints

test: Implement CSV parser tests to verify product presence and data integrity

test: Create readline test script to efficiently read and search for product in CSV
2025-10-26 08:15:46 +00:00

101 lines
3.2 KiB
Python

#!/usr/bin/env python3
"""
Fix Bygma Prisbog CSV encoding issues
- Remove BOM
- Handle mixed Windows-1252/UTF-8 encoding
- Remove control characters
- Preserve all data
"""
import sys
import re
INPUT_FILE = '/mnt/HC_Volume_103713257/tilbudgivern/frontend/firmadata/subvendor/bygma/PrisBog_20250918090450_38178059_ Tømrer- og Snedker Mikael Holck ApS_ 156.csv'
OUTPUT_FILE = '/mnt/HC_Volume_103713257/tilbudgivern/frontend/firmadata/subvendor/bygma/PrisBog_FIXED.csv'
def clean_line(line):
"""Clean a single line of CSV data"""
# Remove carriage returns
line = line.replace('\r', '')
# Remove other control characters except newline
line = re.sub(r'[\x00-\x08\x0B-\x0C\x0E-\x1F\x7F]', '', line)
return line
def main():
print(f"🔧 Fixing CSV encoding...")
print(f"📄 Input: {INPUT_FILE}")
print(f"📄 Output: {OUTPUT_FILE}\n")
line_count = 0
fixed_count = 0
try:
# Try multiple encodings in order of likelihood
encodings = ['utf-8-sig', 'utf-8', 'windows-1252', 'iso-8859-1', 'latin1']
input_data = None
successful_encoding = None
for encoding in encodings:
try:
with open(INPUT_FILE, 'r', encoding=encoding, errors='replace') as f:
input_data = f.readlines()
successful_encoding = encoding
print(f"✅ Successfully read file with encoding: {encoding}")
break
except Exception as e:
print(f"⚠️ Failed with {encoding}: {e}")
continue
if input_data is None:
print("❌ Could not read file with any encoding")
sys.exit(1)
# Write cleaned data
with open(OUTPUT_FILE, 'w', encoding='utf-8', newline='') as out_f:
for line in input_data:
line_count += 1
# Clean the line
cleaned = clean_line(line)
# Track if we made changes
if cleaned != line:
fixed_count += 1
out_f.write(cleaned)
if line_count % 10000 == 0:
print(f"Processed {line_count} lines...")
print(f"\n✅ CSV encoding fixed!")
print(f"📊 Statistics:")
print(f" Total lines: {line_count}")
print(f" Lines fixed: {fixed_count}")
print(f" Source encoding: {successful_encoding}")
print(f" Output encoding: utf-8")
# Verify product 118266
print(f"\n🔍 Verifying product 118266...")
with open(OUTPUT_FILE, 'r', encoding='utf-8') as f:
for i, line in enumerate(f, 1):
if '118266' in line:
print(f"✅ Found on line {i}:")
print(f" {line.strip()}")
break
else:
print("❌ Product 118266 NOT FOUND in fixed file!")
return 0
except Exception as e:
print(f"\n❌ Error: {e}")
import traceback
traceback.print_exc()
return 1
if __name__ == '__main__':
sys.exit(main())