Рефакторинг проекта: разделение на модули, улучшенный интерфейс, обработка Excel
This commit is contained in:
171
services/excel_processor.py
Normal file
171
services/excel_processor.py
Normal file
@@ -0,0 +1,171 @@
|
||||
"""
|
||||
Сервис для работы с Excel файлами.
|
||||
"""
|
||||
import pandas as pd
|
||||
from pathlib import Path
|
||||
from typing import List, Set, Dict, Optional
|
||||
|
||||
class ExcelProcessor:
|
||||
def __init__(self, excel_file_path: str):
|
||||
self.excel_file_path = Path(excel_file_path)
|
||||
self.cadastral_numbers: Set[str] = set()
|
||||
self.df: Optional[pd.DataFrame] = None
|
||||
self.cadastral_column: Optional[str] = None
|
||||
|
||||
def load_excel_file(self) -> bool:
|
||||
"""Загружает Excel файл и находит колонку с кадастровыми номерами"""
|
||||
try:
|
||||
if not self.excel_file_path.exists():
|
||||
print(f"Excel файл {self.excel_file_path} не найден")
|
||||
return False
|
||||
|
||||
# Пробуем разные форматы Excel файлов
|
||||
try:
|
||||
self.df = pd.read_excel(self.excel_file_path, engine='openpyxl')
|
||||
except:
|
||||
self.df = pd.read_excel(self.excel_file_path, engine='xlrd')
|
||||
|
||||
# Ищем колонку с кадастровыми номерами
|
||||
self.cadastral_column = self._find_cadastral_column()
|
||||
|
||||
if self.cadastral_column:
|
||||
# Извлекаем кадастровые номера
|
||||
self._extract_cadastral_numbers()
|
||||
print(f"Загружено {len(self.cadastral_numbers)} кадастровых номеров из Excel")
|
||||
return True
|
||||
else:
|
||||
print("Не найдена колонка с кадастровыми номерами в Excel файле")
|
||||
return False
|
||||
|
||||
except Exception as e:
|
||||
print(f"Ошибка при загрузке Excel файла: {e}")
|
||||
return False
|
||||
|
||||
def _find_cadastral_column(self) -> Optional[str]:
|
||||
"""Находит колонку с кадастровыми номерами"""
|
||||
if self.df is None:
|
||||
return None
|
||||
|
||||
# Ключевые слова для поиска колонки
|
||||
keywords = ['кадастр', 'номер', 'cadastral', 'number', 'кн', 'cad']
|
||||
|
||||
for column in self.df.columns:
|
||||
column_str = str(column).lower()
|
||||
if any(keyword in column_str for keyword in keywords):
|
||||
return column
|
||||
|
||||
# Если не найдено по названию, ищем по содержимому
|
||||
for column in self.df.columns:
|
||||
# Проверяем первые несколько значений в колонке
|
||||
sample_values = self.df[column].dropna().head(5)
|
||||
cadastral_count = 0
|
||||
|
||||
for value in sample_values:
|
||||
if self._is_cadastral_number_format(str(value)):
|
||||
cadastral_count += 1
|
||||
|
||||
# Если больше половины значений похожи на кадастровые номера
|
||||
if cadastral_count >= len(sample_values) * 0.5:
|
||||
return column
|
||||
|
||||
return None
|
||||
|
||||
def _extract_cadastral_numbers(self):
|
||||
"""Извлекает кадастровые номера из найденной колонки"""
|
||||
if self.df is None or self.cadastral_column is None:
|
||||
return
|
||||
|
||||
cadastral_numbers = set()
|
||||
|
||||
for value in self.df[self.cadastral_column].dropna():
|
||||
value_str = str(value).strip()
|
||||
if self._is_cadastral_number_format(value_str):
|
||||
cadastral_numbers.add(value_str)
|
||||
|
||||
self.cadastral_numbers = cadastral_numbers
|
||||
|
||||
def _is_cadastral_number_format(self, text: str) -> bool:
|
||||
"""Проверяет, соответствует ли текст формату кадастрового номера"""
|
||||
import re
|
||||
pattern = r'^\d{1,2}:\d{1,2}:\d{6,8}:\d{1,5}$'
|
||||
return bool(re.match(pattern, text))
|
||||
|
||||
def update_excel_with_file_info(self, cadastral_to_file_map: Dict[str, str],
|
||||
output_file_path: str = None) -> bool:
|
||||
"""Обновляет Excel файл информацией о найденных файлах и смежных номерах"""
|
||||
try:
|
||||
if self.df is None or self.cadastral_column is None:
|
||||
print("Excel файл не загружен или не найдена колонка с номерами")
|
||||
return False
|
||||
|
||||
# Создаем карту файл -> множество кадастровых номеров для поиска смежных
|
||||
file_to_numbers_map = {}
|
||||
for number, file_path in cadastral_to_file_map.items():
|
||||
if file_path not in file_to_numbers_map:
|
||||
file_to_numbers_map[file_path] = set()
|
||||
file_to_numbers_map[file_path].add(number)
|
||||
|
||||
# Добавляем новые колонки
|
||||
self.df['Найденные файлы'] = ''
|
||||
self.df['Смежные кадастровые номера'] = ''
|
||||
self.df['Кадастровый блок'] = ''
|
||||
|
||||
# Заполняем информацию для каждой строки
|
||||
for index, row in self.df.iterrows():
|
||||
cadastral_number = str(row[self.cadastral_column]).strip()
|
||||
|
||||
if cadastral_number in cadastral_to_file_map:
|
||||
# 1. Получаем файл(ы) где найден номер
|
||||
main_file_path = cadastral_to_file_map[cadastral_number]
|
||||
file_name = Path(main_file_path).name
|
||||
self.df.at[index, 'Найденные файлы'] = file_name
|
||||
|
||||
# 2. Находим все смежные кадастровые номера из того же файла
|
||||
if main_file_path in file_to_numbers_map:
|
||||
all_numbers_in_file = file_to_numbers_map[main_file_path]
|
||||
# Исключаем текущий номер из списка смежных
|
||||
adjacent_numbers = all_numbers_in_file - {cadastral_number}
|
||||
if adjacent_numbers:
|
||||
self.df.at[index, 'Смежные кадастровые номера'] = ', '.join(sorted(adjacent_numbers))
|
||||
|
||||
# 3. Определяем кадастровый блок (первые три части номера)
|
||||
cadastral_block = self._extract_cadastral_block(cadastral_number)
|
||||
if cadastral_block:
|
||||
self.df.at[index, 'Кадастровый блок'] = cadastral_block
|
||||
|
||||
# Сохраняем обновленный файл
|
||||
output_path = output_file_path or str(self.excel_file_path.with_suffix('.updated.xlsx'))
|
||||
self.df.to_excel(output_path, index=False, engine='openpyxl')
|
||||
|
||||
print(f"Обновленный Excel файл сохранен: {output_path}")
|
||||
return True
|
||||
|
||||
except Exception as e:
|
||||
print(f"Ошибка при обновлении Excel файла: {e}")
|
||||
return False
|
||||
|
||||
def _extract_cadastral_block(self, cadastral_number: str) -> str:
|
||||
"""Извлекает кадастровый блок из номера (первые три части)"""
|
||||
try:
|
||||
parts = cadastral_number.split(':')
|
||||
if len(parts) >= 3:
|
||||
return ':'.join(parts[:3])
|
||||
return cadastral_number
|
||||
except:
|
||||
return cadastral_number
|
||||
|
||||
def get_cadastral_numbers(self) -> Set[str]:
|
||||
"""Возвращает множество кадастровых номеров из Excel"""
|
||||
return self.cadastral_numbers.copy()
|
||||
|
||||
def get_statistics(self) -> Dict[str, any]:
|
||||
"""Возвращает статистику по Excel файлу"""
|
||||
if self.df is None:
|
||||
return {}
|
||||
|
||||
return {
|
||||
'total_rows': len(self.df),
|
||||
'cadastral_column': self.cadastral_column,
|
||||
'cadastral_numbers_count': len(self.cadastral_numbers),
|
||||
'columns': list(self.df.columns)
|
||||
}
|
||||
Reference in New Issue
Block a user