Рефакторинг проекта: разделение на модули, улучшенный интерфейс, обработка Excel

This commit is contained in:
2025-08-04 00:31:05 +05:00
parent 059f174dd5
commit a882a81180
22 changed files with 1936 additions and 530 deletions

171
services/excel_processor.py Normal file
View File

@@ -0,0 +1,171 @@
"""
Сервис для работы с Excel файлами.
"""
import pandas as pd
from pathlib import Path
from typing import List, Set, Dict, Optional
class ExcelProcessor:
def __init__(self, excel_file_path: str):
self.excel_file_path = Path(excel_file_path)
self.cadastral_numbers: Set[str] = set()
self.df: Optional[pd.DataFrame] = None
self.cadastral_column: Optional[str] = None
def load_excel_file(self) -> bool:
"""Загружает Excel файл и находит колонку с кадастровыми номерами"""
try:
if not self.excel_file_path.exists():
print(f"Excel файл {self.excel_file_path} не найден")
return False
# Пробуем разные форматы Excel файлов
try:
self.df = pd.read_excel(self.excel_file_path, engine='openpyxl')
except:
self.df = pd.read_excel(self.excel_file_path, engine='xlrd')
# Ищем колонку с кадастровыми номерами
self.cadastral_column = self._find_cadastral_column()
if self.cadastral_column:
# Извлекаем кадастровые номера
self._extract_cadastral_numbers()
print(f"Загружено {len(self.cadastral_numbers)} кадастровых номеров из Excel")
return True
else:
print("Не найдена колонка с кадастровыми номерами в Excel файле")
return False
except Exception as e:
print(f"Ошибка при загрузке Excel файла: {e}")
return False
def _find_cadastral_column(self) -> Optional[str]:
"""Находит колонку с кадастровыми номерами"""
if self.df is None:
return None
# Ключевые слова для поиска колонки
keywords = ['кадастр', 'номер', 'cadastral', 'number', 'кн', 'cad']
for column in self.df.columns:
column_str = str(column).lower()
if any(keyword in column_str for keyword in keywords):
return column
# Если не найдено по названию, ищем по содержимому
for column in self.df.columns:
# Проверяем первые несколько значений в колонке
sample_values = self.df[column].dropna().head(5)
cadastral_count = 0
for value in sample_values:
if self._is_cadastral_number_format(str(value)):
cadastral_count += 1
# Если больше половины значений похожи на кадастровые номера
if cadastral_count >= len(sample_values) * 0.5:
return column
return None
def _extract_cadastral_numbers(self):
"""Извлекает кадастровые номера из найденной колонки"""
if self.df is None or self.cadastral_column is None:
return
cadastral_numbers = set()
for value in self.df[self.cadastral_column].dropna():
value_str = str(value).strip()
if self._is_cadastral_number_format(value_str):
cadastral_numbers.add(value_str)
self.cadastral_numbers = cadastral_numbers
def _is_cadastral_number_format(self, text: str) -> bool:
"""Проверяет, соответствует ли текст формату кадастрового номера"""
import re
pattern = r'^\d{1,2}:\d{1,2}:\d{6,8}:\d{1,5}$'
return bool(re.match(pattern, text))
def update_excel_with_file_info(self, cadastral_to_file_map: Dict[str, str],
output_file_path: str = None) -> bool:
"""Обновляет Excel файл информацией о найденных файлах и смежных номерах"""
try:
if self.df is None or self.cadastral_column is None:
print("Excel файл не загружен или не найдена колонка с номерами")
return False
# Создаем карту файл -> множество кадастровых номеров для поиска смежных
file_to_numbers_map = {}
for number, file_path in cadastral_to_file_map.items():
if file_path not in file_to_numbers_map:
file_to_numbers_map[file_path] = set()
file_to_numbers_map[file_path].add(number)
# Добавляем новые колонки
self.df['Найденные файлы'] = ''
self.df['Смежные кадастровые номера'] = ''
self.df['Кадастровый блок'] = ''
# Заполняем информацию для каждой строки
for index, row in self.df.iterrows():
cadastral_number = str(row[self.cadastral_column]).strip()
if cadastral_number in cadastral_to_file_map:
# 1. Получаем файл(ы) где найден номер
main_file_path = cadastral_to_file_map[cadastral_number]
file_name = Path(main_file_path).name
self.df.at[index, 'Найденные файлы'] = file_name
# 2. Находим все смежные кадастровые номера из того же файла
if main_file_path in file_to_numbers_map:
all_numbers_in_file = file_to_numbers_map[main_file_path]
# Исключаем текущий номер из списка смежных
adjacent_numbers = all_numbers_in_file - {cadastral_number}
if adjacent_numbers:
self.df.at[index, 'Смежные кадастровые номера'] = ', '.join(sorted(adjacent_numbers))
# 3. Определяем кадастровый блок (первые три части номера)
cadastral_block = self._extract_cadastral_block(cadastral_number)
if cadastral_block:
self.df.at[index, 'Кадастровый блок'] = cadastral_block
# Сохраняем обновленный файл
output_path = output_file_path or str(self.excel_file_path.with_suffix('.updated.xlsx'))
self.df.to_excel(output_path, index=False, engine='openpyxl')
print(f"Обновленный Excel файл сохранен: {output_path}")
return True
except Exception as e:
print(f"Ошибка при обновлении Excel файла: {e}")
return False
def _extract_cadastral_block(self, cadastral_number: str) -> str:
"""Извлекает кадастровый блок из номера (первые три части)"""
try:
parts = cadastral_number.split(':')
if len(parts) >= 3:
return ':'.join(parts[:3])
return cadastral_number
except:
return cadastral_number
def get_cadastral_numbers(self) -> Set[str]:
"""Возвращает множество кадастровых номеров из Excel"""
return self.cadastral_numbers.copy()
def get_statistics(self) -> Dict[str, any]:
"""Возвращает статистику по Excel файлу"""
if self.df is None:
return {}
return {
'total_rows': len(self.df),
'cadastral_column': self.cadastral_column,
'cadastral_numbers_count': len(self.cadastral_numbers),
'columns': list(self.df.columns)
}