This commit is contained in:
2026-03-10 20:21:31 +05:00
commit ec82b528a3
14 changed files with 1041 additions and 0 deletions

229
read_mif_file.py Normal file
View File

@@ -0,0 +1,229 @@
from dataclasses import dataclass, field
from pathlib import Path
from typing import List, Dict, Any, Optional, Tuple, Union
@dataclass
class MifFile:
name: str
path: str
headers: Dict[str, Any] = field(default_factory=dict)
features: List[Dict[str, Any]] = field(default_factory=list)
def _unquote(s: str) -> str:
s = s.strip()
if s.startswith('"') and s.endswith('"'):
return s[1:-1]
return s
def parse_mif(file_path: str) -> Union[MifFile, List[MifFile]]:
"""
Примитивный парсер MIF-файла. Обрабатывает заголовки (Version, Charset, Delimiter,
CoordSys, Columns) и простые объекты в блоке Data: Region, Text, Center/Centre, Pen, Brush.
Не претендует на полную спецификацию MapInfo MIF, но удобен для быстрого извлечения структур.
"""
p = Path(file_path)
if not p.exists():
raise FileNotFoundError(file_path)
# If a directory is passed, parse all .mif files inside and return a list
if p.is_dir():
return load_mif_files(str(p))
mif = MifFile(name=p.name, path=str(p.resolve()))
lines = [ln.rstrip('\n') for ln in p.read_text(encoding='utf-8', errors='replace').splitlines()]
i = 0
n = len(lines)
# parse header until "Data"
while i < n:
line = lines[i].strip()
if not line:
i += 1
continue
up = line.upper()
if up.startswith('VERSION'):
mif.headers['version'] = line.split(None, 1)[1].strip() if len(line.split()) > 1 else None
i += 1
continue
if up.startswith('CHARSET'):
# Charset "WindowsCyrillic"
rest = line.split(None, 1)[1] if len(line.split()) > 1 else ''
mif.headers['charset'] = _unquote(rest)
i += 1
continue
if up.startswith('DELIMITER'):
rest = line.split(None, 1)[1] if len(line.split()) > 1 else ''
mif.headers['delimiter'] = _unquote(rest)
i += 1
continue
if up.startswith('COORDSYS'):
mif.headers['coordsys'] = line[len('CoordSys'):].strip()
i += 1
continue
if up.startswith('COLUMNS'):
# next N lines are column definitions
parts = line.split()
cnt = int(parts[1]) if len(parts) > 1 else 0
i += 1
cols: List[str] = []
for _ in range(cnt):
if i >= n:
break
cols.append(lines[i].strip())
i += 1
mif.headers['columns'] = cols
continue
if up == 'DATA':
i += 1
break
i += 1
# parse data features
while i < n:
raw = lines[i].lstrip()
if not raw:
i += 1
continue
parts = raw.split(None, 1)
typ = parts[0]
if typ.upper() == 'REGION':
feat: Dict[str, Any] = {'type': 'Region', 'props': {}}
# Region <N_polygons> — количество контуров (полигонов) в объекте
try:
n_polygons = int(parts[1]) if len(parts) > 1 else 1
except (ValueError, IndexError):
n_polygons = 1
i += 1
all_points: List[Tuple[float, float]] = []
contours: List[List[Tuple[float, float]]] = []
for _poly in range(n_polygons):
# Первая строка каждого контура — количество точек
if i >= n:
break
cnt_line = lines[i].strip()
try:
pts_cnt = int(cnt_line)
i += 1
except ValueError:
pts_cnt = 0
poly_pts: List[Tuple[float, float]] = []
for _ in range(pts_cnt):
if i >= n:
break
xy = lines[i].strip().split()
if len(xy) >= 2:
try:
poly_pts.append((float(xy[0]), float(xy[1])))
except ValueError:
pass
i += 1
if poly_pts:
contours.append(poly_pts)
all_points.extend(poly_pts)
# Если один контур — points == список точек (как было раньше)
# Если несколько — points == точки всех контуров подряд, contours — отдельно
feat['props']['points'] = all_points
feat['props']['contours'] = contours # List[List[(x,y)]]
# Читаем атрибуты (Pen, Brush, Center) до начала следующего объекта
while i < n:
t = lines[i].lstrip()
if not t:
i += 1
continue
up2 = t.upper()
if up2.startswith('PEN'):
feat['props']['pen'] = t[len('Pen'):].strip()
i += 1
continue
if up2.startswith('BRUSH'):
feat['props']['brush'] = t[len('Brush'):].strip()
i += 1
continue
if up2.startswith('CENTER') or up2.startswith('CENTRE'):
rest = t.split(None, 1)[1] if len(t.split()) > 1 else ''
coords = rest.split()
try:
cx = float(coords[0]); cy = float(coords[1])
feat['props']['center'] = (cx, cy)
except Exception:
pass
i += 1
continue
if up2.split()[0] in ('REGION', 'TEXT', 'POINT', 'PLINE', 'LINE'):
break
i += 1
mif.features.append(feat)
continue
if typ.upper() == 'TEXT':
feat = {'type': 'Text', 'props': {}}
# next line: quoted string
i += 1
if i < n:
txt_line = lines[i].strip()
feat['props']['text'] = _unquote(txt_line)
i += 1
# next line: bounding coordinates (often 4 numbers)
if i < n:
coords = lines[i].strip().split()
try:
nums = [float(x) for x in coords]
feat['props']['bbox'] = nums
except Exception:
pass
i += 1
# optional Font line
if i < n and lines[i].lstrip().upper().startswith('FONT'):
feat['props']['font'] = lines[i].lstrip()[len('Font'):].strip()
i += 1
mif.features.append(feat)
continue
# simple POINT parser
if typ.upper() == 'POINT':
feat = {'type': 'Point', 'props': {}}
rest = parts[1] if len(parts) > 1 else ''
coords = rest.split()
try:
feat['props']['xy'] = (float(coords[0]), float(coords[1]))
except Exception:
feat['props']['xy'] = None
i += 1
mif.features.append(feat)
continue
# unknown / unsupported object: skip one line
i += 1
return mif
def load_mif_files(dir_path: str) -> List[MifFile]:
"""Найти все .mif файлы в директории (рекурсивно) и распарсить их в объекты MifFile."""
root = Path(dir_path)
if not root.exists():
raise FileNotFoundError(dir_path)
mif_files: List[MifFile] = []
for p in root.rglob('*.mif'):
try:
mif_files.append(parse_mif(str(p)))
except Exception:
# при ошибке парсинга — пропустить файл
continue
return mif_files
__all__ = ["MifFile", "parse_mif", "load_mif_files"]