feat(etl): извлечение прайсов по форматам (xlsx/xls/docx/pdf) и распознавание сложных страниц через Gemini Vision

This commit is contained in:
2026-06-26 15:10:00 +05:00
parent 3834e4811d
commit 26283d5432
8 changed files with 849 additions and 0 deletions
+192
View File
@@ -0,0 +1,192 @@
"""Извлечение позиций прайса из «сетки» строк.
Алгоритм: найти строку заголовков → определить роли колонок (название, код,
единица, ценовые тарифы) → пройти строки, отделяя заголовки секций от позиций.
Один и тот же код работает для xlsx, xls, docx и таблиц, восстановленных из PDF.
Распознавание ролей колонок учитывает реальные ловушки прайсов:
- колонка «№» (последовательные 1, 2, 3…) — это индекс, а не цена;
- настоящая цена медуслуги — сотни и тысячи тенге, а не мелкое число;
- название услуги — самая «кириллическая» колонка, код — цифро-точечная.
"""
from __future__ import annotations
import statistics
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[2]))
from contracts.models import RawRow # noqa: E402
from etl.extractors.common import ( # noqa: E402
_NAME_KEYS,
_UNIT_KEYS,
Grid,
classify_price_tier,
clean,
find_header_row,
is_real_name,
looks_like_code,
parse_price,
)
_TIER_ORDER = ["resident", "nonresident", "cis", "far", "insurance", "partner"]
_CODE_HEADER_KEYS = ("код", "тарификатор")
_INDEX_HEADER_KEYS = ("", "п/п", "n п")
_MIN_REAL_PRICE = 100 # медуслуга дешевле сотни тенге не бывает — отсекает индексы и мелочь
def _col_values(sample: list[list[str]], c: int) -> list[str]:
return [r[c] for r in sample if c < len(r) and r[c]]
def _col_prices(sample: list[list[str]], c: int) -> list[float]:
return [p for r in sample if c < len(r) and (p := parse_price(r[c]))]
def _cyrillic_score(sample: list[list[str]], c: int) -> float:
"""Средняя доля кириллических букв в колонке — признак колонки названий."""
total = 0
for r in sample:
if c < len(r):
total += sum(1 for ch in r[c] if "а" <= ch.lower() <= "я" or ch.lower() == "ё")
return total / max(1, len(sample))
def _is_index_column(values: list[float]) -> bool:
"""Колонка-индекс: целые, монотонные, шаг около 1, старт у единицы."""
if len(values) < 5:
return False
integers = [v for v in values if abs(v - round(v)) < 1e-9]
if len(integers) < 0.9 * len(values):
return False
ordered = sorted(round(v) for v in values)
steps = [ordered[i + 1] - ordered[i] for i in range(len(ordered) - 1)]
return ordered[0] <= 3 and statistics.mean(steps) < 1.5
def _assign_columns(grid: Grid, header_idx: int):
"""Определить роли колонок по заголовку и выборке строк под ним."""
header = [clean(c) for c in grid[header_idx]]
ncol = max((len(r) for r in grid), default=0)
header += [""] * (ncol - len(header))
sample = [[clean(c) for c in r] for r in grid[header_idx + 1 : header_idx + 60]]
# --- ценовые колонки ---
price_cols: dict[int, str | None] = {}
for c in range(ncol):
head = header[c].lower()
if any(k in head for k in _INDEX_HEADER_KEYS) or any(k in head for k in _CODE_HEADER_KEYS):
continue # «№» и «Код» — не цены
prices = _col_prices(sample, c)
nonempty = _col_values(sample, c)
ratio = len(prices) / len(nonempty) if nonempty else 0
if not prices or _is_index_column(prices):
continue
tier = classify_price_tier(header[c])
if tier:
price_cols[c] = tier
elif ratio >= 0.6 and statistics.median(prices) >= _MIN_REAL_PRICE:
price_cols[c] = None # числовая колонка без явного тарифа — назначим по позиции
used = {t for t in price_cols.values() if t}
fallback = [t for t in _TIER_ORDER if t not in used]
for c in sorted(price_cols):
if price_cols[c] is None:
price_cols[c] = fallback.pop(0) if fallback else f"extra_{c}"
# --- колонка названия: явный заголовок, иначе самая кириллическая ---
name_col = next(
(
c
for c in range(ncol)
if c not in price_cols and any(k in header[c].lower() for k in _NAME_KEYS)
),
None,
)
if name_col is None:
candidates = [c for c in range(ncol) if c not in price_cols]
name_col = max(candidates, key=lambda c: _cyrillic_score(sample, c)) if candidates else 0
# --- колонка кода: заголовок «Код»/«тарификатор» или код-подобные значения ---
code_col = next(
(
c
for c in range(ncol)
if c != name_col
and c not in price_cols
and any(k in header[c].lower() for k in _CODE_HEADER_KEYS)
),
None,
)
if code_col is None:
for c in range(ncol):
if c == name_col or c in price_cols:
continue
vals = _col_values(sample, c)
if vals and sum(1 for v in vals if looks_like_code(v)) >= max(2, 0.5 * len(vals)):
code_col = c
break
unit_col = next(
(
c
for c in range(ncol)
if c not in price_cols and any(k in header[c].lower() for k in _UNIT_KEYS)
),
None,
)
return name_col, code_col, unit_col, price_cols
def extract_rows_from_grid(grid: Grid) -> list[RawRow]:
"""Превратить сетку в список позиций прайса."""
grid = [[clean(c) for c in r] for r in grid if any(clean(c) for c in r)]
if len(grid) < 2:
return []
header_idx = find_header_row(grid)
name_col, code_col, unit_col, price_cols = _assign_columns(grid, header_idx)
rows: list[RawRow] = []
section: str | None = None
for r in grid[header_idx + 1 :]:
nonempty = [c for c in r if c]
if not nonempty:
continue
prices: dict[str, float] = {}
for c, tier in price_cols.items():
if c < len(r):
price = parse_price(r[c])
if price:
prices[tier] = price
# Заголовок секции: одна осмысленная ячейка и ни одной цены.
if not prices and len(nonempty) == 1:
section = nonempty[0]
continue
name = r[name_col] if name_col < len(r) else ""
if not prices or not is_real_name(name):
continue # не позиция: нет цены либо в «названии» код/число, а не услуга
rows.append(
RawRow(
service_name_raw=name,
service_code_source=(
r[code_col]
if code_col is not None and code_col < len(r) and r[code_col]
else None
),
prices=prices,
unit=(
r[unit_col]
if unit_col is not None and unit_col < len(r) and r[unit_col]
else None
),
section=section,
)
)
return rows