import os
import pandas as pd
import numpy as np
import xgboost as xgb
from numpy import floating
from sklearn.model_selection import train_test_split, StratifiedKFold
from sklearn.preprocessing import LabelEncoder, StandardScaler
from sklearn.metrics import (
    precision_recall_fscore_support,
    accuracy_score,
    log_loss,
)
from sqlalchemy.orm import Session
from models import ArticleSize, Alias, SizeBoundary
import logging
import json
import shutil
from typing import List, Dict, Optional, Union, Tuple, Any
from time import time
from datetime import datetime
from pathlib import Path
from sqlalchemy import text
from concurrent.futures import ThreadPoolExecutor
import threading
import re
import joblib
from sklearn.utils.class_weight import compute_sample_weight
from sklearn.calibration import CalibratedClassifierCV
from logging_config import get_training_logger, setup_logging
from training_status import TrainingStatusStore
from training_webhook import TrainingWebhookClient
from prediction_log import PredictionLogger
from model_report import build_models_report, build_version_report, compare_versions, load_metrics

setup_logging()
logger = logging.getLogger(__name__)
train_logger = get_training_logger()

_SERVICE_LOCK = threading.Lock()
_SERVICE_INSTANCE: Optional["ModelService"] = None
_SHARED_EXECUTOR = ThreadPoolExecutor(max_workers=1)
_TRAINING_LOCK = threading.Lock()
_CANCEL_EVENT = threading.Event()
_TRAINING_ALIVE = False
_PENDING_RESTART = False
_PENDING_RESTART_FORCE = False


class TrainingCancelled(Exception):
    """Кооперативная отмена обучения по запросу API."""


LETTER_SIZE_ORDER = [
    "XXS", "XS", "S", "SQ", "SK", "M", "MK", "ML", "MQ", "L", "LK", "LL",
    "XL", "XLK", "XLL", "XXL", "XXLK", "XXXL", "OS",
]
SHOE_IN_SIZE_RE = re.compile(r"\((\d+)\)")
CONFIDENT_GAP_PP = 30
OVERLAP_SCORE_RATIO = 0.70
MODEL_SCORE_WEIGHT = 0.55
BOUNDARY_SCORE_WEIGHT = 0.30
SALES_SCORE_WEIGHT = 0.15
RARE_CLASS_SUPPORT = 50
BOUNDARY_MIN_GRID_COVERAGE = 0.5
HIP_K_SCORE_WEIGHT = 0.25
HEIGHT_PROFILE_TOLERANCE_CM = 8.0
WEIGHT_PROFILE_MAX_OFFSET_KG = 15.0


def _normalize_size_label(size: str) -> str:
    """Нормализует кириллические буквы в ярлыке размера без схлопывания перчаточных размеров."""
    if size is None:
        return ""
    text_size = str(size).strip()
    text_size = text_size.replace("МК", "MK").replace("мк", "MK")
    text_size = text_size.replace("Х", "X").replace("х", "x")
    return text_size


def _normalize_gender(gender: Optional[str]) -> str:
    """
    Приводит пол из БД/артефактов к канону: male|female|child|universal.

    В БД встречаются и «Мужской», и ошибочные «Мужская».
    """
    text_gender = (gender or "").strip().lower()
    if text_gender.startswith("муж"):
        return "male"
    if text_gender.startswith("жен"):
        return "female"
    if text_gender.startswith("дет"):
        return "child"
    if text_gender.startswith("унив"):
        return "universal"
    return "universal"


def _item_type_bucket(item_type: Optional[str]) -> str:
    """Грубая корзина типа товара для признаков модели."""
    text_type = (item_type or "").lower()
    if "вейдерсы" in text_type:
        return "wader"
    if any(token in text_type for token in ("сапоги", "ботинки", "кроссовки", "носки")):
        return "footwear"
    if "перчат" in text_type:
        return "gloves"
    return "other"


def _size_letter_key(size: str) -> str:
    """Буквенный ключ размера (L, LK, XLK…) без числового префикса обуви."""
    text_size = _normalize_size_label(size).upper()
    shoe = SHOE_IN_SIZE_RE.search(text_size)
    if shoe:
        text_size = text_size[shoe.end() :]
    text_size = re.sub(r"^[\d\.\-\(\)\s]+", "", text_size)
    return text_size.strip("- ")


def _is_hip_k_letter(letter: str) -> bool:
    """Размер с K — отдельный размер под большой низ/вес (MK, LK, XLK…), не «длинный»."""
    return bool(letter) and letter.endswith("K")


def _base_letter_from_k(letter: str) -> Optional[str]:
    """LK→L, MK→M, XLK→XL, XXLK→XXL."""
    if not _is_hip_k_letter(letter):
        return None
    return letter[:-1]


def _category_matches(item_type: Optional[str], boundary_category: Optional[str]) -> bool:
    """Совпадение типа товара с category границы без грубого bucket other."""
    left = (item_type or "").strip().lower()
    right = (boundary_category or "").strip().lower()
    if not left or not right:
        return False
    return left == right or left in right or right in left


def _value_in_range(value: float, low: Optional[float], high: Optional[float]) -> Optional[bool]:
    """Проверка попадания в диапазон. None — диапазон не задан."""
    if low is None and high is None:
        return None
    if low is not None and high is not None:
        return bool(low <= value <= high)
    if low is not None:
        return bool(value >= low)
    return bool(value <= high)


LEGACY_FEATURE_COLUMNS = [
    "height",
    "weight",
    "shoe_size",
    "bmi",
    "weight_height_ratio",
    "height_is_range",
    "weight_is_range",
    "shoe_is_range",
    "gender_male",
    "gender_female",
    "gender_child",
    "gender_universal",
]


def _topk_accuracy(y_true: np.ndarray, proba: np.ndarray, k: int) -> floating[Any]:
    """Top-k accuracy: true класс входит в k крупнейших вероятностей."""
    k = min(k, proba.shape[1])
    topk = np.argpartition(-proba, kth=k - 1, axis=1)[:, :k]
    return np.mean([y_true[i] in topk[i] for i in range(len(y_true))])


def _confident_share(proba: np.ndarray, gap: float = 0.30) -> float:
    """Доля примеров, где разница top1-top2 вероятностей >= gap."""
    if proba.shape[1] < 2:
        return 1.0
    part = np.partition(-proba, kth=1, axis=1)
    top1 = -part[:, 0]
    top2 = -part[:, 1]
    return float(np.mean((top1 - top2) >= gap))


def _per_class_metrics(y_true: np.ndarray, y_pred: np.ndarray, labels: List[int], label_names: List[str]) -> Dict[str, Dict]:
    p, r, f1, s = precision_recall_fscore_support(y_true, y_pred, labels=labels, zero_division=0)
    res = {}
    for i, lab in enumerate(labels):
        res[label_names[i]] = {
            "precision": float(p[i]),
            "recall": float(r[i]),
            "f1": float(f1[i]),
            "support": int(s[i])
        }
    return res


def _size_sort_key(size: str) -> Tuple[int, float, str]:
    """Ключ сортировки размерной сетки артикула."""
    upper = _normalize_size_label(size).upper()
    shoe = SHOE_IN_SIZE_RE.search(size)
    shoe_num = float(shoe.group(1)) if shoe else -1.0
    if upper in LETTER_SIZE_ORDER:
        return (0, float(LETTER_SIZE_ORDER.index(upper)), size)
    nums = re.findall(r"\d+(?:[.,]\d+)?", size.replace(",", "."))
    if nums:
        try:
            return (1, float(nums[0]), size)
        except ValueError:
            pass
    return (2, shoe_num, size)


def get_model_service(db: Session, model_dir: Optional[str] = None) -> "ModelService":
    """
    Возвращает process-wide singleton ModelService с обновлённой DB-сессией.

    :param db: SQLAlchemy session текущего запроса.
    :param model_dir: Опциональный путь к каталогу моделей.
    """
    global _SERVICE_INSTANCE
    with _SERVICE_LOCK:
        if _SERVICE_INSTANCE is None:
            path = model_dir or os.getenv("MODEL_DIR", "/app/models")
            _SERVICE_INSTANCE = ModelService(db, path)
        else:
            _SERVICE_INSTANCE.db = db
        return _SERVICE_INSTANCE


class ModelService:
    def __init__(self, db: Session, model_dir: str = "/app/models"):
        self.db = db
        self.model_dir = Path(model_dir)
        self.model_dir.mkdir(parents=True, exist_ok=True, mode=0o777)
        self.model = None
        self.label_encoder = None
        self.scaler = None
        self.article_genders_map: Dict[str, str] = {}
        self.article_item_types_map: Dict[str, str] = {}
        self.article_sizes_map: Dict[str, List[str]] = {}
        self.article_size_sales: Dict[str, Dict[str, int]] = {}
        self.size_boundaries: List[Dict[str, Any]] = []
        self.size_profiles = None
        self.class_index_map: Dict[str, int] = {}
        self.class_index_map_norm: Dict[str, int] = {}
        self.class_support: Dict[str, int] = {}
        self.holdout_meta: Optional[pd.DataFrame] = None
        self.imputation_medians: Dict[str, float] = {
            "height": 175.0,
            "weight": 75.0,
            "shoe_size": 42.0,
        }
        self.base_feature_columns = ["height", "weight", "shoe_size"]
        self.engineered_feature_columns = [
            "bmi",
            "weight_height_ratio",
            "height_is_range",
            "weight_is_range",
            "shoe_is_range",
            "height_missing",
            "weight_missing",
            "shoe_missing",
            "gender_male",
            "gender_female",
            "gender_child",
            "gender_universal",
            "item_type_wader",
            "item_type_footwear",
            "item_type_gloves",
            "item_type_other",
        ]
        self.feature_columns = self.base_feature_columns + self.engineered_feature_columns
        self.keep_versions = max(3, int(os.getenv("MODEL_KEEP_VERSIONS", "5")))
        self.calibration_method = os.getenv("CALIBRATION_METHOD", "isotonic")
        self.status_store = TrainingStatusStore(self.model_dir / "training_status.json")
        self.webhook = TrainingWebhookClient()
        self.prediction_logger = PredictionLogger(self.model_dir / "predictions.jsonl")
        self.current_version = self._read_active_version()
        self._initialize_dirs()
        if self.current_version > 0:
            try:
                self._load_artifacts(self.current_version)
            except Exception as e:
                logger.warning("Initial model load failed: %s", e)

    def _schema_feature_columns(self) -> List[str]:
        """Актуальный список признаков текущего кода (не из старых metrics)."""
        return list(self.base_feature_columns + self.engineered_feature_columns)

    def _reset_feature_columns_for_training(self) -> None:
        """Сбрасывает feature_columns перед train, чтобы не наследовать список из активной версии."""
        self.feature_columns = self._schema_feature_columns()
        train_logger.info(
            "Training feature_columns reset to schema (%s): %s",
            len(self.feature_columns),
            self.feature_columns,
        )

    def _initialize_dirs(self):
        try:
            if not os.access(self.model_dir, os.W_OK):
                logger.error(f"No write permissions for: {self.model_dir}")
                raise PermissionError(f"No write permissions for: {self.model_dir}")
            logger.info(f"Model directory: {self.model_dir}")
        except Exception as e:
            logger.error(f"Directory initialization error: {str(e)}")
            raise

    def _current_path(self) -> Path:
        return self.model_dir / "current.json"

    def _read_active_version(self) -> int:
        path = self._current_path()
        if path.exists():
            try:
                with open(path, "r", encoding="utf-8") as f:
                    data = json.load(f)
                ver = int(data.get("version", 0))
                if ver > 0 and (self.model_dir / f"v{ver}").exists():
                    return ver
            except Exception as e:
                logger.warning("Failed to read current.json: %s", e)
        return self._get_latest_version()

    def _write_active_version(self, version: int) -> None:
        payload = {"version": version, "updated_at": datetime.utcnow().isoformat()}
        tmp = self._current_path().with_suffix(".tmp")
        with open(tmp, "w", encoding="utf-8") as f:
            json.dump(payload, f, ensure_ascii=False, indent=2)
        tmp.replace(self._current_path())
        self.current_version = version

    def _train_in_background(self, force: bool = False):
        global _TRAINING_ALIVE, _PENDING_RESTART, _PENDING_RESTART_FORCE
        from database import SessionLocal
        db = SessionLocal()
        _TRAINING_ALIVE = True
        try:
            self.db = db
            train_logger.info("Starting background training")
            result = self.train(force=force)
            train_logger.info(f"Training completed: {result}")
        except TrainingCancelled as e:
            train_logger.warning("Training cancelled: %s", e)
            self.status_store.cancel(str(e) or "Обучение отменено")
            self.webhook.notify("cancelled", "Обучение модели размеров отменено", progress=0)
        except Exception as e:
            train_logger.error(f"Training error: {str(e)}", exc_info=True)
            short = str(e)[:240]
            self.status_store.fail(short)
            self.webhook.notify("failed", f"Ошибка обучения: {short}", progress=0)
        finally:
            db.close()
            _TRAINING_ALIVE = False
            _CANCEL_EVENT.clear()
            should_restart = False
            restart_force = False
            with _TRAINING_LOCK:
                if _PENDING_RESTART:
                    _PENDING_RESTART = False
                    restart_force = _PENDING_RESTART_FORCE
                    _PENDING_RESTART_FORCE = False
                    should_restart = True
            if should_restart:
                train_logger.info("Starting pending restart after cancel")
                try:
                    self.start_training(force=restart_force)
                except Exception as restart_error:
                    train_logger.error("Pending restart failed: %s", restart_error, exc_info=True)

    def start_training(self, force: bool = False) -> Dict:
        """
        Запускает асинхронное обучение.

        :param force: Активировать кандидата даже при провале gate.
        """
        with _TRAINING_LOCK:
            return self._start_training_locked(force=force)

    def _start_training_locked(self, force: bool = False) -> Dict:
        if self.status_store.is_running():
            status = self.status_store.get()
            train_logger.warning("Training already in progress")
            return {
                "run_id": status.get("run_id"),
                "message": "Training already in progress",
                "status": status.get("status") or "running",
                "status_endpoint": "/admin/train/status",
            }
        _CANCEL_EVENT.clear()
        status = self.status_store.begin_run("Обучение модели размеров запущено")
        self.webhook.notify("started", "Обучение модели размеров запущено")
        _SHARED_EXECUTOR.submit(self._train_in_background, force)
        return {
            "run_id": status.get("run_id"),
            "message": "Training started",
            "status": "running",
            "status_endpoint": "/admin/train/status",
        }

    def cancel_training(self) -> Dict:
        """
        Запрашивает отмену текущего обучения или сбрасывает зависший статус.

        :return: Актуальный статус отмены.
        """
        global _PENDING_RESTART
        with _TRAINING_LOCK:
            status = self.status_store.get()
            current = status.get("status")
            if current not in {"running", "cancelling"}:
                return {
                    "run_id": status.get("run_id"),
                    "message": "Nothing to cancel",
                    "status": current or "idle",
                    "status_endpoint": "/admin/train/status",
                    "forced": False,
                }

            _CANCEL_EVENT.set()
            if _TRAINING_ALIVE:
                updated = self.status_store.mark_cancelling("Отмена обучения запрошена")
                return {
                    "run_id": updated.get("run_id"),
                    "message": "Cancel requested",
                    "status": "cancelling",
                    "status_endpoint": "/admin/train/status",
                    "forced": False,
                }

            _PENDING_RESTART = False
            updated = self.status_store.cancel("Обучение сброшено (процесс обучения не активен)")
            _CANCEL_EVENT.clear()
            self.webhook.notify("cancelled", updated.get("message") or "Обучение сброшено", progress=0)
            return {
                "run_id": updated.get("run_id"),
                "message": updated.get("message") or "Training reset",
                "status": "cancelled",
                "status_endpoint": "/admin/train/status",
                "forced": True,
            }

    def restart_training(self, force: bool = False) -> Dict:
        """
        Отменяет текущий/зависший прогон и запускает обучение заново.

        :param force: Передаётся в следующий start_training.
        """
        global _PENDING_RESTART, _PENDING_RESTART_FORCE
        with _TRAINING_LOCK:
            status = self.status_store.get()
            current = status.get("status")
            if current in {"running", "cancelling"}:
                _CANCEL_EVENT.set()
                if _TRAINING_ALIVE:
                    _PENDING_RESTART = True
                    _PENDING_RESTART_FORCE = force
                    updated = self.status_store.mark_cancelling(
                        "Перезапуск: ожидание остановки текущего прогона"
                    )
                    return {
                        "run_id": updated.get("run_id"),
                        "message": "Restart scheduled after cancel",
                        "status": "restarting",
                        "status_endpoint": "/admin/train/status",
                        "forced": False,
                    }
                self.status_store.cancel("Обучение сброшено перед перезапуском")
                _CANCEL_EVENT.clear()
                self.webhook.notify("cancelled", "Обучение сброшено перед перезапуском", progress=0)

            return self._start_training_locked(force=force)

    def _get_latest_version(self) -> int:
        versions = [int(v.name[1:]) for v in self.model_dir.glob("v*") if v.is_dir() and v.name[1:].isdigit()]
        return max(versions, default=0)

    def _raise_if_cancelled(self) -> None:
        if _CANCEL_EVENT.is_set():
            raise TrainingCancelled("Обучение отменено")

    def _notify_progress(self, stage: str, stage_label: str, progress: int) -> None:
        self._raise_if_cancelled()
        self.status_store.set_stage(stage, stage_label, progress, message=stage_label)
        self.webhook.notify("progress", stage_label, progress=progress)

    def _parse_input_article(self, article: str) -> str:
        """Всегда берём первые 4 цифры артикула (валидируем вход)."""
        logger.debug(f"Processing input article: {article}")
        match = re.match(r'^(\d{4})(\d*.*)?$', article)
        if not match:
            logger.error(f"Invalid article format: {article}")
            raise ValueError("Недопустимый артикул: должен начинаться с минимум 4 цифр")
        extracted_article = match.group(1)
        logger.debug(f"Extracted article: {extracted_article}")
        return extracted_article

    def _pick_history_article(self, catalog_article: str, old_candidates: List[str]) -> str:
        """
        Выбирает старый артикул с историей заказов для нового каталожного.

        :param catalog_article: Артикул из запроса (новый).
        :param old_candidates: Старые артикулы из size_chart_aliases.
        """
        if not old_candidates:
            return catalog_article
        best_article = old_candidates[0]
        best_count = -1
        for old_article in old_candidates:
            sales = self.article_size_sales.get(str(old_article)) or {}
            total = int(sum(sales.values()))
            if total > best_count:
                best_count = total
                best_article = str(old_article)
        return best_article

    def _parse_parameter(self, value: Optional[Union[float, str]]) -> Tuple[Optional[float], Optional[float], Optional[float]]:
        """
        Возвращает (exact, min, max).
        exact — если число
        min — если строка вида '>160'
        max — если строка вида '<260'
        """
        if value is None:
            return None, None, None

        if isinstance(value, (int, float)):
            return float(value), None, None

        if isinstance(value, str):
            value = value.strip()
            if value.startswith('>'):
                try:
                    return None, float(value[1:].strip()), None
                except ValueError:
                    return None, None, None
            elif value.startswith('<'):
                try:
                    return None, None, float(value[1:].strip())
                except ValueError:
                    return None, None, None
            try:
                return float(value), None, None
            except ValueError:
                return None, None, None

        return None, None, None

    def _load_article_meta_maps(self) -> None:
        """Загружает gender, item_type и сетки размеров по артикулам из ArticleSize."""
        rows = self.db.query(
            ArticleSize.article,
            ArticleSize.size,
            ArticleSize.gender,
            ArticleSize.item_type,
        ).all()
        genders: Dict[str, str] = {}
        item_types: Dict[str, str] = {}
        sizes_map: Dict[str, List[str]] = {}
        for article, size, gender, item_type in rows:
            art = str(article)
            if art not in genders:
                genders[art] = gender or "Универсальный"
            if art not in item_types:
                item_types[art] = item_type or "unknown"
            sizes_map.setdefault(art, [])
            if size is not None and size not in sizes_map[art]:
                sizes_map[art].append(size)
        self.article_genders_map = genders
        self.article_item_types_map = item_types
        self.article_sizes_map = sizes_map

    def _load_size_boundaries(self) -> None:
        """Загружает границы размеров из SizeBoundary."""
        rows = self.db.query(SizeBoundary).all()
        boundaries: List[Dict[str, Any]] = []
        for row in rows:
            boundaries.append(
                {
                    "size": _normalize_size_label(row.size) if row.size else "",
                    "size_raw": row.size,
                    "article": str(row.article) if row.article else None,
                    "category": row.category,
                    "height_min": float(row.height_min) if row.height_min is not None else None,
                    "height_max": float(row.height_max) if row.height_max is not None else None,
                    "weight_min": float(row.weight_min) if row.weight_min is not None else None,
                    "weight_max": float(row.weight_max) if row.weight_max is not None else None,
                }
            )
        self.size_boundaries = boundaries
        train_logger.info("Loaded %s size boundaries", len(boundaries))

    def _build_article_size_sales(self, df: pd.DataFrame) -> None:
        """Строит prior продаж размера по артикулу из заказов."""
        sales: Dict[str, Dict[str, int]] = {}
        grouped = df.groupby(["article", "size"]).size()
        for (article, size), count in grouped.items():
            art = str(article)
            sales.setdefault(art, {})
            sales[art][str(size)] = int(count)
        self.article_size_sales = sales

    def _load_data_batch(self, batch_size: int = 50000) -> pd.DataFrame:
        offset = 0
        dfs = []
        while True:
            query = text("""
                         SELECT o.id,
                                o.article,
                                o.size,
                                o.height,
                                o.height_min,
                                o.height_max,
                                o.weight,
                                o.weight_min,
                                o.weight_max,
                                o.shoe_size,
                                o.shoe_size_min,
                                o.shoe_size_max,
                                o.status
                         FROM size_chart_orders o
                         WHERE o.status = TRUE
                           AND o.size IS NOT NULL
                           AND (o.height IS NOT NULL OR o.height_min IS NOT NULL OR o.height_max IS NOT NULL
                             OR o.weight IS NOT NULL OR o.weight_min IS NOT NULL OR o.weight_max IS NOT NULL
                             OR o.shoe_size IS NOT NULL OR o.shoe_size_min IS NOT NULL OR o.shoe_size_max IS NOT NULL)
                         ORDER BY o.id LIMIT :limit
                         OFFSET :offset
                         """)
            df = pd.read_sql(query, self.db.bind, params={"limit": batch_size, "offset": offset})
            if df.empty:
                break
            dfs.append(df)
            offset += batch_size
        if not dfs:
            raise ValueError("No data available from size_chart_orders")
        orders_df = pd.concat(dfs, ignore_index=True)
        orders_df["size"] = orders_df["size"].astype(str).map(_normalize_size_label)
        self._load_article_meta_maps()
        self._load_size_boundaries()
        training_data = []
        for _, row in orders_df.iterrows():
            article = str(row["article"])
            training_data.append(
                {
                    "article": article,
                    "size": row["size"],
                    "height": row["height"],
                    "weight": row["weight"],
                    "shoe_size": row["shoe_size"],
                    "height_min": row["height_min"],
                    "height_max": row["height_max"],
                    "weight_min": row["weight_min"],
                    "weight_max": row["weight_max"],
                    "shoe_size_min": row["shoe_size_min"],
                    "shoe_size_max": row["shoe_size_max"],
                    "gender": self.article_genders_map.get(article, "Универсальный"),
                    "item_type": self.article_item_types_map.get(article, "unknown"),
                    "source": "orders",
                }
            )
        training_df = pd.DataFrame(training_data)
        self._build_article_size_sales(training_df)
        return training_df

    def _get_possible_sizes(self, article: str) -> List[str]:
        sizes = [size for size, in self.db.query(ArticleSize.size)
                 .filter_by(article=article).distinct()]
        logger.debug(f"Possible sizes for article {article}: {sizes}")
        return sorted(list(set(sizes)))

    def _get_article_gender(self, article: str) -> str:
        if article in self.article_genders_map:
            return self.article_genders_map[article]
        gender = self.db.query(ArticleSize.gender).filter_by(article=article).first()
        return gender[0] if gender else "Универсальный"

    def _get_article_item_type(self, article: str) -> str:
        if article in self.article_item_types_map:
            return self.article_item_types_map[article]
        row = self.db.query(ArticleSize.item_type).filter_by(article=article).first()
        return row[0] if row else "unknown"
    def _save_artifacts(self, version: int, metrics: Dict):
        """
        Сохраняет артефакты версии модели.

        :param version: Номер версии.
        :param metrics: Метрики и метаданные.
        """
        try:
            version_dir = self.model_dir / f"v{version}"
            version_dir.mkdir(parents=True, exist_ok=True)
            with open(version_dir / "metrics.json", "w", encoding="utf-8") as f:
                json.dump(metrics, f, ensure_ascii=False, indent=2)
            joblib.dump(self.label_encoder, version_dir / "label_encoder.pkl")
            joblib.dump(self.scaler, version_dir / "scaler.pkl")
            joblib.dump(self.model, version_dir / "model.pkl")
            with open(version_dir / "imputation_medians.json", "w", encoding="utf-8") as f:
                json.dump(self.imputation_medians, f, ensure_ascii=False, indent=2)
            base_model_path = version_dir / "model.json"
            if hasattr(self.model, "estimator"):
                base_xgb = self.model.estimator
            else:
                base_xgb = self.model
            if hasattr(base_xgb, "save_model"):
                base_xgb.save_model(str(base_model_path))
            else:
                joblib.dump(base_xgb, version_dir / "model_base.pkl")
            with open(version_dir / "article_genders.json", "w", encoding="utf-8") as f:
                json.dump(self.article_genders_map, f, ensure_ascii=False, indent=2)
            with open(version_dir / "article_item_types.json", "w", encoding="utf-8") as f:
                json.dump(self.article_item_types_map, f, ensure_ascii=False, indent=2)
            with open(version_dir / "article_sizes_map.json", "w", encoding="utf-8") as f:
                json.dump(self.article_sizes_map, f, ensure_ascii=False, indent=2)
            with open(version_dir / "article_size_sales.json", "w", encoding="utf-8") as f:
                json.dump(self.article_size_sales, f, ensure_ascii=False, indent=2)
            with open(version_dir / "size_boundaries.json", "w", encoding="utf-8") as f:
                json.dump(self.size_boundaries, f, ensure_ascii=False, indent=2)
            with open(version_dir / "size_profiles.json", "w", encoding="utf-8") as f:
                json.dump(self.size_profiles or {}, f, ensure_ascii=False, indent=2)
            train_logger.info(f"Артефакты сохранены в {version_dir}")
        except Exception as e:
            train_logger.error(f"Error saving artifacts: {e}", exc_info=True)
            raise

    def _load_artifacts(self, version: int):
        """
        Загружает артефакты версии в память.

        :param version: Номер версии.
        """
        try:
            version_dir = self.model_dir / f"v{version}"
            if not version_dir.exists():
                raise FileNotFoundError(f"Директория модели не найдена: {version_dir}")
            le_path = version_dir / "label_encoder.pkl"
            if not le_path.exists():
                raise FileNotFoundError(f"Не найден label_encoder: {le_path}")
            self.label_encoder = joblib.load(le_path)
            scaler_path = version_dir / "scaler.pkl"
            if not scaler_path.exists():
                raise FileNotFoundError(f"Не найден scaler: {scaler_path}")
            self.scaler = joblib.load(scaler_path)
            model_pkl_path = version_dir / "model.pkl"
            model_json_path = version_dir / "model.json"
            if model_pkl_path.exists():
                self.model = joblib.load(model_pkl_path)
                logger.info(f"Загружена калиброванная модель из {model_pkl_path}")
            elif model_json_path.exists():
                self.model = xgb.XGBClassifier()
                self.model.load_model(str(model_json_path))
                logger.info(f"Загружена старая модель из {model_json_path}")
            else:
                raise FileNotFoundError("Не найдена модель: ни model.pkl, ни model.json")
            metrics = load_metrics(self.model_dir, version) or {}
            if metrics.get("feature_columns"):
                self.feature_columns = list(metrics["feature_columns"])
            elif hasattr(self.scaler, "n_features_in_") and int(self.scaler.n_features_in_) == len(LEGACY_FEATURE_COLUMNS):
                self.feature_columns = list(LEGACY_FEATURE_COLUMNS)
            medians_path = version_dir / "imputation_medians.json"
            if medians_path.exists():
                with open(medians_path, "r", encoding="utf-8") as f:
                    self.imputation_medians = json.load(f)
            gender_path = version_dir / "article_genders.json"
            if gender_path.exists():
                with open(gender_path, "r", encoding="utf-8") as f:
                    self.article_genders_map = json.load(f)
            else:
                self.article_genders_map = {}
            item_types_path = version_dir / "article_item_types.json"
            if item_types_path.exists():
                with open(item_types_path, "r", encoding="utf-8") as f:
                    self.article_item_types_map = json.load(f)
            else:
                self.article_item_types_map = {}
            sizes_map_path = version_dir / "article_sizes_map.json"
            if sizes_map_path.exists():
                with open(sizes_map_path, "r", encoding="utf-8") as f:
                    self.article_sizes_map = json.load(f)
            else:
                self.article_sizes_map = {}
            sales_path = version_dir / "article_size_sales.json"
            if sales_path.exists():
                with open(sales_path, "r", encoding="utf-8") as f:
                    self.article_size_sales = json.load(f)
            else:
                self.article_size_sales = {}
            boundaries_path = version_dir / "size_boundaries.json"
            if boundaries_path.exists():
                with open(boundaries_path, "r", encoding="utf-8") as f:
                    self.size_boundaries = json.load(f)
            else:
                try:
                    self._load_size_boundaries()
                except Exception:
                    self.size_boundaries = []
            size_profiles_path = version_dir / "size_profiles.json"
            if size_profiles_path.exists():
                with open(size_profiles_path, "r", encoding="utf-8") as f:
                    raw_profiles = json.load(f)
                self.size_profiles = {
                    k: {
                        "height": float(v["height"]),
                        "weight": float(v["weight"]),
                        "count": int(v["count"]),
                    }
                    for k, v in raw_profiles.items()
                }
            else:
                self.size_profiles = None
            self.class_index_map = {
                str(name): idx for idx, name in enumerate(self.label_encoder.classes_)
            }
            self.class_index_map_norm = {
                _normalize_size_label(str(name)): idx
                for idx, name in enumerate(self.label_encoder.classes_)
            }
            dist = (metrics.get("class_distribution") or {})
            self.class_support = {str(k): int(v) for k, v in dist.items()}
            self.current_version = version
            logger.info(f"Артефакты загружены из v{version}")
        except Exception as e:
            logger.error(f"Error loading artifacts: {e}", exc_info=True)
            raise

    def _build_features(self, df: pd.DataFrame) -> pd.DataFrame:
        df = df.copy()
        if "item_type" not in df.columns:
            df["item_type"] = df["article"].map(self.article_item_types_map).fillna("unknown")
        buckets = df["item_type"].map(_item_type_bucket)
        is_footwear = buckets == "footwear"
        df["height_is_range"] = ((df["height"].isna()) & (df["height_min"].notna() | df["height_max"].notna())).astype(
            "float32"
        )
        df["weight_is_range"] = ((df["weight"].isna()) & (df["weight_min"].notna() | df["weight_max"].notna())).astype(
            "float32"
        )
        df["shoe_is_range"] = (
            (df["shoe_size"].isna()) & (df["shoe_size_min"].notna() | df["shoe_size_max"].notna())
        ).astype("float32")
        for param in ["height", "weight", "shoe_size"]:
            both = df[f"{param}_min"].notna() & df[f"{param}_max"].notna()
            if both.any():
                mid = (df.loc[both, f"{param}_min"].astype("float32") + df.loc[both, f"{param}_max"].astype("float32")) / 2.0
                df.loc[both, param] = mid
            only_min = df[param].isna() & df[f"{param}_min"].notna()
            only_max = df[param].isna() & df[f"{param}_max"].notna()
            df.loc[only_min, param] = df.loc[only_min, f"{param}_min"]
            df.loc[only_max, param] = df.loc[only_max, f"{param}_max"]
        df.loc[is_footwear, "height"] = np.nan
        df.loc[is_footwear, "weight"] = np.nan
        df.loc[is_footwear, "height_min"] = np.nan
        df.loc[is_footwear, "height_max"] = np.nan
        df.loc[is_footwear, "weight_min"] = np.nan
        df.loc[is_footwear, "weight_max"] = np.nan
        df.loc[is_footwear, "height_is_range"] = 0.0
        df.loc[is_footwear, "weight_is_range"] = 0.0
        for col in ["height", "weight", "shoe_size"]:
            if col == "height":
                miss_col = "height_missing"
            elif col == "weight":
                miss_col = "weight_missing"
            else:
                miss_col = "shoe_missing"
            df[miss_col] = df[col].isna().astype("float32")
            if col in ("height", "weight"):
                df.loc[is_footwear, miss_col] = 1.0
                df.loc[is_footwear, col] = 0.0
                non_fw = ~is_footwear
                median = float(self.imputation_medians.get(col, 0.0))
                df.loc[non_fw, col] = df.loc[non_fw, col].fillna(median).astype("float32")
            else:
                median = float(self.imputation_medians.get(col, 0.0))
                df[col] = df[col].fillna(median).astype("float32")
        h_m = df["height"].astype("float32") / 100.0
        w = df["weight"].astype("float32")
        with np.errstate(divide="ignore", invalid="ignore"):
            df["bmi"] = (w / (h_m ** 2)).replace([np.inf, -np.inf], np.nan).fillna(0).astype("float32")
            df["weight_height_ratio"] = (
                (w / df["height"].astype("float32")).replace([np.inf, -np.inf], np.nan).fillna(0).astype("float32")
            )
        df.loc[is_footwear, "bmi"] = 0.0
        df.loc[is_footwear, "weight_height_ratio"] = 0.0
        if "gender" not in df.columns:
            df["gender"] = df["article"].map(self.article_genders_map).fillna("Универсальный")
        gender_canon = df["gender"].map(_normalize_gender)
        df["gender_male"] = (gender_canon == "male").astype("float32")
        df["gender_female"] = (gender_canon == "female").astype("float32")
        df["gender_child"] = (gender_canon == "child").astype("float32")
        df["gender_universal"] = (gender_canon == "universal").astype("float32")
        df["item_type_wader"] = (buckets == "wader").astype("float32")
        df["item_type_footwear"] = (buckets == "footwear").astype("float32")
        df["item_type_gloves"] = (buckets == "gloves").astype("float32")
        df["item_type_other"] = (buckets == "other").astype("float32")
        df = df.drop(columns=[c for c in ["gender", "item_type"] if c in df.columns], errors="ignore")
        return df

    def _prepare_xy(self, df: pd.DataFrame, min_class_count: int = 2) -> Tuple[pd.DataFrame, np.ndarray, pd.DataFrame]:
        """
        Готовит X,y и метаданные строк для product-метрик.

        :param df: Датафрейм с признаками.
        :param min_class_count: Минимальное число примеров класса.
        """
        df = df[df["size"].notna()].copy()
        df["size"] = df["size"].astype(str).map(_normalize_size_label)
        self.label_encoder = LabelEncoder()
        df["target"] = self.label_encoder.fit_transform(df["size"])
        vc = df["target"].value_counts()
        valid_targets = vc[vc >= min_class_count].index
        df = df[df["target"].isin(valid_targets)].copy()
        self.label_encoder = LabelEncoder()
        df["target"] = self.label_encoder.fit_transform(df["size"])
        for col in self.feature_columns:
            if col not in df.columns:
                df[col] = 0.0
        meta = pd.DataFrame(
            {
                "article": df["article"].astype(str).to_numpy(copy=True) if "article" in df.columns else np.array([""] * len(df)),
                "size": df["size"].astype(str).to_numpy(copy=True),
                "height": df["height"].astype("float32").to_numpy(copy=True) if "height" in df.columns else np.full(len(df), np.nan),
                "weight": df["weight"].astype("float32").to_numpy(copy=True) if "weight" in df.columns else np.full(len(df), np.nan),
                "shoe_size": df["shoe_size"].astype("float32").to_numpy(copy=True) if "shoe_size" in df.columns else np.full(len(df), np.nan),
                "height_missing": df["height_missing"].astype("float32").to_numpy(copy=True) if "height_missing" in df.columns else np.zeros(len(df), dtype="float32"),
                "weight_missing": df["weight_missing"].astype("float32").to_numpy(copy=True) if "weight_missing" in df.columns else np.zeros(len(df), dtype="float32"),
                "shoe_missing": df["shoe_missing"].astype("float32").to_numpy(copy=True) if "shoe_missing" in df.columns else np.zeros(len(df), dtype="float32"),
            }
        )
        self.scaler = StandardScaler()
        df[self.feature_columns] = self.scaler.fit_transform(df[self.feature_columns].astype("float32"))
        X = df[self.feature_columns].reset_index(drop=True)
        y = df["target"].to_numpy()
        return X, y, meta.reset_index(drop=True)

    def _collect_class_distribution(self, y: np.ndarray) -> Dict[str, int]:
        inv = self.label_encoder.inverse_transform(np.arange(len(self.label_encoder.classes_)))
        counts = pd.Series(y).value_counts().to_dict()
        by_name = {inv[k]: int(v) for k, v in counts.items()}
        return by_name

    def _cross_validate(self, X: pd.DataFrame, y: np.ndarray, n_splits: int = 5) -> Dict:
        skf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)
        n_classes = len(np.unique(y))
        labels = list(range(n_classes))
        label_names = list(self.label_encoder.classes_)

        acc1_list, acc3_list = [], []
        prec_mac_list, rec_mac_list, f1_mac_list = [], [], []
        prec_w_list, rec_w_list, f1_w_list = [], [], []

        per_class_agg = {name: {"precision": [], "recall": [], "f1": [], "support": []} for name in label_names}

        fold_idx = 0
        for train_idx, val_idx in skf.split(X, y):
            fold_idx += 1
            X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]
            y_train, y_val = y[train_idx], y[val_idx]

            model = self._make_xgb(n_classes=n_classes, early_stopping_rounds=50)
            sample_weight = self._sqrt_balanced_weights(y_train)
            model.fit(
                X_train, y_train,
                sample_weight=sample_weight,
                eval_set=[(X_val, y_val)],
                verbose=False
            )

            proba = model.predict_proba(X_val)
            y_pred = np.argmax(proba, axis=1)

            # Метрики по фолду
            acc1 = accuracy_score(y_val, y_pred)
            acc3 = _topk_accuracy(y_val, proba, k=3)
            p_mac, r_mac, f1_mac, _ = precision_recall_fscore_support(y_val, y_pred, average='macro', zero_division=0)
            p_w, r_w, f1_w, _ = precision_recall_fscore_support(y_val, y_pred, average='weighted', zero_division=0)

            acc1_list.append(acc1)
            acc3_list.append(acc3)
            prec_mac_list.append(p_mac); rec_mac_list.append(r_mac); f1_mac_list.append(f1_mac)
            prec_w_list.append(p_w);     rec_w_list.append(r_w);     f1_w_list.append(f1_w)

            # По классам
            fold_per_cls = _per_class_metrics(y_val, y_pred, labels=labels, label_names=label_names)
            for name in label_names:
                per_class_agg[name]["precision"].append(fold_per_cls[name]["precision"])
                per_class_agg[name]["recall"].append(fold_per_cls[name]["recall"])
                per_class_agg[name]["f1"].append(fold_per_cls[name]["f1"])
                per_class_agg[name]["support"].append(fold_per_cls[name]["support"])

            train_logger.info(f"CV fold {fold_idx}: top1={acc1:.4f}, top3={acc3:.4f}, f1_macro={f1_mac:.4f}")

        per_class_mean = {
            name: {
                "precision": float(np.mean(vals["precision"])) if vals["precision"] else 0.0,
                "recall": float(np.mean(vals["recall"])) if vals["recall"] else 0.0,
                "f1": float(np.mean(vals["f1"])) if vals["f1"] else 0.0,
                "support": int(np.sum(vals["support"])) if vals["support"] else 0
            }
            for name, vals in per_class_agg.items()
        }

        return {
            "n_splits": n_splits,
            "accuracy_top1_mean": float(np.mean(acc1_list)),
            "accuracy_top1_std": float(np.std(acc1_list)),
            "accuracy_top3_mean": float(np.mean(acc3_list)),
            "accuracy_top3_std": float(np.std(acc3_list)),
            "precision_macro_mean": float(np.mean(prec_mac_list)),
            "recall_macro_mean": float(np.mean(rec_mac_list)),
            "f1_macro_mean": float(np.mean(f1_mac_list)),
            "precision_weighted_mean": float(np.mean(prec_w_list)),
            "recall_weighted_mean": float(np.mean(rec_w_list)),
            "f1_weighted_mean": float(np.mean(f1_w_list)),
            "per_class": per_class_mean
        }

    def _step_load_and_preprocess_data(self) -> pd.DataFrame:
        """Этап 1-3: Загрузка, очистка и построение признаков."""
        self._notify_progress("load_data", "Загрузка данных", 10)
        train_logger.info("Этап 1/7: Загрузка данных...")
        df = self._load_data_batch()
        if len(df) == 0:
            raise ValueError("Нет данных для обучения")
        self._notify_progress("clean_data", "Очистка данных", 20)
        train_logger.info("Этап 2/7: Очистка данных...")
        df = df[df["size"].notna()].copy()
        for col in ["height", "weight", "shoe_size"]:
            series = df[col].dropna()
            if len(series) > 0:
                self.imputation_medians[col] = float(series.median())
        size_data = df.dropna(subset=["height", "weight"]).copy()
        size_data = size_data.groupby("size").agg(
            mean_height=("height", "mean"),
            mean_weight=("weight", "mean"),
            count=("size", "count"),
        ).reset_index()
        self.size_profiles = {
            row["size"]: {
                "height": float(row["mean_height"]),
                "weight": float(row["mean_weight"]),
                "count": int(row["count"]),
            }
            for _, row in size_data.iterrows()
        }
        train_logger.info(f"Создан профиль размеров для {len(self.size_profiles)} размеров")
        self._notify_progress("features", "Построение признаков", 30)
        train_logger.info("Этап 3/7: Построение признаков...")
        return self._build_features(df)

    def _step_prepare_xy(self, df: pd.DataFrame) -> Tuple[pd.DataFrame, np.ndarray, pd.DataFrame]:
        """Этап 4: Подготовка X, y с масштабированием."""
        self._notify_progress("prepare_xy", "Подготовка признаков и целевой переменной", 40)
        train_logger.info("Этап 4/7: Подготовка X,y...")
        X_all, y_all, meta_all = self._prepare_xy(df, min_class_count=2)
        if len(np.unique(y_all)) < 2:
            raise ValueError("Слишком мало классов после фильтрации")
        self.holdout_meta = meta_all
        self.class_support = self._collect_class_distribution(y_all)
        return X_all, y_all, meta_all

    def _step_cross_validation(
        self, X_all: pd.DataFrame, y_all: np.ndarray, meta_all: pd.DataFrame
    ) -> Dict:
        """Этап 5: Быстрая оценка на holdout."""
        self._notify_progress("cv", "Оценка качества на отложенной выборке", 55)
        train_logger.info("Этап 5/7: Кросс-валидация...")
        indices = np.arange(len(y_all))
        idx_train, idx_val = train_test_split(
            indices, test_size=0.2, random_state=42, stratify=y_all
        )
        return self._fast_holdout_metrics(
            X_all.iloc[idx_train],
            X_all.iloc[idx_val],
            y_all[idx_train],
            y_all[idx_val],
            meta_all.iloc[idx_val].reset_index(drop=True),
        )

    def _make_xgb(self, n_classes: int, early_stopping_rounds: int = 50) -> xgb.XGBClassifier:
        """Создаёт XGBClassifier с параметрами, ориентированными на hit_first."""
        return xgb.XGBClassifier(
            objective="multi:softprob",
            num_class=n_classes,
            n_estimators=400,
            max_depth=6,
            learning_rate=0.05,
            min_child_weight=5,
            gamma=0.2,
            subsample=0.85,
            colsample_bytree=0.85,
            reg_alpha=0.05,
            reg_lambda=0.5,
            n_jobs=4,
            random_state=42,
            early_stopping_rounds=early_stopping_rounds,
            eval_metric="mlogloss",
        )

    def _sqrt_balanced_weights(self, y_train: np.ndarray) -> np.ndarray:
        """Веса классов: корень из balanced — меньше давит majority top1."""
        balanced = compute_sample_weight("balanced", y_train)
        return np.sqrt(balanced)

    def _step_train_final_model(
        self, X_all: pd.DataFrame, y_all: np.ndarray, meta_all: pd.DataFrame
    ) -> Dict[str, Any]:
        """Этап 6: Обучение финальной модели и калибровка на отдельном сплите."""
        self._notify_progress("fit", "Обучение и калибровка модели", 70)
        train_logger.info("Этап 6/7: Финальное обучение с holdout...")
        indices = np.arange(len(y_all))
        idx_trainval, idx_test = train_test_split(
            indices, test_size=0.2, random_state=42, stratify=y_all
        )
        y_trainval = y_all[idx_trainval]
        idx_fit, idx_cal = train_test_split(
            idx_trainval, test_size=0.2, random_state=42, stratify=y_trainval
        )
        X_fit, y_fit = X_all.iloc[idx_fit], y_all[idx_fit]
        X_cal, y_cal = X_all.iloc[idx_cal], y_all[idx_cal]
        X_val, y_val = X_all.iloc[idx_test], y_all[idx_test]
        meta_val = meta_all.iloc[idx_test].reset_index(drop=True)
        n_classes = len(np.unique(y_all))
        self.model = self._make_xgb(n_classes=n_classes, early_stopping_rounds=50)
        sample_weight = self._sqrt_balanced_weights(y_fit)
        self.model.fit(
            X_fit,
            y_fit,
            sample_weight=sample_weight,
            eval_set=[(X_cal, y_cal)],
            verbose=False,
        )
        method = self.calibration_method if self.calibration_method in {"isotonic", "sigmoid"} else "isotonic"
        train_logger.info("Калибровка вероятностей (%s) на отдельном сплите...", method)
        calibrated_model = CalibratedClassifierCV(self.model, method=method, cv="prefit")
        calibrated_model.fit(X_cal, y_cal)
        self.model = calibrated_model
        proba_val = self.model.predict_proba(X_val)
        y_pred_val = np.argmax(proba_val, axis=1)
        return {
            "y_val": y_val,
            "y_pred_val": y_pred_val,
            "proba_val": proba_val,
            "X_val": X_val,
            "meta_val": meta_val,
        }

    def _evaluate_gate(self, candidate_holdout: Dict[str, Any]) -> Dict[str, Any]:
        """
        Сравнивает holdout кандидата с активной версией (приоритет product API-метрик).

        :param candidate_holdout: Метрики кандидата.
        :return: Результат gate.
        """
        baseline_metrics = load_metrics(self.model_dir, self.current_version) if self.current_version else None
        baseline = (baseline_metrics or {}).get("holdout") or {}
        cand_api1 = candidate_holdout.get("hit_first_api")
        cand_api2 = candidate_holdout.get("hit_top2_api")
        base_api1 = baseline.get("hit_first_api")
        base_api2 = baseline.get("hit_top2_api")
        reasons = []
        passed = True
        if base_api1 is not None and cand_api1 is not None:
            if float(cand_api1) + 1e-9 < float(base_api1):
                passed = False
                reasons.append(f"hit_first_api {float(cand_api1):.4f} < baseline {float(base_api1):.4f}")
            if base_api2 is not None and cand_api2 is not None and float(cand_api2) + 1e-9 < float(base_api2):
                passed = False
                reasons.append(f"hit_top2_api {float(cand_api2):.4f} < baseline {float(base_api2):.4f}")
        elif cand_api1 is not None and base_api1 is None and baseline_metrics is not None:
            passed = True
            reasons = ["baseline without product API metrics — activate new evaluation protocol"]
        else:
            cand_top1 = float(candidate_holdout.get("accuracy_top1") or 0.0)
            cand_top3 = float(candidate_holdout.get("accuracy_top3") or 0.0)
            cand_ll = candidate_holdout.get("log_loss")
            base_top1 = baseline.get("accuracy_top1")
            base_top3 = baseline.get("accuracy_top3")
            base_ll = baseline.get("log_loss")
            if base_top1 is not None and cand_top1 + 1e-9 < float(base_top1):
                passed = False
                reasons.append(f"top1 {cand_top1:.4f} < baseline {float(base_top1):.4f}")
            if base_top3 is not None and cand_top3 + 1e-9 < float(base_top3):
                passed = False
                reasons.append(f"top3 {cand_top3:.4f} < baseline {float(base_top3):.4f}")
            if base_ll is not None and cand_ll is not None and float(cand_ll) > float(base_ll) + 1e-9:
                passed = False
                reasons.append(f"log_loss {float(cand_ll):.4f} > baseline {float(base_ll):.4f}")
        if baseline_metrics is None:
            passed = True
            reasons = ["no baseline — first/activate candidate"]
        return {
            "passed": passed,
            "reason": "; ".join(reasons) if reasons else None,
            "metrics_candidate": {
                "hit_first_api": cand_api1,
                "hit_top2_api": cand_api2,
                "accuracy_top1": candidate_holdout.get("accuracy_top1"),
                "accuracy_top3": candidate_holdout.get("accuracy_top3"),
                "log_loss": candidate_holdout.get("log_loss"),
            },
            "metrics_baseline": {
                "hit_first_api": base_api1,
                "hit_top2_api": base_api2,
                "accuracy_top1": baseline.get("accuracy_top1"),
                "accuracy_top3": baseline.get("accuracy_top3"),
                "log_loss": baseline.get("log_loss"),
            },
        }

    def _prune_old_versions(self) -> None:
        """Удаляет самые старые версии сверх MODEL_KEEP_VERSIONS, не трогая активную."""
        versions = sorted(
            int(v.name[1:])
            for v in self.model_dir.glob("v*")
            if v.is_dir() and v.name[1:].isdigit()
        )
        while len(versions) > self.keep_versions:
            to_delete = None
            for ver in versions:
                if ver != self.current_version:
                    to_delete = ver
                    break
            if to_delete is None:
                break
            shutil.rmtree(self.model_dir / f"v{to_delete}", ignore_errors=True)
            train_logger.info("Pruned old model version v%s", to_delete)
            versions.remove(to_delete)

    def _compute_product_api_metrics(
        self,
        proba_val: np.ndarray,
        meta_val: pd.DataFrame,
    ) -> Dict[str, float]:
        """
        Считает hit_first_api / hit_top2_api по сетке артикула и логике ответа API.

        :param proba_val: Вероятности модели на holdout.
        :param meta_val: Метаданные строк holdout.
        """
        hit_first = 0
        hit_top2 = 0
        total = 0
        for i in range(len(meta_val)):
            article = str(meta_val.iloc[i]["article"])
            true_size = _normalize_size_label(str(meta_val.iloc[i]["size"]))
            sizes = list(self.article_sizes_map.get(article) or [])
            if not sizes:
                continue
            item_type = self.article_item_types_map.get(article, "unknown")
            height = meta_val.iloc[i]["height"]
            weight = meta_val.iloc[i]["weight"]
            height_missing = float(meta_val.iloc[i].get("height_missing", 0.0) or 0.0)
            weight_missing = float(meta_val.iloc[i].get("weight_missing", 0.0) or 0.0)
            h_val = float(height) if height_missing < 0.5 and pd.notna(height) else None
            w_val = float(weight) if weight_missing < 0.5 and pd.notna(weight) else None
            scored = self._combine_size_scores(
                sizes,
                proba_val[i],
                article,
                item_type,
                h_val,
                w_val,
                height_missing,
                weight_missing,
            )
            finalized = self._finalize_size_probs(scored, article)
            returned = [_normalize_size_label(s["size"]) for s in finalized.get("sizes", [])]
            if not returned:
                continue
            total += 1
            if returned[0] == true_size:
                hit_first += 1
            if true_size in returned:
                hit_top2 += 1
        if total == 0:
            return {"hit_first_api": 0.0, "hit_top2_api": 0.0, "product_eval_rows": 0}
        return {
            "hit_first_api": float(hit_first) / float(total),
            "hit_top2_api": float(hit_top2) / float(total),
            "product_eval_rows": int(total),
        }

    def _step_save_artifacts(
        self,
        X_all: pd.DataFrame,
        y_all: np.ndarray,
        val_results: Dict,
        cv_metrics: Dict,
        force: bool = False,
    ) -> Dict[str, Any]:
        """Этап 7: Метрики, gate, сохранение и опциональная активация."""
        self._notify_progress("gate", "Проверка метрик и сохранение", 90)
        n_classes = len(np.unique(y_all))
        y_val, y_pred_val, proba_val = val_results["y_val"], val_results["y_pred_val"], val_results["proba_val"]
        meta_val = val_results.get("meta_val")
        holdout: Dict[str, Any] = {}
        holdout["accuracy_top1"] = float(accuracy_score(y_val, y_pred_val))
        holdout["accuracy_top2"] = float(_topk_accuracy(y_val, proba_val, k=2))
        holdout["accuracy_top3"] = float(_topk_accuracy(y_val, proba_val, k=3))
        holdout["confident_share"] = float(_confident_share(proba_val, gap=CONFIDENT_GAP_PP / 100.0))
        try:
            holdout["log_loss"] = float(log_loss(y_val, proba_val, labels=list(range(n_classes))))
        except Exception:
            holdout["log_loss"] = None
        pm, rm, f1m, _ = precision_recall_fscore_support(y_val, y_pred_val, average="macro", zero_division=0)
        pw, rw, f1w, _ = precision_recall_fscore_support(y_val, y_pred_val, average="weighted", zero_division=0)
        holdout["precision_macro"] = float(pm)
        holdout["recall_macro"] = float(rm)
        holdout["f1_macro"] = float(f1m)
        holdout["precision_weighted"] = float(pw)
        holdout["recall_weighted"] = float(rw)
        holdout["f1_weighted"] = float(f1w)
        holdout["per_class"] = _per_class_metrics(
            y_val,
            y_pred_val,
            labels=list(range(n_classes)),
            label_names=list(self.label_encoder.classes_),
        )
        self.class_index_map = {str(name): idx for idx, name in enumerate(self.label_encoder.classes_)}
        self.class_index_map_norm = {
            _normalize_size_label(str(name)): idx for idx, name in enumerate(self.label_encoder.classes_)
        }
        if meta_val is not None and len(meta_val) > 0:
            product_metrics = self._compute_product_api_metrics(proba_val, meta_val)
            holdout.update(product_metrics)
        new_version = self._get_latest_version() + 1
        metrics_payload = {
            "records_used": int(len(X_all)),
            "classes": list(self.label_encoder.classes_),
            "n_classes": int(len(self.label_encoder.classes_)),
            "class_distribution": self._collect_class_distribution(y_all),
            "cv": cv_metrics,
            "holdout": holdout,
            "trained_at": datetime.utcnow().isoformat(),
            "feature_columns": self.feature_columns,
        }
        self._save_artifacts(new_version, metrics_payload)
        gate = self._evaluate_gate(holdout)
        activated = False
        rejected = False
        if gate["passed"] or force:
            self._write_active_version(new_version)
            self._load_artifacts(new_version)
            activated = True
            message = f"Обучение завершено, версия v{new_version} активирована"
            if force and not gate["passed"]:
                message = f"Обучение завершено, версия v{new_version} активирована принудительно (gate не пройден)"
        else:
            rejected = True
            message = (
                f"Обучение завершено, активация v{new_version} отклонена gate"
                + (f": {gate['reason']}" if gate.get("reason") else "")
            )
            if self.current_version > 0:
                try:
                    self._load_artifacts(self.current_version)
                except Exception:
                    pass
        self._prune_old_versions()
        return {
            "version_candidate": new_version,
            "version_activated": self.current_version,
            "activated": activated,
            "rejected": rejected,
            "gate": gate,
            "records_used": len(X_all),
            "message": message,
            "holdout": holdout,
        }

    def train(self, force: bool = False) -> Dict:
        """
        Полный цикл обучения с обновлением статуса и webhook.

        :param force: Игнорировать gate при активации.
        """
        start_time = time()
        train_logger.info("Начало обучения модели")
        try:
            self._raise_if_cancelled()
            self._reset_feature_columns_for_training()
            df_feat = self._step_load_and_preprocess_data()
            X_all, y_all, meta_all = self._step_prepare_xy(df_feat)
            cv_metrics = self._step_cross_validation(X_all, y_all, meta_all)
            val_results = self._step_train_final_model(X_all, y_all, meta_all)
            save_result = self._step_save_artifacts(X_all, y_all, val_results, cv_metrics, force=force)
            training_time = (time() - start_time) / 60
            self.status_store.complete(
                activated=save_result["activated"],
                version_candidate=save_result["version_candidate"],
                version_activated=save_result["version_activated"],
                gate=save_result["gate"],
                records_used=save_result["records_used"],
                training_time_min=round(training_time, 1),
                message=save_result["message"],
                rejected=save_result["rejected"],
            )
            self.webhook.notify("completed", save_result["message"], progress=100)
            train_logger.info(
                f"Обучение завершено за {training_time:.1f} минут, использовано {len(X_all)} строк"
            )
            return {
                "message": save_result["message"],
                "version": save_result["version_candidate"],
                "activated": save_result["activated"],
                "gate": save_result["gate"],
                "records_used": len(X_all),
                "training_time_min": round(training_time, 1),
            }
        except TrainingCancelled:
            raise
        except Exception as e:
            error_msg = f"Ошибка обучения: {str(e)}"
            train_logger.error(error_msg, exc_info=True)
            short = str(e)[:240]
            self.status_store.fail(short)
            self.webhook.notify("failed", f"Ошибка обучения: {short}", progress=0)
            raise

    def _select_boundary_row(
        self,
        size: str,
        article: str,
        item_type: str,
    ) -> Optional[Dict[str, Any]]:
        """
        Выбирает строку границы только по артикулу или той же категории.

        :param size: Размер сетки.
        :param article: Артикул.
        :param item_type: Тип товара.
        """
        if not self.size_boundaries:
            return None
        size_norm = _normalize_size_label(size)
        article_key = str(article)
        article_matches = []
        category_matches = []
        for boundary in self.size_boundaries:
            if _normalize_size_label(boundary.get("size") or "") != size_norm:
                continue
            if boundary.get("article") and str(boundary["article"]) == article_key:
                article_matches.append(boundary)
            elif _category_matches(item_type, boundary.get("category")):
                category_matches.append(boundary)
        if article_matches:
            return article_matches[0]
        if category_matches:
            return category_matches[0]
        return None

    def _boundary_fit_score(
        self,
        size: str,
        article: str,
        item_type: str,
        height: Optional[float],
        weight: Optional[float],
        height_missing: float,
        weight_missing: float,
    ) -> float:
        """
        Жёсткая оценка попадания в границы размера (0 или 1).

        :param size: Размер из сетки артикула.
        :param article: Артикул.
        :param item_type: Тип товара.
        :param height: Рост.
        :param weight: Вес.
        :param height_missing: Флаг пропуска роста.
        :param weight_missing: Флаг пропуска веса.
        """
        best = self._select_boundary_row(size, article, item_type)
        if best is None:
            return 0.0
        if height_missing >= 0.5 and weight_missing >= 0.5:
            return 0.0
        checks: List[bool] = []
        if height_missing < 0.5 and height is not None:
            height_ok = _value_in_range(float(height), best.get("height_min"), best.get("height_max"))
            if height_ok is not None:
                checks.append(bool(height_ok))
        if weight_missing < 0.5 and weight is not None:
            weight_ok = _value_in_range(float(weight), best.get("weight_min"), best.get("weight_max"))
            if weight_ok is not None:
                checks.append(bool(weight_ok))
        if not checks:
            return 0.0
        return 1.0 if all(checks) else 0.0

    def _sales_prior_score(self, article: str, size: str) -> float:
        """Prior доли продаж размера на артикуле."""
        art_sales = self.article_size_sales.get(str(article)) or {}
        if not art_sales:
            return 0.0
        total = float(sum(art_sales.values()))
        if total <= 0:
            return 0.0
        size_norm = _normalize_size_label(size)
        count = 0
        for key, value in art_sales.items():
            if _normalize_size_label(key) == size_norm or key == size:
                count += int(value)
        return float(count) / total

    def _model_prob_for_size(self, probs: np.ndarray, size: str) -> float:
        """Вероятность модели для размера с нормализацией кириллицы."""
        idx = self.class_index_map.get(size)
        if idx is None:
            idx = self.class_index_map_norm.get(_normalize_size_label(size))
        if idx is None:
            return 0.0
        return float(probs[idx])

    def _profile_for_size(self, size: str) -> Optional[Dict[str, float]]:
        """Профиль размера из обучения (mean height/weight) с нормализацией ярлыка."""
        if not self.size_profiles:
            return None
        if size in self.size_profiles:
            return self.size_profiles[size]
        size_norm = _normalize_size_label(size)
        if size_norm in self.size_profiles:
            return self.size_profiles[size_norm]
        for key, value in self.size_profiles.items():
            if _normalize_size_label(str(key)) == size_norm:
                return value
        return None

    def _height_fit_distance(
        self,
        size: str,
        article: str,
        item_type: str,
        height: float,
    ) -> Optional[float]:
        """
        Насколько рост подходит размеру: меньше = лучше.
        None — нет данных для оценки.
        """
        row = self._select_boundary_row(size, article, item_type)
        if row is not None:
            h_min = row.get("height_min")
            h_max = row.get("height_max")
            height_ok = _value_in_range(float(height), h_min, h_max)
            if height_ok is True:
                if h_min is not None and h_max is not None:
                    mid = (float(h_min) + float(h_max)) / 2.0
                    return abs(float(height) - mid)
                if h_min is not None:
                    return abs(float(height) - float(h_min))
                if h_max is not None:
                    return abs(float(height) - float(h_max))
                return 0.0
            if height_ok is False:
                return None
        profile = self._profile_for_size(size)
        if profile and profile.get("height") is not None:
            dist = abs(float(height) - float(profile["height"]))
            if dist <= HEIGHT_PROFILE_TOLERANCE_CM:
                return dist
            return None
        return None

    def _pick_base_size_by_height(
        self,
        sizes: List[str],
        article: str,
        item_type: str,
        height: float,
    ) -> Optional[str]:
        """Один обычный (не *K) размер сетки, лучший по росту."""
        best_size: Optional[str] = None
        best_dist: Optional[float] = None
        for size in sizes:
            letter = _size_letter_key(size)
            if not letter or _is_hip_k_letter(letter):
                continue
            dist = self._height_fit_distance(size, article, item_type, height)
            if dist is None:
                continue
            if best_dist is None or dist < best_dist:
                best_dist = dist
                best_size = size
        return best_size

    def _weight_max_for_base_size(
        self,
        size: str,
        article: str,
        item_type: str,
    ) -> Optional[float]:
        """Потолок веса базового размера: граница, иначе mean(заказы)+offset."""
        row = self._select_boundary_row(size, article, item_type)
        if row is not None and row.get("weight_max") is not None:
            return float(row["weight_max"])
        profile = self._profile_for_size(size)
        if profile and profile.get("weight") is not None:
            return float(profile["weight"]) + WEIGHT_PROFILE_MAX_OFFSET_KG
        return None

    def _hip_k_scores(
        self,
        sizes: List[str],
        article: str,
        item_type: str,
        height: Optional[float],
        weight: Optional[float],
        height_missing: float,
        weight_missing: float,
    ) -> Dict[str, float]:
        """
        Один базовый размер по росту; при перевесе — его *K, иначе сам базовый.

        :return: Словарь size→0..1 для усиления выбранного размера в скоринге.
        """
        scores = {size: 0.0 for size in sizes}
        if height_missing >= 0.5 or weight_missing >= 0.5 or height is None or weight is None:
            return scores
        base_size = self._pick_base_size_by_height(sizes, article, item_type, float(height))
        if not base_size:
            return scores
        weight_max = self._weight_max_for_base_size(base_size, article, item_type)
        if weight_max is None or float(weight) <= float(weight_max):
            return scores
        letter_to_size: Dict[str, str] = {}
        for size in sizes:
            letter = _size_letter_key(size)
            if letter and letter not in letter_to_size:
                letter_to_size[letter] = size
        base_letter = _size_letter_key(base_size)
        k_size = letter_to_size.get(base_letter + "K") if base_letter else None
        if k_size:
            scores[k_size] = 1.0
        else:
            scores[base_size] = 1.0
        return scores

    def _combine_size_scores(
        self,
        sizes: List[str],
        probs: np.ndarray,
        article: str,
        item_type: str,
        height: Optional[float],
        weight: Optional[float],
        height_missing: float,
        weight_missing: float,
        sales_article: Optional[str] = None,
    ) -> List[Tuple[str, float]]:
        """
        Смешивает модель, границы, продажи и правило перевес→K по сетке артикула.

        :param sales_article: Артикул для prior продаж (старый при алиасе new→old).
        :return: Список (size, score) по убыванию.
        """
        if not sizes:
            return []
        sales_key = str(sales_article or article)
        boundary_scores = {
            size: self._boundary_fit_score(
                size, article, item_type, height, weight, height_missing, weight_missing
            )
            for size in sizes
        }
        sizes_with_boundary_row = sum(
            1 for size in sizes if self._select_boundary_row(size, article, item_type) is not None
        )
        coverage = float(sizes_with_boundary_row) / float(len(sizes))
        if coverage >= BOUNDARY_MIN_GRID_COVERAGE:
            boundary_w = BOUNDARY_SCORE_WEIGHT
        else:
            boundary_w = 0.0
        hip_scores = self._hip_k_scores(
            sizes, article, item_type, height, weight, height_missing, weight_missing
        )
        hip_w = HIP_K_SCORE_WEIGHT if any(v > 0 for v in hip_scores.values()) else 0.0
        model_w = MODEL_SCORE_WEIGHT
        sales_w = SALES_SCORE_WEIGHT
        weight_sum = model_w + boundary_w + sales_w + hip_w
        if weight_sum <= 0:
            weight_sum = 1.0
        model_w /= weight_sum
        boundary_w /= weight_sum
        sales_w /= weight_sum
        hip_w /= weight_sum
        scored: List[Tuple[str, float]] = []
        for size in sizes:
            model_p = self._model_prob_for_size(probs, size)
            boundary_p = boundary_scores[size] if boundary_w > 0 else 0.0
            sales_p = self._sales_prior_score(sales_key, size)
            hip_p = hip_scores.get(size, 0.0) if hip_w > 0 else 0.0
            score = model_w * model_p + boundary_w * boundary_p + sales_w * sales_p + hip_w * hip_p
            scored.append((size, float(score)))
        total = sum(s for _, s in scored)
        if total > 0:
            scored = [(size, s / total) for size, s in scored]
        scored.sort(key=lambda x: x[1], reverse=True)
        return scored

    def _build_inference_features(
        self,
        gender: str,
        item_type: str,
        height: float,
        weight: float,
        shoe_size: float,
        height_is_range: float,
        weight_is_range: float,
        shoe_is_range: float,
        height_missing: float,
        weight_missing: float,
        shoe_missing: float,
    ) -> Dict[str, float]:
        feat = {col: 0.0 for col in self.feature_columns}
        bucket = _item_type_bucket(item_type)
        is_footwear = bucket == "footwear"
        if is_footwear:
            height_missing = 1.0
            weight_missing = 1.0
            height_is_range = 0.0
            weight_is_range = 0.0
        med_h = float(self.imputation_medians.get("height", 175.0))
        med_w = float(self.imputation_medians.get("weight", 75.0))
        med_s = float(self.imputation_medians.get("shoe_size", 42.0))
        h = 0.0 if is_footwear or height_missing >= 0.5 else float(height)
        w = 0.0 if is_footwear or weight_missing >= 0.5 else float(weight)
        if not is_footwear and height_missing >= 0.5:
            h = med_h
        if not is_footwear and weight_missing >= 0.5:
            w = med_w
        s = float(shoe_size) if shoe_missing < 0.5 else med_s
        if "height" in feat:
            feat["height"] = h
        if "weight" in feat:
            feat["weight"] = w
        if "shoe_size" in feat:
            feat["shoe_size"] = s
        if "height_is_range" in feat:
            feat["height_is_range"] = height_is_range
        if "weight_is_range" in feat:
            feat["weight_is_range"] = weight_is_range
        if "shoe_is_range" in feat:
            feat["shoe_is_range"] = shoe_is_range
        if "height_missing" in feat:
            feat["height_missing"] = height_missing
        if "weight_missing" in feat:
            feat["weight_missing"] = weight_missing
        if "shoe_missing" in feat:
            feat["shoe_missing"] = shoe_missing
        h_m = h / 100.0 if h > 0 else 0.0
        if "bmi" in feat:
            feat["bmi"] = (w / (h_m ** 2)) if h_m > 0 and not is_footwear else 0.0
        if "weight_height_ratio" in feat:
            feat["weight_height_ratio"] = (w / h) if h > 0 and not is_footwear else 0.0
        gender_canon = _normalize_gender(gender)
        if "gender_male" in feat:
            feat["gender_male"] = 1.0 if gender_canon == "male" else 0.0
        if "gender_female" in feat:
            feat["gender_female"] = 1.0 if gender_canon == "female" else 0.0
        if "gender_child" in feat:
            feat["gender_child"] = 1.0 if gender_canon == "child" else 0.0
        if "gender_universal" in feat:
            feat["gender_universal"] = 1.0 if gender_canon == "universal" else 0.0
        if "item_type_wader" in feat:
            feat["item_type_wader"] = 1.0 if bucket == "wader" else 0.0
        if "item_type_footwear" in feat:
            feat["item_type_footwear"] = 1.0 if bucket == "footwear" else 0.0
        if "item_type_gloves" in feat:
            feat["item_type_gloves"] = 1.0 if bucket == "gloves" else 0.0
        if "item_type_other" in feat:
            feat["item_type_other"] = 1.0 if bucket == "other" else 0.0
        return feat

    def _finalize_size_probs(
        self,
        size_probs: List[Tuple[str, float]],
        final_article: str,
    ) -> Dict[str, Any]:
        """
        Формирует 1–2 размера по скору без подмены соседями.

        :param size_probs: Отсортированные (size, score).
        :param final_article: Артикул для SKU.
        """
        size_probs = sorted(size_probs, key=lambda x: x[1], reverse=True)
        if not size_probs:
            return {"sizes": [], "confident": False, "confidence_gap": 0}
        selected = [size_probs[0]]
        if len(size_probs) > 1:
            top_score = size_probs[0][1]
            second_score = size_probs[1][1]
            if top_score <= 0:
                selected.append(size_probs[1])
            elif second_score / top_score >= OVERLAP_SCORE_RATIO:
                selected.append(size_probs[1])
        total_prob = sum(p for _, p in selected)
        if total_prob > 0:
            final_probs = [(s, p / total_prob * 100.0) for s, p in selected]
        else:
            final_probs = [(s, 100.0 / len(selected)) for s, _ in selected]
        if len(final_probs) == 1:
            final_probs = [(final_probs[0][0], 100.0)]
        elif len(final_probs) > 1:
            rounded_probs = [round(p) for _, p in final_probs]
            diff = 100 - sum(rounded_probs)
            max_idx = max(range(len(final_probs)), key=lambda i: final_probs[i][1])
            rounded_probs[max_idx] += diff
            final_probs = [(final_probs[i][0], float(rounded_probs[i])) for i in range(len(final_probs))]
            if any(p > 80 for p in rounded_probs):
                final_probs = [(final_probs[0][0], 100.0)]
        confident = False
        if len(final_probs) == 1:
            confident = True
        elif len(final_probs) >= 2:
            confident = (final_probs[0][1] - final_probs[1][1]) >= CONFIDENT_GAP_PP
        return {
            "sizes": [
                {"size": s, "probability": int(p), "sku": f"{final_article}-{s}"}
                for s, p in final_probs
            ],
            "confident": confident,
            "confidence_gap": (
                int(round(final_probs[0][1] - final_probs[1][1])) if len(final_probs) >= 2 else 100
            ),
        }

    def predict(
        self,
        articles: List[str],
        height: Optional[Union[float, str]] = None,
        weight: Optional[Union[float, str]] = None,
        shoe_size: Optional[Union[float, str]] = None,
    ) -> List[Dict]:
        try:
            if self.model is None or self.label_encoder is None or self.scaler is None:
                if self.current_version <= 0:
                    return [
                        {
                            "article": a,
                            "sizes": [],
                            "message": "Модель не обучена",
                            "code": "model_not_trained",
                            "confident": False,
                        }
                        for a in articles
                    ]
                self._load_artifacts(self.current_version)
            if not self.size_boundaries:
                try:
                    self._load_size_boundaries()
                except Exception:
                    pass
            if not self.article_item_types_map or not self.article_genders_map:
                try:
                    self._load_article_meta_maps()
                except Exception:
                    pass
            parsed_articles = []
            for article in articles:
                try:
                    base = self._parse_input_article(article)
                    parsed_articles.append({"input_article": article, "base_article": base})
                except ValueError as ve:
                    parsed_articles.append({"input_article": article, "error": str(ve), "code": "invalid_article"})
            if not parsed_articles:
                return []
            base_article_list = [pa["base_article"] for pa in parsed_articles if "base_article" in pa]
            if not base_article_list:
                return [
                    {
                        "article": pa["input_article"],
                        "sizes": [],
                        "message": pa.get("error", "Invalid article"),
                        "code": pa.get("code", "invalid_article"),
                        "confident": False,
                    }
                    for pa in parsed_articles
                ]
            alias_rows = (
                self.db.query(Alias.old_article, Alias.new_article)
                .filter(Alias.new_article.in_(base_article_list))
                .all()
            )
            new_to_olds: Dict[str, List[str]] = {}
            for old_article, new_article in alias_rows:
                new_to_olds.setdefault(str(new_article), []).append(str(old_article))
            for pa in parsed_articles:
                if "base_article" not in pa:
                    continue
                catalog_article = pa["base_article"]
                history_candidates = new_to_olds.get(catalog_article, [])
                history_article = self._pick_history_article(catalog_article, history_candidates)
                pa["catalog_article"] = catalog_article
                pa["final_article"] = catalog_article
                pa["history_article"] = history_article
                pa["used_alias"] = bool(history_candidates) and history_article != catalog_article
            final_articles = [pa["final_article"] for pa in parsed_articles if "final_article" in pa]
            lookup_articles = set(final_articles)
            size_data = (
                self.db.query(ArticleSize.article, ArticleSize.size, ArticleSize.gender, ArticleSize.item_type)
                .filter(ArticleSize.article.in_(list(lookup_articles)))
                .all()
            )
            article_info = {}
            for art, size, gender, item_type in size_data:
                if art not in article_info:
                    article_info[art] = {
                        "sizes": set(),
                        "gender": gender,
                        "item_type": item_type.lower() if item_type else "unknown",
                    }
                article_info[art]["sizes"].add(size)
            for art in article_info:
                article_info[art]["sizes"] = sorted(article_info[art]["sizes"])
            h_exact, h_min, h_max = self._parse_parameter(height)
            w_exact, w_min, w_max = self._parse_parameter(weight)
            shoe_exact, shoe_min, shoe_max = self._parse_parameter(shoe_size)
            results = []
            for pa in parsed_articles:
                input_article = pa["input_article"]
                if "error" in pa:
                    results.append(
                        {
                            "article": input_article,
                            "sizes": [],
                            "message": pa["error"],
                            "code": pa.get("code", "invalid_article"),
                            "confident": False,
                        }
                    )
                    continue
                final_article = pa["final_article"]
                history_article = pa.get("history_article", final_article)
                info = article_info.get(final_article)
                if not info:
                    logger.warning(
                        "Article not found in size chart: input=%s catalog=%s history=%s alias=%s",
                        input_article,
                        final_article,
                        history_article,
                        pa.get("used_alias"),
                    )
                    results.append(
                        {
                            "article": input_article,
                            "sizes": [],
                            "message": "Артикул не найден",
                            "code": "article_not_found",
                            "confident": False,
                        }
                    )
                    continue
                sizes = sorted(set(info["sizes"]))

                gender = info["gender"]
                item_type = info["item_type"]
                if not sizes:
                    results.append(
                        {
                            "article": input_article,
                            "sizes": [],
                            "message": "Нет доступных размеров",
                            "code": "no_sizes",
                            "confident": False,
                        }
                    )
                    continue
                if len(sizes) == 1:
                    results.append(
                        {
                            "article": input_article,
                            "sizes": [
                                {
                                    "size": sizes[0],
                                    "probability": 100,
                                    "sku": f"{final_article}-{sizes[0]}",
                                }
                            ],
                            "confident": True,
                            "confidence_gap": 100,
                        }
                    )
                    continue
                h_val = w_val = shoe_val = 0.0
                height_is_range = weight_is_range = shoe_is_range = 0.0
                height_missing = weight_missing = shoe_missing = 1.0
                if "вейдерсы с сапогом" in item_type:
                    if h_exact is None and h_min is None and h_max is None:
                        results.append(
                            {
                                "article": input_article,
                                "sizes": [],
                                "message": "Требуется рост",
                                "code": "height_required",
                                "confident": False,
                            }
                        )
                        continue
                    if w_exact is None and w_min is None and w_max is None:
                        results.append(
                            {
                                "article": input_article,
                                "sizes": [],
                                "message": "Требуется вес",
                                "code": "weight_required",
                                "confident": False,
                            }
                        )
                        continue
                    if shoe_exact is None and shoe_min is None and shoe_max is None:
                        results.append(
                            {
                                "article": input_article,
                                "sizes": [],
                                "message": "Требуется размер обуви",
                                "code": "shoe_size_required",
                                "confident": False,
                            }
                        )
                        continue
                    if h_exact is not None:
                        h_val = float(h_exact)
                    elif h_min is not None and h_max is not None:
                        h_val = (float(h_min) + float(h_max)) / 2.0
                    else:
                        h_val = float(h_min if h_min is not None else h_max)
                    if w_exact is not None:
                        w_val = float(w_exact)
                    elif w_min is not None and w_max is not None:
                        w_val = (float(w_min) + float(w_max)) / 2.0
                    else:
                        w_val = float(w_min if w_min is not None else w_max)
                    if shoe_exact is not None:
                        shoe_val = float(shoe_exact)
                    elif shoe_min is not None and shoe_max is not None:
                        shoe_val = (float(shoe_min) + float(shoe_max)) / 2.0
                    else:
                        shoe_val = float(shoe_min if shoe_min is not None else shoe_max)
                    height_missing = weight_missing = shoe_missing = 0.0
                    height_is_range = 1.0 if (h_exact is None and (h_min or h_max)) else 0.0
                    weight_is_range = 1.0 if (w_exact is None and (w_min or w_max)) else 0.0
                    shoe_is_range = 1.0 if (shoe_exact is None and (shoe_min or shoe_max)) else 0.0
                    client_shoe_min = shoe_min + 1 if shoe_min is not None else float("-inf")
                    filtered_sizes = []
                    for size in sizes:
                        match = SHOE_IN_SIZE_RE.search(size)
                        if match:
                            parsed_shoe = float(match.group(1))
                            if (shoe_max is None or parsed_shoe <= shoe_max - 1) and parsed_shoe >= client_shoe_min:
                                filtered_sizes.append(size)
                    if not filtered_sizes:
                        results.append(
                            {
                                "article": input_article,
                                "sizes": [],
                                "message": "Нет подходящих размеров",
                                "code": "no_matching_sizes",
                                "confident": False,
                            }
                        )
                        continue
                    sizes = filtered_sizes
                elif any(t in item_type for t in ["сапоги", "ботинки", "кроссовки", "носки"]):
                    if shoe_exact is None and shoe_min is None and shoe_max is None:
                        results.append(
                            {
                                "article": input_article,
                                "sizes": [],
                                "message": "Требуется размер обуви",
                                "code": "shoe_size_required",
                                "confident": False,
                            }
                        )
                        continue
                    if shoe_exact is not None:
                        shoe_val = float(shoe_exact)
                    elif shoe_min is not None and shoe_max is not None:
                        shoe_val = (float(shoe_min) + float(shoe_max)) / 2.0
                    else:
                        shoe_val = float(shoe_min if shoe_min is not None else shoe_max)
                    shoe_missing = 0.0
                    shoe_is_range = 1.0 if (shoe_exact is None and (shoe_min or shoe_max)) else 0.0
                else:
                    if h_exact is None and h_min is None and h_max is None:
                        results.append(
                            {
                                "article": input_article,
                                "sizes": [],
                                "message": "Требуется рост",
                                "code": "height_required",
                                "confident": False,
                            }
                        )
                        continue
                    if w_exact is None and w_min is None and w_max is None:
                        results.append(
                            {
                                "article": input_article,
                                "sizes": [],
                                "message": "Требуется вес",
                                "code": "weight_required",
                                "confident": False,
                            }
                        )
                        continue
                    if h_exact is not None:
                        h_val = float(h_exact)
                    elif h_min is not None and h_max is not None:
                        h_val = (float(h_min) + float(h_max)) / 2.0
                    else:
                        h_val = float(h_min if h_min is not None else h_max)
                    if w_exact is not None:
                        w_val = float(w_exact)
                    elif w_min is not None and w_max is not None:
                        w_val = (float(w_min) + float(w_max)) / 2.0
                    else:
                        w_val = float(w_min if w_min is not None else w_max)
                    height_missing = weight_missing = 0.0
                    height_is_range = 1.0 if (h_exact is None and (h_min or h_max)) else 0.0
                    weight_is_range = 1.0 if (w_exact is None and (w_min or w_max)) else 0.0
                feat = self._build_inference_features(
                    gender,
                    item_type,
                    h_val,
                    w_val,
                    shoe_val,
                    height_is_range,
                    weight_is_range,
                    shoe_is_range,
                    height_missing,
                    weight_missing,
                    shoe_missing,
                )
                input_df = pd.DataFrame([feat])[self.feature_columns].astype("float32")
                input_scaled = self.scaler.transform(input_df)
                probs = self.model.predict_proba(input_scaled)[0]
                size_probs = self._combine_size_scores(
                    sizes,
                    probs,
                    final_article,
                    item_type,
                    h_val if height_missing < 0.5 else None,
                    w_val if weight_missing < 0.5 else None,
                    height_missing,
                    weight_missing,
                    sales_article=history_article,
                )
                finalized = self._finalize_size_probs(size_probs, final_article)
                results.append({"article": input_article, **finalized})
            self.prediction_logger.log(
                articles=articles,
                height=height,
                weight=weight,
                shoe_size=shoe_size,
                results=results,
                model_version=self.current_version,
            )
            return results
        except Exception as e:
            logger.error("Predict failed: %s", e, exc_info=True)
            return [
                {
                    "article": article,
                    "sizes": [],
                    "message": "Системная ошибка при подборе размера",
                    "code": "system_error",
                    "confident": False,
                }
                for article in articles
            ]

    def switch_version(self, version: int) -> Dict:
        """
        Активирует указанную версию модели (hot-reload).

        :param version: Номер версии.
        """
        if not (self.model_dir / f"v{version}").exists():
            raise ValueError(f"Version {version} does not exist")
        self._load_artifacts(version)
        self._write_active_version(version)
        logger.info(f"Switched to model version: v{version}")
        return {"message": f"Switched to version v{version}", "current_version": version}

    def delete_version(self, version: int) -> Dict:
        """
        Удаляет версию модели с диска.

        :param version: Номер версии.
        """
        version_dir = self.model_dir / f"v{version}"
        if not version_dir.exists():
            raise ValueError(f"Version {version} does not exist")
        if version == self.current_version:
            versions = [
                int(v.name[1:])
                for v in self.model_dir.glob("v*")
                if v.is_dir() and v.name[1:].isdigit() and int(v.name[1:]) != version
            ]
            if not versions:
                raise ValueError("Cannot delete active version without alternatives")
            new_active = max(versions)
            self._load_artifacts(new_active)
            self._write_active_version(new_active)
        shutil.rmtree(version_dir)
        logger.info(f"Deleted model version: v{version}")
        return {"message": f"Version v{version} deleted", "current_version": self.current_version}

    def get_progress(self) -> Dict:
        """Возвращает персистентный статус обучения."""
        return self.status_store.get()

    def get_status(self) -> Dict:
        """Статус модели с метриками из metrics.json и кратким human-слоем."""
        versions = []
        for v in self.model_dir.glob("v*"):
            if not (v.is_dir() and v.name[1:].isdigit()):
                continue
            ver = int(v.name[1:])
            metrics = load_metrics(self.model_dir, ver) or {}
            holdout = metrics.get("holdout") or {}
            hit_first = holdout.get("hit_first_api")
            if hit_first is None:
                hit_first = holdout.get("accuracy_top1")
            hit_top2 = holdout.get("hit_top2_api")
            if hit_top2 is None:
                hit_top2 = holdout.get("accuracy_top2")
            versions.append(
                {
                    "version": ver,
                    "trained_at": metrics.get("trained_at", "unknown"),
                    "records_used": metrics.get("records_used", 0),
                    "hit_first_pct": round(float(hit_first) * 100, 1) if hit_first is not None else None,
                    "hit_top2_pct": round(float(hit_top2) * 100, 1) if hit_top2 is not None else None,
                    "hit_top3_pct": (
                        round(float(holdout["accuracy_top3"]) * 100, 1)
                        if holdout.get("accuracy_top3") is not None
                        else None
                    ),
                }
            )
        versions.sort(key=lambda x: x["version"], reverse=True)
        active = next((x for x in versions if x["version"] == self.current_version), None)
        plain = None
        if active and active.get("hit_first_pct") is not None:
            plain = (
                f"Активна v{self.current_version}: примерно {int(round(active['hit_first_pct']))} "
                f"из 100 — верный размер сразу"
            )
        return {
            "trained": bool(versions),
            "current_version": self.current_version,
            "versions": versions,
            "plain": plain,
        }

    def get_models_report(self) -> Dict:
        """Человекочитаемый отчёт по всем версиям."""
        return build_models_report(self.model_dir, self.current_version)

    def get_version_report(self, version: int) -> Dict:
        """Человекочитаемый отчёт по одной версии."""
        metrics = load_metrics(self.model_dir, version)
        if not metrics:
            raise ValueError(f"No metrics for version v{version}")
        prev = load_metrics(self.model_dir, version - 1) if version > 1 else None
        return build_version_report(version, metrics, self.current_version, prev)

    def compare_model_versions(self, a: int, b: int) -> Dict:
        """Сравнение двух версий для не-ML пользователя."""
        return compare_versions(self.model_dir, a, b)

    def _fast_holdout_metrics(
        self,
        X_train: pd.DataFrame,
        X_val: pd.DataFrame,
        y_train: np.ndarray,
        y_val: np.ndarray,
        meta_val: Optional[pd.DataFrame] = None,
    ) -> Dict:
        """
        Быстрая оценка метрик на holdout-выборке (вместо 5-фолдовой CV).
        """
        train_logger.info("Оценка метрик на holdout (вместо CV)...")
        model = self._make_xgb(n_classes=len(self.label_encoder.classes_), early_stopping_rounds=30)
        sample_weight = self._sqrt_balanced_weights(y_train)
        model.fit(
            X_train,
            y_train,
            sample_weight=sample_weight,
            eval_set=[(X_val, y_val)],
            verbose=False,
        )
        proba = model.predict_proba(X_val)
        y_pred = np.argmax(proba, axis=1)
        acc1 = accuracy_score(y_val, y_pred)
        acc2 = _topk_accuracy(y_val, proba, k=2)
        acc3 = _topk_accuracy(y_val, proba, k=3)
        p_mac, r_mac, f1_mac, _ = precision_recall_fscore_support(y_val, y_pred, average="macro", zero_division=0)
        p_w, r_w, f1_w, _ = precision_recall_fscore_support(y_val, y_pred, average="weighted", zero_division=0)
        try:
            ll = float(log_loss(y_val, proba, labels=list(range(len(self.label_encoder.classes_)))))
        except Exception:
            ll = None
        per_class = _per_class_metrics(
            y_val,
            y_pred,
            labels=list(range(len(self.label_encoder.classes_))),
            label_names=list(self.label_encoder.classes_),
        )
        result = {
            "n_splits": 1,
            "accuracy_top1_mean": float(acc1),
            "accuracy_top1_std": 0.0,
            "accuracy_top2_mean": float(acc2),
            "accuracy_top3_mean": float(acc3),
            "accuracy_top3_std": 0.0,
            "log_loss_mean": ll,
            "precision_macro_mean": float(p_mac),
            "recall_macro_mean": float(r_mac),
            "f1_macro_mean": float(f1_mac),
            "precision_weighted_mean": float(p_w),
            "recall_weighted_mean": float(r_w),
            "f1_weighted_mean": float(f1_w),
            "per_class": per_class,
        }
        if meta_val is not None and len(meta_val) > 0:
            prev_map = self.class_index_map
            prev_norm = self.class_index_map_norm
            self.class_index_map = {str(name): idx for idx, name in enumerate(self.label_encoder.classes_)}
            self.class_index_map_norm = {
                _normalize_size_label(str(name)): idx for idx, name in enumerate(self.label_encoder.classes_)
            }
            product_metrics = self._compute_product_api_metrics(proba, meta_val)
            result.update(product_metrics)
            self.class_index_map = prev_map
            self.class_index_map_norm = prev_norm
        return result

    def get_metrics(self, version: Optional[int] = None) -> Dict:
        """Вернуть метрики для версии (или текущей)."""
        ver = version if version is not None else self.current_version
        version_dir = self.model_dir / f'v{ver}'
        metrics_path = version_dir / 'metrics.json'
        if not metrics_path.exists():
            raise ValueError(f"No metrics for version v{ver}")
        with open(metrics_path, 'r') as f:
            return json.load(f)