Files
test/BotBoard/app/border.py
T

191 lines
5.9 KiB
Python

from __future__ import annotations
import re
from dataclasses import dataclass
from datetime import datetime
import httpx
from bs4 import BeautifulSoup
POLAND_CHECKPOINTS = (
"Брузги",
"Берестовица",
"Песчатка",
"Козловичи",
"Брест",
"Домачево",
)
SECTION_NAMES = ("Легковые", "Грузовые", "Автобусы")
@dataclass(frozen=True)
class QueueInfo:
checkpoint: str
queue_text: str
queue_value: int | None
electronic_queue: bool
@dataclass(frozen=True)
class BorderSection:
name: str
checkpoints: tuple[QueueInfo, ...]
@dataclass(frozen=True)
class BorderSnapshot:
updated_at: str | None
fetched_at: datetime
sections: tuple[BorderSection, ...]
source_url: str
class BorderParseError(RuntimeError):
pass
async def fetch_border_snapshot(url: str, timeout: float = 20.0) -> BorderSnapshot:
async with httpx.AsyncClient(timeout=timeout, follow_redirects=True) as client:
response = await client.get(url)
response.raise_for_status()
return parse_border_page(response.text, url)
def parse_border_page(html: str, source_url: str) -> BorderSnapshot:
soup = BeautifulSoup(html, "html.parser")
for tag in soup(["script", "style", "noscript"]):
tag.decompose()
lines = [line.strip() for line in soup.get_text("\n").splitlines() if line.strip()]
updated_at = _extract_updated_at(lines)
poland_blocks = _extract_poland_blocks(lines)
if not poland_blocks:
raise BorderParseError("Не удалось найти блок очередей по Польше на странице ГПК")
sections = tuple(
BorderSection(SECTION_NAMES[index] if index < len(SECTION_NAMES) else f"Таблица {index + 1}", block)
for index, block in enumerate(poland_blocks[:3])
)
return BorderSnapshot(
updated_at=updated_at,
fetched_at=datetime.now().astimezone(),
sections=sections,
source_url=source_url,
)
def format_snapshot(snapshot: BorderSnapshot) -> str:
updated = snapshot.updated_at or "не указано"
parts = [f"Очереди на границе Беларуси с Польшей", f"Обновлено ГПК: {updated}"]
for section in snapshot.sections:
parts.append("")
parts.append(f"{section.name}:")
for item in section.checkpoints:
suffix = " ЭО" if item.electronic_queue else ""
parts.append(f"- {item.checkpoint}: {item.queue_text}{suffix}")
parts.append("")
parts.append("ЭО - электронная очередь. '-' означает, что значение не опубликовано или пункт не работает.")
return "\n".join(parts)
def format_best_options(snapshot: BorderSnapshot) -> str:
lines = ["Самые свободные варианты по текущим данным:"]
for section in snapshot.sections:
numeric = [item for item in section.checkpoints if item.queue_value is not None]
if not numeric:
lines.append(f"{section.name}: нет числовых данных")
continue
best_value = min(item.queue_value for item in numeric)
best_names = ", ".join(item.checkpoint for item in numeric if item.queue_value == best_value)
lines.append(f"{section.name}: {best_names} ({best_value})")
lines.append("")
lines.append("Проверяй официальную страницу перед выездом: ситуация может измениться за минуты.")
return "\n".join(lines)
def snapshot_context(snapshot: BorderSnapshot) -> str:
return format_snapshot(snapshot)
def _extract_updated_at(lines: list[str]) -> str | None:
date_pattern = re.compile(r"\b\d{2}\.\d{2}\.\d{4}\s+\d{2}:\d{2}\b")
for line in lines:
match = date_pattern.search(line)
if match:
return match.group(0)
return None
def _extract_poland_blocks(lines: list[str]) -> list[tuple[QueueInfo, ...]]:
blocks: list[tuple[QueueInfo, ...]] = []
for index, line in enumerate(lines):
if line != "Польша":
continue
block_lines: list[str] = []
for next_line in lines[index + 1 :]:
if next_line == "Украина":
break
block_lines.append(next_line)
parsed = _parse_checkpoint_block(block_lines)
if parsed:
blocks.append(tuple(parsed))
return blocks
def _parse_checkpoint_block(lines: list[str]) -> list[QueueInfo]:
result: list[QueueInfo] = []
for index, line in enumerate(lines):
if line not in POLAND_CHECKPOINTS:
continue
electronic = False
value_line = None
cursor = index + 1
while cursor < len(lines):
candidate = lines[cursor]
if candidate in POLAND_CHECKPOINTS:
break
if "Электронная очередь" in candidate:
electronic = True
elif _looks_like_queue_value(candidate):
value_line = candidate
break
cursor += 1
if value_line is None:
continue
result.append(
QueueInfo(
checkpoint=line,
queue_text=_clean_queue_text(value_line),
queue_value=_queue_value(value_line),
electronic_queue=electronic,
)
)
return result
def _looks_like_queue_value(value: str) -> bool:
return bool(re.fullmatch(r"\(?\d+\)?\*?|[-—]", value.strip()))
def _clean_queue_text(value: str) -> str:
cleaned = value.strip().replace("*", "").strip()
if cleaned.startswith("(") and cleaned.endswith(")"):
cleaned = cleaned[1:-1]
return cleaned
def _queue_value(value: str) -> int | None:
match = re.search(r"\d+", value)
if not match:
return None
return int(match.group(0))