"""Конвертер Markdown → VK format_data. Поддерживает: **жирный**, *курсив*, `код`, [ссылки], #заголовки, > цитаты, ```блоки кода```, ---. Разбивает длинные тексты на чанки (VK лимит ~4096 символов). """ from __future__ import annotations import re from dataclasses import dataclass, field from typing import Optional @dataclass class FormatItem: type: str # bold | italic | link | inline_code offset: int length: int url: Optional[str] = None @dataclass class Chunk: text: str items: list[FormatItem] = field(default_factory=list) # ─── AST-узлы для промежуточного представления ─────────────────────────── @dataclass class TextNode: text: str @dataclass class BoldNode: children: list = field(default_factory=list) @dataclass class ItalicNode: children: list = field(default_factory=list) @dataclass class BoldItalicNode: children: list = field(default_factory=list) @dataclass class CodeNode: text: str @dataclass class LinkNode: text: str url: str @dataclass class HeaderNode: level: int text: str @dataclass class BlockquoteNode: text: str @dataclass class CodeBlockNode: text: str @dataclass class HrNode: pass @dataclass class ParagraphNode: children: list = field(default_factory=list) @dataclass class DocumentNode: children: list = field(default_factory=list) # ─── Парсер Markdown (блочный + строчный) ───────────────────────────────── def parse_block(text: str) -> list: """Разбивает текст на блочные элементы.""" lines = text.split("\n") blocks = [] i = 0 while i < len(lines): line = lines[i] # Горизонтальная линия if re.match(r"^-{3,}$", line.strip()): blocks.append(HrNode()) i += 1 continue # Заголовок hm = re.match(r"^(#{1,6})\s+(.+)$", line) if hm: blocks.append(HeaderNode(level=len(hm.group(1)), text=hm.group(2))) i += 1 continue # Цитата if line.startswith("> "): quote_lines = [] while i < len(lines) and lines[i].startswith("> "): quote_lines.append(lines[i][2:]) i += 1 blocks.append(BlockquoteNode(text="\n".join(quote_lines))) continue # Блок кода if line.startswith("```"): code_lines = [] i += 1 while i < len(lines) and not lines[i].startswith("```"): code_lines.append(lines[i]) i += 1 i += 1 # пропускаем закрывающие ``` blocks.append(CodeBlockNode(text="\n".join(code_lines))) continue # Пустая строка — разделитель параграфов if line.strip() == "": i += 1 continue # Обычный параграф para_lines = [] while i < len(lines) and lines[i].strip() != "" and not lines[i].startswith("```") and not re.match(r"^-{3,}$", lines[i].strip()): # Проверка на заголовок внутри — не разрываем параграф if re.match(r"^#{1,6}\s+", lines[i]) and len(para_lines) > 0: break para_lines.append(lines[i]) i += 1 blocks.append(ParagraphNode(children=parse_inline("\n".join(para_lines)))) # Не инкрементим i, т.к. цикл while уже продвинул return blocks def parse_inline(text: str) -> list: """Парсит строчные элементы: **жирный**, *курсив*, `код`, [ссылки], ***жирный+курсив***.""" result = [] pos = 0 while pos < len(text): # ***жирный+курсив*** m = re.match(r"\*\*\*(.+?)\*\*\*", text[pos:]) if m: result.append(BoldItalicNode(children=[TextNode(text=m.group(1))])) pos += len(m.group(0)) continue # **жирный** m = re.match(r"\*\*(.+?)\*\*", text[pos:]) if m: result.append(BoldNode(children=[TextNode(text=m.group(1))])) pos += len(m.group(0)) continue # __жирный__ m = re.match(r"__(.+?)__", text[pos:]) if m: result.append(BoldNode(children=[TextNode(text=m.group(1))])) pos += len(m.group(0)) continue # *курсив* m = re.match(r"\*(.+?)\*", text[pos:]) if m: # Убедимся, что это не ** if not text[pos:].startswith("**"): result.append(ItalicNode(children=[TextNode(text=m.group(1))])) pos += len(m.group(0)) continue # _курсив_ m = re.match(r"_(.+?)_", text[pos:]) if m: if not text[pos:].startswith("__"): result.append(ItalicNode(children=[TextNode(text=m.group(1))])) pos += len(m.group(0)) continue # `код` m = re.match(r"`([^`]+)`", text[pos:]) if m: result.append(CodeNode(text=m.group(1))) pos += len(m.group(0)) continue # [ссылка](url) m = re.match(r"\[([^\]]+)\]\(([^)]+)\)", text[pos:]) if m: result.append(LinkNode(text=m.group(1), url=m.group(2))) pos += len(m.group(0)) continue # Обычный текст m = re.match(r"[^*_`\[<]+", text[pos:]) if m: result.append(TextNode(text=m.group(0))) pos += len(m.group(0)) continue # Одиночный символ (если не подошло ни одно правило) result.append(TextNode(text=text[pos])) pos += 1 return result # ─── Генерация VK-формата ───────────────────────────────────────────────── def _render_node(node, plain_text: list[str], format_items: list[FormatItem], base_offset: int) -> int: """Рендерит AST-узел в plain_text и format_items. Возвращает новый offset.""" if isinstance(node, TextNode): plain_text.append(node.text) return base_offset + len(node.text) elif isinstance(node, BoldNode): inner_start = base_offset offset = inner_start for child in node.children: offset = _render_node(child, plain_text, format_items, offset) if offset > inner_start: format_items.append(FormatItem(type="bold", offset=inner_start, length=offset - inner_start)) return offset elif isinstance(node, ItalicNode): inner_start = base_offset offset = inner_start for child in node.children: offset = _render_node(child, plain_text, format_items, offset) if offset > inner_start: format_items.append(FormatItem(type="italic", offset=inner_start, length=offset - inner_start)) return offset elif isinstance(node, BoldItalicNode): inner_start = base_offset offset = inner_start for child in node.children: offset = _render_node(child, plain_text, format_items, offset) if offset > inner_start: format_items.append(FormatItem(type="bold", offset=inner_start, length=offset - inner_start)) format_items.append(FormatItem(type="italic", offset=inner_start, length=offset - inner_start)) return offset elif isinstance(node, CodeNode): plain_text.append(node.text) length = len(node.text) format_items.append(FormatItem(type="inline_code", offset=base_offset, length=length)) return base_offset + length elif isinstance(node, LinkNode): offset = base_offset for child in parse_inline(node.text): offset = _render_node(child, plain_text, format_items, offset) length = offset - base_offset format_items.append(FormatItem(type="link", offset=base_offset, length=length, url=node.url)) return offset elif isinstance(node, HeaderNode): # Заголовки → жирный + uppercase text = node.text.upper() plain_text.append(text) format_items.append(FormatItem(type="bold", offset=base_offset, length=len(text))) return base_offset + len(text) elif isinstance(node, BlockquoteNode): # Цитата → italic text = node.text plain_text.append(text) format_items.append(FormatItem(type="italic", offset=base_offset, length=len(text))) return base_offset + len(text) elif isinstance(node, CodeBlockNode): text = node.text plain_text.append(text) return base_offset + len(text) elif isinstance(node, HrNode): plain_text.append("───") return base_offset + 3 elif isinstance(node, ParagraphNode): offset = base_offset for child in node.children: offset = _render_node(child, plain_text, format_items, offset) return offset return base_offset def render_document(blocks: list) -> tuple[str, list[FormatItem]]: """Рендерит список блоков в плоский текст + format_items.""" plain_text: list[str] = [] format_items: list[FormatItem] = [] offset = 0 for i, block in enumerate(blocks): if i > 0: plain_text.append("\n\n") offset += 2 offset = _render_node(block, plain_text, format_items, offset) return "".join(plain_text), format_items # ─── Разбиение на чанки ──────────────────────────────────────────────────── def _vk_char_len(text: str) -> int: """VK считает @ за 2 символа. Учитываем это при подсчёте длины.""" count = 0 for ch in text: count += 2 if ch == "@" else 1 return count def _split_into_chunks(text: str, items: list[FormatItem], chunk_size: int = 4096) -> list[Chunk]: """Разбивает текст на чанки, корректируя format_items на границах.""" if not text: return [Chunk(text="", items=[])] # Определяем границы разбиения по \n\n (абзацы) # Если текст влезает целиком — один чанк if _vk_char_len(text) <= chunk_size: return [Chunk(text=text, items=_adjust_items(items, 0, len(text)))] chunks: list[Chunk] = [] start = 0 while start < len(text): # Ищем границу: \n\n в пределах chunk_size end = start + int(chunk_size * 0.9) # 90% от лимита — запас if end >= len(text): end = len(text) # Ищем \n\n назад от end split_pos = text.rfind("\n\n", start, end) if split_pos == -1 or split_pos <= start: # Если нет \n\n — ищем последний пробел split_pos = text.rfind(" ", start, end) if split_pos == -1 or split_pos <= start: split_pos = end chunk_text = text[start:split_pos].strip() if chunk_text: chunk_items = _adjust_items(items, start, split_pos) chunks.append(Chunk(text=chunk_text, items=chunk_items)) start = split_pos + 1 # пропускаем разделитель return chunks if chunks else [Chunk(text=text, items=[])] def _adjust_items(items: list[FormatItem], start: int, end: int) -> list[FormatItem]: """Обрезает format_items для диапазона [start, end) и сдвигает offset.""" result = [] for item in items: item_end = item.offset + item.length # Проверяем пересечение if item_end <= start or item.offset >= end: continue new_offset = max(item.offset, start) - start new_length = min(item_end, end) - max(item.offset, start) result.append(FormatItem( type=item.type, offset=new_offset, length=new_length, url=item.url, )) return result # ─── Публичный API ───────────────────────────────────────────────────────── def markdown_to_vk(text: str, chunk_size: int = 4096) -> list[Chunk]: """Конвертирует Markdown в список чанков, готовых к отправке в VK API. Каждый чанк содержит: - text: plain text для поля message - items: список FormatItem для format_data VK принимает format_data через поле format_data в wall.post, которое должно быть JSON-строкой вида: {"version": 1, "items": [{"type": "bold", "offset": 0, "length": 5}]} """ if not text or not text.strip(): return [Chunk(text="", items=[])] blocks = parse_block(text) plain_text, items = render_document(blocks) return _split_into_chunks(plain_text, items, chunk_size) def format_data_json(items: list[FormatItem]) -> str: """Сериализует format_items в JSON для VK API.""" import json vk_items = [] for item in items: d = {"type": item.type, "offset": item.offset, "length": item.length} if item.url: d["url"] = item.url vk_items.append(d) return json.dumps({"version": 1, "items": vk_items}, ensure_ascii=False)