mirror of
https://gitverse.ru/kpa39l/md2vk.git
synced 2026-09-29 18:05:04 +00:00
422 lines
14 KiB
Python
422 lines
14 KiB
Python
"""Конвертер Markdown → VK format_data.
|
|
|
|
Поддерживает: **жирный**, *курсив*, `код`, [ссылки], #заголовки, > цитаты, ```блоки кода```, ---.
|
|
Разбивает длинные тексты на чанки (VK лимит ~4096 символов).
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
from dataclasses import dataclass, field
|
|
from typing import Optional
|
|
|
|
|
|
@dataclass
|
|
class FormatItem:
|
|
type: str # bold | italic | link | inline_code
|
|
offset: int
|
|
length: int
|
|
url: Optional[str] = None
|
|
|
|
|
|
@dataclass
|
|
class Chunk:
|
|
text: str
|
|
items: list[FormatItem] = field(default_factory=list)
|
|
|
|
|
|
# ─── AST-узлы для промежуточного представления ───────────────────────────
|
|
|
|
@dataclass
|
|
class TextNode:
|
|
text: str
|
|
|
|
|
|
@dataclass
|
|
class BoldNode:
|
|
children: list = field(default_factory=list)
|
|
|
|
|
|
@dataclass
|
|
class ItalicNode:
|
|
children: list = field(default_factory=list)
|
|
|
|
|
|
@dataclass
|
|
class BoldItalicNode:
|
|
children: list = field(default_factory=list)
|
|
|
|
|
|
@dataclass
|
|
class CodeNode:
|
|
text: str
|
|
|
|
|
|
@dataclass
|
|
class LinkNode:
|
|
text: str
|
|
url: str
|
|
|
|
|
|
@dataclass
|
|
class HeaderNode:
|
|
level: int
|
|
text: str
|
|
|
|
|
|
@dataclass
|
|
class BlockquoteNode:
|
|
text: str
|
|
|
|
|
|
@dataclass
|
|
class CodeBlockNode:
|
|
text: str
|
|
|
|
|
|
@dataclass
|
|
class HrNode:
|
|
pass
|
|
|
|
|
|
@dataclass
|
|
class ParagraphNode:
|
|
children: list = field(default_factory=list)
|
|
|
|
|
|
@dataclass
|
|
class DocumentNode:
|
|
children: list = field(default_factory=list)
|
|
|
|
|
|
# ─── Парсер Markdown (блочный + строчный) ─────────────────────────────────
|
|
|
|
|
|
def parse_block(text: str) -> list:
|
|
"""Разбивает текст на блочные элементы."""
|
|
lines = text.split("\n")
|
|
blocks = []
|
|
i = 0
|
|
while i < len(lines):
|
|
line = lines[i]
|
|
|
|
# Горизонтальная линия
|
|
if re.match(r"^-{3,}$", line.strip()):
|
|
blocks.append(HrNode())
|
|
i += 1
|
|
continue
|
|
|
|
# Заголовок
|
|
hm = re.match(r"^(#{1,6})\s+(.+)$", line)
|
|
if hm:
|
|
blocks.append(HeaderNode(level=len(hm.group(1)), text=hm.group(2)))
|
|
i += 1
|
|
continue
|
|
|
|
# Цитата
|
|
if line.startswith("> "):
|
|
quote_lines = []
|
|
while i < len(lines) and lines[i].startswith("> "):
|
|
quote_lines.append(lines[i][2:])
|
|
i += 1
|
|
blocks.append(BlockquoteNode(text="\n".join(quote_lines)))
|
|
continue
|
|
|
|
# Блок кода
|
|
if line.startswith("```"):
|
|
code_lines = []
|
|
i += 1
|
|
while i < len(lines) and not lines[i].startswith("```"):
|
|
code_lines.append(lines[i])
|
|
i += 1
|
|
i += 1 # пропускаем закрывающие ```
|
|
blocks.append(CodeBlockNode(text="\n".join(code_lines)))
|
|
continue
|
|
|
|
# Пустая строка — разделитель параграфов
|
|
if line.strip() == "":
|
|
i += 1
|
|
continue
|
|
|
|
# Обычный параграф
|
|
para_lines = []
|
|
while i < len(lines) and lines[i].strip() != "" and not lines[i].startswith("```") and not re.match(r"^-{3,}$", lines[i].strip()):
|
|
# Проверка на заголовок внутри — не разрываем параграф
|
|
if re.match(r"^#{1,6}\s+", lines[i]) and len(para_lines) > 0:
|
|
break
|
|
para_lines.append(lines[i])
|
|
i += 1
|
|
blocks.append(ParagraphNode(children=parse_inline("\n".join(para_lines))))
|
|
# Не инкрементим i, т.к. цикл while уже продвинул
|
|
|
|
return blocks
|
|
|
|
|
|
def parse_inline(text: str) -> list:
|
|
"""Парсит строчные элементы: **жирный**, *курсив*, `код`, [ссылки], ***жирный+курсив***."""
|
|
result = []
|
|
pos = 0
|
|
while pos < len(text):
|
|
# ***жирный+курсив***
|
|
m = re.match(r"\*\*\*(.+?)\*\*\*", text[pos:])
|
|
if m:
|
|
result.append(BoldItalicNode(children=[TextNode(text=m.group(1))]))
|
|
pos += len(m.group(0))
|
|
continue
|
|
|
|
# **жирный**
|
|
m = re.match(r"\*\*(.+?)\*\*", text[pos:])
|
|
if m:
|
|
result.append(BoldNode(children=[TextNode(text=m.group(1))]))
|
|
pos += len(m.group(0))
|
|
continue
|
|
|
|
# __жирный__
|
|
m = re.match(r"__(.+?)__", text[pos:])
|
|
if m:
|
|
result.append(BoldNode(children=[TextNode(text=m.group(1))]))
|
|
pos += len(m.group(0))
|
|
continue
|
|
|
|
# *курсив*
|
|
m = re.match(r"\*(.+?)\*", text[pos:])
|
|
if m:
|
|
# Убедимся, что это не **
|
|
if not text[pos:].startswith("**"):
|
|
result.append(ItalicNode(children=[TextNode(text=m.group(1))]))
|
|
pos += len(m.group(0))
|
|
continue
|
|
|
|
# _курсив_
|
|
m = re.match(r"_(.+?)_", text[pos:])
|
|
if m:
|
|
if not text[pos:].startswith("__"):
|
|
result.append(ItalicNode(children=[TextNode(text=m.group(1))]))
|
|
pos += len(m.group(0))
|
|
continue
|
|
|
|
# `код`
|
|
m = re.match(r"`([^`]+)`", text[pos:])
|
|
if m:
|
|
result.append(CodeNode(text=m.group(1)))
|
|
pos += len(m.group(0))
|
|
continue
|
|
|
|
# [ссылка](url)
|
|
m = re.match(r"\[([^\]]+)\]\(([^)]+)\)", text[pos:])
|
|
if m:
|
|
result.append(LinkNode(text=m.group(1), url=m.group(2)))
|
|
pos += len(m.group(0))
|
|
continue
|
|
|
|
# Обычный текст
|
|
m = re.match(r"[^*_`\[<]+", text[pos:])
|
|
if m:
|
|
result.append(TextNode(text=m.group(0)))
|
|
pos += len(m.group(0))
|
|
continue
|
|
|
|
# Одиночный символ (если не подошло ни одно правило)
|
|
result.append(TextNode(text=text[pos]))
|
|
pos += 1
|
|
|
|
return result
|
|
|
|
|
|
# ─── Генерация VK-формата ─────────────────────────────────────────────────
|
|
|
|
|
|
def _render_node(node, plain_text: list[str], format_items: list[FormatItem], base_offset: int) -> int:
|
|
"""Рендерит AST-узел в plain_text и format_items. Возвращает новый offset."""
|
|
if isinstance(node, TextNode):
|
|
plain_text.append(node.text)
|
|
return base_offset + len(node.text)
|
|
|
|
elif isinstance(node, BoldNode):
|
|
inner_start = base_offset
|
|
offset = inner_start
|
|
for child in node.children:
|
|
offset = _render_node(child, plain_text, format_items, offset)
|
|
if offset > inner_start:
|
|
format_items.append(FormatItem(type="bold", offset=inner_start, length=offset - inner_start))
|
|
return offset
|
|
|
|
elif isinstance(node, ItalicNode):
|
|
inner_start = base_offset
|
|
offset = inner_start
|
|
for child in node.children:
|
|
offset = _render_node(child, plain_text, format_items, offset)
|
|
if offset > inner_start:
|
|
format_items.append(FormatItem(type="italic", offset=inner_start, length=offset - inner_start))
|
|
return offset
|
|
|
|
elif isinstance(node, BoldItalicNode):
|
|
inner_start = base_offset
|
|
offset = inner_start
|
|
for child in node.children:
|
|
offset = _render_node(child, plain_text, format_items, offset)
|
|
if offset > inner_start:
|
|
format_items.append(FormatItem(type="bold", offset=inner_start, length=offset - inner_start))
|
|
format_items.append(FormatItem(type="italic", offset=inner_start, length=offset - inner_start))
|
|
return offset
|
|
|
|
elif isinstance(node, CodeNode):
|
|
plain_text.append(node.text)
|
|
length = len(node.text)
|
|
format_items.append(FormatItem(type="inline_code", offset=base_offset, length=length))
|
|
return base_offset + length
|
|
|
|
elif isinstance(node, LinkNode):
|
|
offset = base_offset
|
|
for child in parse_inline(node.text):
|
|
offset = _render_node(child, plain_text, format_items, offset)
|
|
length = offset - base_offset
|
|
format_items.append(FormatItem(type="link", offset=base_offset, length=length, url=node.url))
|
|
return offset
|
|
|
|
elif isinstance(node, HeaderNode):
|
|
# Заголовки → жирный + uppercase
|
|
text = node.text.upper()
|
|
plain_text.append(text)
|
|
format_items.append(FormatItem(type="bold", offset=base_offset, length=len(text)))
|
|
return base_offset + len(text)
|
|
|
|
elif isinstance(node, BlockquoteNode):
|
|
# Цитата → italic
|
|
text = node.text
|
|
plain_text.append(text)
|
|
format_items.append(FormatItem(type="italic", offset=base_offset, length=len(text)))
|
|
return base_offset + len(text)
|
|
|
|
elif isinstance(node, CodeBlockNode):
|
|
text = node.text
|
|
plain_text.append(text)
|
|
return base_offset + len(text)
|
|
|
|
elif isinstance(node, HrNode):
|
|
plain_text.append("───")
|
|
return base_offset + 3
|
|
|
|
elif isinstance(node, ParagraphNode):
|
|
offset = base_offset
|
|
for child in node.children:
|
|
offset = _render_node(child, plain_text, format_items, offset)
|
|
return offset
|
|
|
|
return base_offset
|
|
|
|
|
|
def render_document(blocks: list) -> tuple[str, list[FormatItem]]:
|
|
"""Рендерит список блоков в плоский текст + format_items."""
|
|
plain_text: list[str] = []
|
|
format_items: list[FormatItem] = []
|
|
offset = 0
|
|
|
|
for i, block in enumerate(blocks):
|
|
if i > 0:
|
|
plain_text.append("\n\n")
|
|
offset += 2
|
|
offset = _render_node(block, plain_text, format_items, offset)
|
|
|
|
return "".join(plain_text), format_items
|
|
|
|
|
|
# ─── Разбиение на чанки ────────────────────────────────────────────────────
|
|
|
|
|
|
def _vk_char_len(text: str) -> int:
|
|
"""VK считает @ за 2 символа. Учитываем это при подсчёте длины."""
|
|
count = 0
|
|
for ch in text:
|
|
count += 2 if ch == "@" else 1
|
|
return count
|
|
|
|
|
|
def _split_into_chunks(text: str, items: list[FormatItem], chunk_size: int = 4096) -> list[Chunk]:
|
|
"""Разбивает текст на чанки, корректируя format_items на границах."""
|
|
if not text:
|
|
return [Chunk(text="", items=[])]
|
|
|
|
# Определяем границы разбиения по \n\n (абзацы)
|
|
# Если текст влезает целиком — один чанк
|
|
if _vk_char_len(text) <= chunk_size:
|
|
return [Chunk(text=text, items=_adjust_items(items, 0, len(text)))]
|
|
|
|
chunks: list[Chunk] = []
|
|
start = 0
|
|
|
|
while start < len(text):
|
|
# Ищем границу: \n\n в пределах chunk_size
|
|
end = start + int(chunk_size * 0.9) # 90% от лимита — запас
|
|
if end >= len(text):
|
|
end = len(text)
|
|
|
|
# Ищем \n\n назад от end
|
|
split_pos = text.rfind("\n\n", start, end)
|
|
if split_pos == -1 or split_pos <= start:
|
|
# Если нет \n\n — ищем последний пробел
|
|
split_pos = text.rfind(" ", start, end)
|
|
if split_pos == -1 or split_pos <= start:
|
|
split_pos = end
|
|
|
|
chunk_text = text[start:split_pos].strip()
|
|
if chunk_text:
|
|
chunk_items = _adjust_items(items, start, split_pos)
|
|
chunks.append(Chunk(text=chunk_text, items=chunk_items))
|
|
|
|
start = split_pos + 1 # пропускаем разделитель
|
|
|
|
return chunks if chunks else [Chunk(text=text, items=[])]
|
|
|
|
|
|
def _adjust_items(items: list[FormatItem], start: int, end: int) -> list[FormatItem]:
|
|
"""Обрезает format_items для диапазона [start, end) и сдвигает offset."""
|
|
result = []
|
|
for item in items:
|
|
item_end = item.offset + item.length
|
|
# Проверяем пересечение
|
|
if item_end <= start or item.offset >= end:
|
|
continue
|
|
new_offset = max(item.offset, start) - start
|
|
new_length = min(item_end, end) - max(item.offset, start)
|
|
result.append(FormatItem(
|
|
type=item.type,
|
|
offset=new_offset,
|
|
length=new_length,
|
|
url=item.url,
|
|
))
|
|
return result
|
|
|
|
|
|
# ─── Публичный API ─────────────────────────────────────────────────────────
|
|
|
|
|
|
def markdown_to_vk(text: str, chunk_size: int = 4096) -> list[Chunk]:
|
|
"""Конвертирует Markdown в список чанков, готовых к отправке в VK API.
|
|
|
|
Каждый чанк содержит:
|
|
- text: plain text для поля message
|
|
- items: список FormatItem для format_data
|
|
|
|
VK принимает format_data через поле format_data в wall.post,
|
|
которое должно быть JSON-строкой вида:
|
|
{"version": 1, "items": [{"type": "bold", "offset": 0, "length": 5}]}
|
|
"""
|
|
if not text or not text.strip():
|
|
return [Chunk(text="", items=[])]
|
|
|
|
blocks = parse_block(text)
|
|
plain_text, items = render_document(blocks)
|
|
return _split_into_chunks(plain_text, items, chunk_size)
|
|
|
|
|
|
def format_data_json(items: list[FormatItem]) -> str:
|
|
"""Сериализует format_items в JSON для VK API."""
|
|
import json
|
|
vk_items = []
|
|
for item in items:
|
|
d = {"type": item.type, "offset": item.offset, "length": item.length}
|
|
if item.url:
|
|
d["url"] = item.url
|
|
vk_items.append(d)
|
|
return json.dumps({"version": 1, "items": vk_items}, ensure_ascii=False) |