Baseline md2vk: docs, audit log, docker 8420, openspec, deploy

This commit is contained in:
estorozhenko
2026-09-18 22:05:01 +00:00
commit 09e960a3a9
39 changed files with 3716 additions and 0 deletions
View File
+422
View File
@@ -0,0 +1,422 @@
"""Конвертер Markdown → VK format_data.
Поддерживает: **жирный**, *курсив*, `код`, [ссылки], #заголовки, > цитаты, ```блоки кода```, ---.
Разбивает длинные тексты на чанки (VK лимит ~4096 символов).
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from typing import Optional
@dataclass
class FormatItem:
type: str # bold | italic | link | inline_code
offset: int
length: int
url: Optional[str] = None
@dataclass
class Chunk:
text: str
items: list[FormatItem] = field(default_factory=list)
# ─── AST-узлы для промежуточного представления ───────────────────────────
@dataclass
class TextNode:
text: str
@dataclass
class BoldNode:
children: list = field(default_factory=list)
@dataclass
class ItalicNode:
children: list = field(default_factory=list)
@dataclass
class BoldItalicNode:
children: list = field(default_factory=list)
@dataclass
class CodeNode:
text: str
@dataclass
class LinkNode:
text: str
url: str
@dataclass
class HeaderNode:
level: int
text: str
@dataclass
class BlockquoteNode:
text: str
@dataclass
class CodeBlockNode:
text: str
@dataclass
class HrNode:
pass
@dataclass
class ParagraphNode:
children: list = field(default_factory=list)
@dataclass
class DocumentNode:
children: list = field(default_factory=list)
# ─── Парсер Markdown (блочный + строчный) ─────────────────────────────────
def parse_block(text: str) -> list:
"""Разбивает текст на блочные элементы."""
lines = text.split("\n")
blocks = []
i = 0
while i < len(lines):
line = lines[i]
# Горизонтальная линия
if re.match(r"^-{3,}$", line.strip()):
blocks.append(HrNode())
i += 1
continue
# Заголовок
hm = re.match(r"^(#{1,6})\s+(.+)$", line)
if hm:
blocks.append(HeaderNode(level=len(hm.group(1)), text=hm.group(2)))
i += 1
continue
# Цитата
if line.startswith("> "):
quote_lines = []
while i < len(lines) and lines[i].startswith("> "):
quote_lines.append(lines[i][2:])
i += 1
blocks.append(BlockquoteNode(text="\n".join(quote_lines)))
continue
# Блок кода
if line.startswith("```"):
code_lines = []
i += 1
while i < len(lines) and not lines[i].startswith("```"):
code_lines.append(lines[i])
i += 1
i += 1 # пропускаем закрывающие ```
blocks.append(CodeBlockNode(text="\n".join(code_lines)))
continue
# Пустая строка — разделитель параграфов
if line.strip() == "":
i += 1
continue
# Обычный параграф
para_lines = []
while i < len(lines) and lines[i].strip() != "" and not lines[i].startswith("```") and not re.match(r"^-{3,}$", lines[i].strip()):
# Проверка на заголовок внутри — не разрываем параграф
if re.match(r"^#{1,6}\s+", lines[i]) and len(para_lines) > 0:
break
para_lines.append(lines[i])
i += 1
blocks.append(ParagraphNode(children=parse_inline("\n".join(para_lines))))
# Не инкрементим i, т.к. цикл while уже продвинул
return blocks
def parse_inline(text: str) -> list:
"""Парсит строчные элементы: **жирный**, *курсив*, `код`, [ссылки], ***жирный+курсив***."""
result = []
pos = 0
while pos < len(text):
# ***жирный+курсив***
m = re.match(r"\*\*\*(.+?)\*\*\*", text[pos:])
if m:
result.append(BoldItalicNode(children=[TextNode(text=m.group(1))]))
pos += len(m.group(0))
continue
# **жирный**
m = re.match(r"\*\*(.+?)\*\*", text[pos:])
if m:
result.append(BoldNode(children=[TextNode(text=m.group(1))]))
pos += len(m.group(0))
continue
# __жирный__
m = re.match(r"__(.+?)__", text[pos:])
if m:
result.append(BoldNode(children=[TextNode(text=m.group(1))]))
pos += len(m.group(0))
continue
# *курсив*
m = re.match(r"\*(.+?)\*", text[pos:])
if m:
# Убедимся, что это не **
if not text[pos:].startswith("**"):
result.append(ItalicNode(children=[TextNode(text=m.group(1))]))
pos += len(m.group(0))
continue
# _курсив_
m = re.match(r"_(.+?)_", text[pos:])
if m:
if not text[pos:].startswith("__"):
result.append(ItalicNode(children=[TextNode(text=m.group(1))]))
pos += len(m.group(0))
continue
# `код`
m = re.match(r"`([^`]+)`", text[pos:])
if m:
result.append(CodeNode(text=m.group(1)))
pos += len(m.group(0))
continue
# [ссылка](url)
m = re.match(r"\[([^\]]+)\]\(([^)]+)\)", text[pos:])
if m:
result.append(LinkNode(text=m.group(1), url=m.group(2)))
pos += len(m.group(0))
continue
# Обычный текст
m = re.match(r"[^*_`\[<]+", text[pos:])
if m:
result.append(TextNode(text=m.group(0)))
pos += len(m.group(0))
continue
# Одиночный символ (если не подошло ни одно правило)
result.append(TextNode(text=text[pos]))
pos += 1
return result
# ─── Генерация VK-формата ─────────────────────────────────────────────────
def _render_node(node, plain_text: list[str], format_items: list[FormatItem], base_offset: int) -> int:
"""Рендерит AST-узел в plain_text и format_items. Возвращает новый offset."""
if isinstance(node, TextNode):
plain_text.append(node.text)
return base_offset + len(node.text)
elif isinstance(node, BoldNode):
inner_start = base_offset
offset = inner_start
for child in node.children:
offset = _render_node(child, plain_text, format_items, offset)
if offset > inner_start:
format_items.append(FormatItem(type="bold", offset=inner_start, length=offset - inner_start))
return offset
elif isinstance(node, ItalicNode):
inner_start = base_offset
offset = inner_start
for child in node.children:
offset = _render_node(child, plain_text, format_items, offset)
if offset > inner_start:
format_items.append(FormatItem(type="italic", offset=inner_start, length=offset - inner_start))
return offset
elif isinstance(node, BoldItalicNode):
inner_start = base_offset
offset = inner_start
for child in node.children:
offset = _render_node(child, plain_text, format_items, offset)
if offset > inner_start:
format_items.append(FormatItem(type="bold", offset=inner_start, length=offset - inner_start))
format_items.append(FormatItem(type="italic", offset=inner_start, length=offset - inner_start))
return offset
elif isinstance(node, CodeNode):
plain_text.append(node.text)
length = len(node.text)
format_items.append(FormatItem(type="inline_code", offset=base_offset, length=length))
return base_offset + length
elif isinstance(node, LinkNode):
offset = base_offset
for child in parse_inline(node.text):
offset = _render_node(child, plain_text, format_items, offset)
length = offset - base_offset
format_items.append(FormatItem(type="link", offset=base_offset, length=length, url=node.url))
return offset
elif isinstance(node, HeaderNode):
# Заголовки → жирный + uppercase
text = node.text.upper()
plain_text.append(text)
format_items.append(FormatItem(type="bold", offset=base_offset, length=len(text)))
return base_offset + len(text)
elif isinstance(node, BlockquoteNode):
# Цитата → italic
text = node.text
plain_text.append(text)
format_items.append(FormatItem(type="italic", offset=base_offset, length=len(text)))
return base_offset + len(text)
elif isinstance(node, CodeBlockNode):
text = node.text
plain_text.append(text)
return base_offset + len(text)
elif isinstance(node, HrNode):
plain_text.append("───")
return base_offset + 3
elif isinstance(node, ParagraphNode):
offset = base_offset
for child in node.children:
offset = _render_node(child, plain_text, format_items, offset)
return offset
return base_offset
def render_document(blocks: list) -> tuple[str, list[FormatItem]]:
"""Рендерит список блоков в плоский текст + format_items."""
plain_text: list[str] = []
format_items: list[FormatItem] = []
offset = 0
for i, block in enumerate(blocks):
if i > 0:
plain_text.append("\n\n")
offset += 2
offset = _render_node(block, plain_text, format_items, offset)
return "".join(plain_text), format_items
# ─── Разбиение на чанки ────────────────────────────────────────────────────
def _vk_char_len(text: str) -> int:
"""VK считает @ за 2 символа. Учитываем это при подсчёте длины."""
count = 0
for ch in text:
count += 2 if ch == "@" else 1
return count
def _split_into_chunks(text: str, items: list[FormatItem], chunk_size: int = 4096) -> list[Chunk]:
"""Разбивает текст на чанки, корректируя format_items на границах."""
if not text:
return [Chunk(text="", items=[])]
# Определяем границы разбиения по \n\n (абзацы)
# Если текст влезает целиком — один чанк
if _vk_char_len(text) <= chunk_size:
return [Chunk(text=text, items=_adjust_items(items, 0, len(text)))]
chunks: list[Chunk] = []
start = 0
while start < len(text):
# Ищем границу: \n\n в пределах chunk_size
end = start + int(chunk_size * 0.9) # 90% от лимита — запас
if end >= len(text):
end = len(text)
# Ищем \n\n назад от end
split_pos = text.rfind("\n\n", start, end)
if split_pos == -1 or split_pos <= start:
# Если нет \n\n — ищем последний пробел
split_pos = text.rfind(" ", start, end)
if split_pos == -1 or split_pos <= start:
split_pos = end
chunk_text = text[start:split_pos].strip()
if chunk_text:
chunk_items = _adjust_items(items, start, split_pos)
chunks.append(Chunk(text=chunk_text, items=chunk_items))
start = split_pos + 1 # пропускаем разделитель
return chunks if chunks else [Chunk(text=text, items=[])]
def _adjust_items(items: list[FormatItem], start: int, end: int) -> list[FormatItem]:
"""Обрезает format_items для диапазона [start, end) и сдвигает offset."""
result = []
for item in items:
item_end = item.offset + item.length
# Проверяем пересечение
if item_end <= start or item.offset >= end:
continue
new_offset = max(item.offset, start) - start
new_length = min(item_end, end) - max(item.offset, start)
result.append(FormatItem(
type=item.type,
offset=new_offset,
length=new_length,
url=item.url,
))
return result
# ─── Публичный API ─────────────────────────────────────────────────────────
def markdown_to_vk(text: str, chunk_size: int = 4096) -> list[Chunk]:
"""Конвертирует Markdown в список чанков, готовых к отправке в VK API.
Каждый чанк содержит:
- text: plain text для поля message
- items: список FormatItem для format_data
VK принимает format_data через поле format_data в wall.post,
которое должно быть JSON-строкой вида:
{"version": 1, "items": [{"type": "bold", "offset": 0, "length": 5}]}
"""
if not text or not text.strip():
return [Chunk(text="", items=[])]
blocks = parse_block(text)
plain_text, items = render_document(blocks)
return _split_into_chunks(plain_text, items, chunk_size)
def format_data_json(items: list[FormatItem]) -> str:
"""Сериализует format_items в JSON для VK API."""
import json
vk_items = []
for item in items:
d = {"type": item.type, "offset": item.offset, "length": item.length}
if item.url:
d["url"] = item.url
vk_items.append(d)
return json.dumps({"version": 1, "items": vk_items}, ensure_ascii=False)