# VESTI — Банк статей: markdown-бандлы + медиа. # Спека: openspec/changes/tg-crawler-publisher-prototype/specs/news-store/spec.md # Путь: bundles/<направление>//.md # Медиа: bundles/<направление>//media// # frontmatter: title, date, direction, source, url, views, reactions, interest, status import json import os import re import shutil import sqlite3 import unicodedata from datetime import datetime from pathlib import Path import yaml BASE = Path(__file__).resolve().parent.parent DB_PATH = BASE / "db" / "vesti.db" BUNDLES_DIR = BASE / "bundles" MEDIA_SRC = BASE / "media" / "media" # каталог, куда краулер складывает файлы DIRECTION_LABELS = { "linux": "Linux", "tech": "Технологии", "politics": "Политика", "games": "Игры", "electronics": "Электроника", "llm": "LLM / AI", } def slugify(text: str, maxlen: int = 60) -> str: """Транслитерация + slug. Убирает URL/служебное. 'Omarchy — Arch Linux...' -> 'omarchy-arch-linux'.""" text = (text or "").strip() # убираем markdown/эмодзи/служебное text = re.sub(r"[*_`#\[\]()>|]", "", text) # убираем URL (http, т.me, ссылки) — из них берём только куски без протокола text = re.sub(r"https?://\S+|t\.me/\S+|ya\.cc/\S+", "", text) text = re.sub(r"\s+", " ", text).strip() # обрезаем до 6 слов — иначе слаг раздувается в белиберду words_ = text.split() if len(words_) > 6: text = " ".join(words_[:6]) # транслитерация кириллицы (правильные буквосочетания) tr_map = { "а": "a", "б": "b", "в": "v", "г": "g", "д": "d", "е": "e", "ё": "e", "ж": "zh", "з": "z", "и": "i", "й": "y", "к": "k", "л": "l", "м": "m", "н": "n", "о": "o", "п": "p", "р": "r", "с": "s", "т": "t", "у": "u", "ф": "f", "х": "h", "ц": "ts", "ч": "ch", "ш": "sh", "щ": "sch", "ъ": "", "ы": "y", "ь": "", "э": "e", "ю": "yu", "я": "ya", "А": "A", "Б": "B", "В": "V", "Г": "G", "Д": "D", "Е": "E", "Ё": "E", "Ж": "Zh", "З": "Z", "И": "I", "Й": "Y", "К": "K", "Л": "L", "М": "M", "Н": "N", "О": "O", "П": "P", "Р": "R", "С": "S", "Т": "T", "У": "U", "Ф": "F", "Х": "H", "Ц": "Ts", "Ч": "Ch", "Ш": "Sh", "Щ": "Sch", "Ъ": "", "Ы": "Y", "Ь": "", "Э": "E", "Ю": "Yu", "Я": "Ya", } text = "".join(tr_map.get(ch, ch) for ch in text) # нормализация юникода, оставляем латиницу/цифры/дефис text = unicodedata.normalize("NFKD", text).encode("ascii", "ignore").decode() words = re.findall(r"[a-z0-9]+", text.lower()) slug = "-".join(words)[:maxlen].strip("-") return slug or "untitled" def get_post(conn, post_id: int): return conn.execute( """SELECT p.*, s.slug AS source_slug, s.name AS source_name, s.direction AS source_direction, s.lang AS source_lang FROM posts p JOIN sources s ON s.id = p.source_id WHERE p.id = ?""", (post_id,), ).fetchone() def frontmatter(post, direction: str) -> dict: """Собирает YAML frontmatter по спеке.""" title = (post["text"] or "").replace("\n", " ").strip() if len(title) > 120: title = title[:117] + "..." if not title: title = f"{post['source_name']} — пост {post['tg_post_id']}" reactions = json.loads(post["reactions"] or "{}") # чистим ключи реакций: Telethon-объекты могли сохраниться как repr clean_reactions = {} for k, v in reactions.items(): key = k if "emoticon" in str(key): m = re.search(r"emoticon='([^']+)'", str(key)) if m: key = m.group(1) clean_reactions[key] = v is_own = int(post.get("is_own") or 0) == 1 fm = { "title": title, "date": (post["published_at"] or "")[:10], # YYYY-MM-DD "direction": direction, "source": post["source_slug"], "source_name": post["source_name"], "url": post["url"], "views": post["views"] or 0, "reactions": clean_reactions, "interest": None, # заполняется классификатором "status": post["status"] or "new", "content_type": post["content_type"], "media": post["media_path"], "tg_channel": post["tg_channel"], "tg_post_id": post["tg_post_id"], } # СВОЙ контент: origin: own, source_url на оригинал, автор if is_own: fm["origin"] = "own" if post.get("tg_channel") == "dedinit" and post.get("tg_post_id"): fm["source_url"] = f"https://t.me/dedinit/{post['tg_post_id']}" elif post.get("url"): fm["source_url"] = post["url"] fm["author"] = "Дед в АйТи" else: fm["origin"] = "external" return fm def make_bundle(post, direction: str, overwrite: bool = False) -> Path: """Создаёт бандл для поста. Возвращает путь к md или None, если уже есть.""" if "id" not in post.keys() and isinstance(post, sqlite3.Row): post = dict(post) slug = slugify(post["text"] or f"post-{post['tg_post_id']}") month = (post["published_at"] or datetime.utcnow().isoformat())[:7] # YYYY-MM bundle_dir = BUNDLES_DIR / direction / month media_dir = bundle_dir / "media" / slug md_path = bundle_dir / f"{slug}.md" # коллизия слага (2 поста с одинаковым заголовком) — добавляем суффикс if md_path.exists() and not overwrite: existing = md_path.read_text(encoding="utf-8") if f"tg_post_id: {post['tg_post_id']}" in existing: return md_path # бандл уже создан для этого поста # другой пост с таким же текстом — суффикс i = 2 while (bundle_dir / f"{slug}-{i}.md").exists(): i += 1 slug = f"{slug}-{i}" md_path = bundle_dir / f"{slug}.md" media_dir = bundle_dir / "media" / slug bundle_dir.mkdir(parents=True, exist_ok=True) # копируем медиа, если есть media_rel = None if post.get("media_path"): src = MEDIA_SRC / os.path.basename(post["media_path"]) if src.exists(): media_dir.mkdir(parents=True, exist_ok=True) dst = media_dir / src.name if not dst.exists(): shutil.copy2(src, dst) media_rel = f"media/{slug}/{src.name}" else: print(f" ! медиа не найдено на диске: {src}") fm = frontmatter(post, direction) if media_rel: fm["media_local"] = media_rel body = post["text"] or "" content = ( "---\n" + yaml.safe_dump(fm, allow_unicode=True, sort_keys=False).strip() + "\n---\n\n" + body + "\n\n## Источники\n\n" + f"- [{post['source_name']}]({post['url']}) — {post['tg_channel']} ({post['tg_post_id']})\n" ) md_path.write_text(content, encoding="utf-8") return md_path def create_bundle(post: dict, bundles_dir: Path | None = None, directions: list[str] | None = None) -> dict: """Создаёт бандл(ы) для поста (для веб-подтверждения). Возвращает {path, slug} или {paths}. directions: список направлений — бандл создаётся по КАЖДОМУ (fan-out). Если directions не задан — направление поста (source_direction/direction/linux). """ global BUNDLES_DIR if bundles_dir is not None: BUNDLES_DIR = Path(bundles_dir) if not directions: directions = [post.get("source_direction") or post.get("direction") or "linux"] dirs = list(dict.fromkeys([d for d in directions if d])) p = dict(post) p.setdefault("source_slug", p.get("source_slug") or p.get("source_name") or "unknown") p.setdefault("source_name", p.get("source_name") or p.get("source_slug") or "") paths = [] first = None for direction in dirs: try: path = make_bundle(p, direction) # храним ОТНОСИТЕЛЬНЫЙ путь от bundles/ (для веба: /bundle/) rel = str(path.relative_to(BUNDLES_DIR)) paths.append(rel) if first is None: first = rel except Exception as e: return {"error": str(e), "path": "", "slug": ""} if first is None: return {"error": "no bundles", "path": "", "slug": ""} return {"path": first, "paths": paths, "slug": Path(first).stem} def build_all(direction: str = "linux", statuses=("new",), limit: int = 0, overwrite: bool = False): """Собирает бандлы для постов заданного направления.""" conn = sqlite3.connect(DB_PATH) conn.row_factory = sqlite3.Row ph = ",".join("?" for _ in statuses) rows = conn.execute( f"""SELECT p.*, s.slug AS source_slug, s.name AS source_name, s.direction AS source_direction, s.lang AS source_lang FROM posts p JOIN sources s ON s.id = p.source_id WHERE s.direction = ? AND p.status IN ({ph}) ORDER BY p.published_at DESC""", (direction, *statuses), ).fetchall() if limit: rows = rows[:limit] made = skipped = 0 for row in rows: p = dict(row) direction_eff = p.get("source_direction") or direction try: path = make_bundle(p, direction_eff, overwrite=overwrite) except Exception as e: print(f" ERR post {p.get('id')}: {e}") continue if path.exists(): made += 1 else: skipped += 1 conn.close() print(f"Бандлы: собрано {made}, пропущено {skipped}") if __name__ == "__main__": import sys direction = sys.argv[1] if len(sys.argv) > 1 else "linux" build_all(direction)