Initial import: vesti.nixg.ru — новостной апрув-проект (web, crawler, classifier, publisher, openspec)

This commit is contained in:
kpa39l
2026-09-13 15:58:32 +00:00
commit c3f59f7b7a
113 changed files with 7065 additions and 0 deletions
+243
View File
@@ -0,0 +1,243 @@
# VESTI — Банк статей: markdown-бандлы + медиа.
# Спека: openspec/changes/tg-crawler-publisher-prototype/specs/news-store/spec.md
# Путь: bundles/<направление>/<YYYY-MM>/<slug>.md
# Медиа: bundles/<направление>/<YYYY-MM>/media/<slug>/
# frontmatter: title, date, direction, source, url, views, reactions, interest, status
import json
import os
import re
import shutil
import sqlite3
import unicodedata
from datetime import datetime
from pathlib import Path
import yaml
BASE = Path(__file__).resolve().parent.parent
DB_PATH = BASE / "db" / "vesti.db"
BUNDLES_DIR = BASE / "bundles"
MEDIA_SRC = BASE / "media" / "media" # каталог, куда краулер складывает файлы
DIRECTION_LABELS = {
"linux": "Linux",
"tech": "Технологии",
"politics": "Политика",
"games": "Игры",
"electronics": "Электроника",
"llm": "LLM / AI",
}
def slugify(text: str, maxlen: int = 60) -> str:
"""Транслитерация + slug. Убирает URL/служебное. 'Omarchy — Arch Linux...' -> 'omarchy-arch-linux'."""
text = (text or "").strip()
# убираем markdown/эмодзи/служебное
text = re.sub(r"[*_`#\[\]()>|]", "", text)
# убираем URL (http, т.me, ссылки) — из них берём только куски без протокола
text = re.sub(r"https?://\S+|t\.me/\S+|ya\.cc/\S+", "", text)
text = re.sub(r"\s+", " ", text).strip()
# обрезаем до 6 слов — иначе слаг раздувается в белиберду
words_ = text.split()
if len(words_) > 6:
text = " ".join(words_[:6])
# транслитерация кириллицы (правильные буквосочетания)
tr_map = {
"а": "a", "б": "b", "в": "v", "г": "g", "д": "d", "е": "e", "ё": "e",
"ж": "zh", "з": "z", "и": "i", "й": "y", "к": "k", "л": "l", "м": "m",
"н": "n", "о": "o", "п": "p", "р": "r", "с": "s", "т": "t", "у": "u",
"ф": "f", "х": "h", "ц": "ts", "ч": "ch", "ш": "sh", "щ": "sch",
"ъ": "", "ы": "y", "ь": "", "э": "e", "ю": "yu", "я": "ya",
"А": "A", "Б": "B", "В": "V", "Г": "G", "Д": "D", "Е": "E", "Ё": "E",
"Ж": "Zh", "З": "Z", "И": "I", "Й": "Y", "К": "K", "Л": "L", "М": "M",
"Н": "N", "О": "O", "П": "P", "Р": "R", "С": "S", "Т": "T", "У": "U",
"Ф": "F", "Х": "H", "Ц": "Ts", "Ч": "Ch", "Ш": "Sh", "Щ": "Sch",
"Ъ": "", "Ы": "Y", "Ь": "", "Э": "E", "Ю": "Yu", "Я": "Ya",
}
text = "".join(tr_map.get(ch, ch) for ch in text)
# нормализация юникода, оставляем латиницу/цифры/дефис
text = unicodedata.normalize("NFKD", text).encode("ascii", "ignore").decode()
words = re.findall(r"[a-z0-9]+", text.lower())
slug = "-".join(words)[:maxlen].strip("-")
return slug or "untitled"
def get_post(conn, post_id: int):
return conn.execute(
"""SELECT p.*, s.slug AS source_slug, s.name AS source_name,
s.direction AS source_direction, s.lang AS source_lang
FROM posts p JOIN sources s ON s.id = p.source_id WHERE p.id = ?""",
(post_id,),
).fetchone()
def frontmatter(post, direction: str) -> dict:
"""Собирает YAML frontmatter по спеке."""
title = (post["text"] or "").replace("\n", " ").strip()
if len(title) > 120:
title = title[:117] + "..."
if not title:
title = f"{post['source_name']} — пост {post['tg_post_id']}"
reactions = json.loads(post["reactions"] or "{}")
# чистим ключи реакций: Telethon-объекты могли сохраниться как repr
clean_reactions = {}
for k, v in reactions.items():
key = k
if "emoticon" in str(key):
m = re.search(r"emoticon='([^']+)'", str(key))
if m:
key = m.group(1)
clean_reactions[key] = v
is_own = int(post.get("is_own") or 0) == 1
fm = {
"title": title,
"date": (post["published_at"] or "")[:10], # YYYY-MM-DD
"direction": direction,
"source": post["source_slug"],
"source_name": post["source_name"],
"url": post["url"],
"views": post["views"] or 0,
"reactions": clean_reactions,
"interest": None, # заполняется классификатором
"status": post["status"] or "new",
"content_type": post["content_type"],
"media": post["media_path"],
"tg_channel": post["tg_channel"],
"tg_post_id": post["tg_post_id"],
}
# СВОЙ контент: origin: own, source_url на оригинал, автор
if is_own:
fm["origin"] = "own"
if post.get("tg_channel") == "dedinit" and post.get("tg_post_id"):
fm["source_url"] = f"https://t.me/dedinit/{post['tg_post_id']}"
elif post.get("url"):
fm["source_url"] = post["url"]
fm["author"] = "Дед в АйТи"
else:
fm["origin"] = "external"
return fm
def make_bundle(post, direction: str, overwrite: bool = False) -> Path:
"""Создаёт бандл для поста. Возвращает путь к md или None, если уже есть."""
if "id" not in post.keys() and isinstance(post, sqlite3.Row):
post = dict(post)
slug = slugify(post["text"] or f"post-{post['tg_post_id']}")
month = (post["published_at"] or datetime.utcnow().isoformat())[:7] # YYYY-MM
bundle_dir = BUNDLES_DIR / direction / month
media_dir = bundle_dir / "media" / slug
md_path = bundle_dir / f"{slug}.md"
# коллизия слага (2 поста с одинаковым заголовком) — добавляем суффикс
if md_path.exists() and not overwrite:
existing = md_path.read_text(encoding="utf-8")
if f"tg_post_id: {post['tg_post_id']}" in existing:
return md_path # бандл уже создан для этого поста
# другой пост с таким же текстом — суффикс
i = 2
while (bundle_dir / f"{slug}-{i}.md").exists():
i += 1
slug = f"{slug}-{i}"
md_path = bundle_dir / f"{slug}.md"
media_dir = bundle_dir / "media" / slug
bundle_dir.mkdir(parents=True, exist_ok=True)
# копируем медиа, если есть
media_rel = None
if post.get("media_path"):
src = MEDIA_SRC / os.path.basename(post["media_path"])
if src.exists():
media_dir.mkdir(parents=True, exist_ok=True)
dst = media_dir / src.name
if not dst.exists():
shutil.copy2(src, dst)
media_rel = f"media/{slug}/{src.name}"
else:
print(f" ! медиа не найдено на диске: {src}")
fm = frontmatter(post, direction)
if media_rel:
fm["media_local"] = media_rel
body = post["text"] or ""
content = (
"---\n"
+ yaml.safe_dump(fm, allow_unicode=True, sort_keys=False).strip()
+ "\n---\n\n"
+ body
+ "\n\n## Источники\n\n"
+ f"- [{post['source_name']}]({post['url']}) — {post['tg_channel']} ({post['tg_post_id']})\n"
)
md_path.write_text(content, encoding="utf-8")
return md_path
def create_bundle(post: dict, bundles_dir: Path | None = None, directions: list[str] | None = None) -> dict:
"""Создаёт бандл(ы) для поста (для веб-подтверждения). Возвращает {path, slug} или {paths}.
directions: список направлений — бандл создаётся по КАЖДОМУ (fan-out).
Если directions не задан — направление поста (source_direction/direction/linux).
"""
global BUNDLES_DIR
if bundles_dir is not None:
BUNDLES_DIR = Path(bundles_dir)
if not directions:
directions = [post.get("source_direction") or post.get("direction") or "linux"]
dirs = list(dict.fromkeys([d for d in directions if d]))
p = dict(post)
p.setdefault("source_slug", p.get("source_slug") or p.get("source_name") or "unknown")
p.setdefault("source_name", p.get("source_name") or p.get("source_slug") or "")
paths = []
first = None
for direction in dirs:
try:
path = make_bundle(p, direction)
# храним ОТНОСИТЕЛЬНЫЙ путь от bundles/ (для веба: /bundle/<rel>)
rel = str(path.relative_to(BUNDLES_DIR))
paths.append(rel)
if first is None:
first = rel
except Exception as e:
return {"error": str(e), "path": "", "slug": ""}
if first is None:
return {"error": "no bundles", "path": "", "slug": ""}
return {"path": first, "paths": paths, "slug": Path(first).stem}
def build_all(direction: str = "linux", statuses=("new",), limit: int = 0, overwrite: bool = False):
"""Собирает бандлы для постов заданного направления."""
conn = sqlite3.connect(DB_PATH)
conn.row_factory = sqlite3.Row
ph = ",".join("?" for _ in statuses)
rows = conn.execute(
f"""SELECT p.*, s.slug AS source_slug, s.name AS source_name,
s.direction AS source_direction, s.lang AS source_lang
FROM posts p JOIN sources s ON s.id = p.source_id
WHERE s.direction = ? AND p.status IN ({ph})
ORDER BY p.published_at DESC""",
(direction, *statuses),
).fetchall()
if limit:
rows = rows[:limit]
made = skipped = 0
for row in rows:
p = dict(row)
direction_eff = p.get("source_direction") or direction
try:
path = make_bundle(p, direction_eff, overwrite=overwrite)
except Exception as e:
print(f" ERR post {p.get('id')}: {e}")
continue
if path.exists():
made += 1
else:
skipped += 1
conn.close()
print(f"Бандлы: собрано {made}, пропущено {skipped}")
if __name__ == "__main__":
import sys
direction = sys.argv[1] if len(sys.argv) > 1 else "linux"
build_all(direction)