mirror of
https://gitverse.ru/kpa39l/chronicle.nixg.ru.git
synced 2026-09-29 09:55:08 +00:00
Phase 0 MVP: Telegram Archiver fully functional (992 posts tested)
This commit is contained in:
@@ -5,6 +5,7 @@ Handles message processing, markdown generation, and file organization.
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import re
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
@@ -19,6 +20,54 @@ from config import Settings, get_settings
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def extract_hashtags(text: str) -> list[str]:
|
||||
"""
|
||||
Extract hashtags from text.
|
||||
|
||||
Matches: #tag, #тег, #tag123, #тэг_с_подчёркиванием
|
||||
Supports: Latin, Cyrillic, digits, underscores
|
||||
|
||||
Args:
|
||||
text: Message text
|
||||
|
||||
Returns:
|
||||
List of unique hashtags (without #)
|
||||
"""
|
||||
if not text:
|
||||
return []
|
||||
|
||||
# Regex for hashtags: # followed by word chars (including Cyrillic)
|
||||
pattern = r'#([\wа-яА-ЯёЁ\d_]+)'
|
||||
matches = re.findall(pattern, text)
|
||||
|
||||
# Return unique tags, lowercase
|
||||
return list(set(tag.lower() for tag in matches))
|
||||
|
||||
|
||||
def remove_hashtags(text: str) -> str:
|
||||
"""
|
||||
Remove hashtags from text.
|
||||
|
||||
Args:
|
||||
text: Message text with hashtags
|
||||
|
||||
Returns:
|
||||
Text without hashtags (extra whitespace cleaned)
|
||||
"""
|
||||
if not text:
|
||||
return text
|
||||
|
||||
# Remove hashtags
|
||||
pattern = r'#\w+[\wа-яА-ЯёЁ\d_]*'
|
||||
text = re.sub(pattern, '', text)
|
||||
|
||||
# Clean up multiple spaces/newlines
|
||||
text = re.sub(r'\n\s*\n', '\n\n', text)
|
||||
text = re.sub(r' {2,}', ' ', text)
|
||||
|
||||
return text.strip()
|
||||
|
||||
|
||||
class ChannelArchiver:
|
||||
"""
|
||||
Handles archiving of a single Telegram channel.
|
||||
@@ -167,52 +216,62 @@ class ChannelArchiver:
|
||||
self.stats["posts_skipped"] += 1
|
||||
return
|
||||
|
||||
logger.debug(f"Processing message {message_id}...")
|
||||
try:
|
||||
logger.debug(f"Processing message {message_id}...")
|
||||
|
||||
# Create bundle directory
|
||||
bundle_dir.mkdir(parents=True, exist_ok=True)
|
||||
# Create bundle directory
|
||||
bundle_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Extract post data
|
||||
post_data = await self._extract_post_data(message)
|
||||
# Extract post data
|
||||
post_data = await self._extract_post_data(message)
|
||||
|
||||
# Download media
|
||||
if message.media:
|
||||
media_info = await self.client.download_media(
|
||||
message, bundle_dir, self.settings.max_file_size
|
||||
)
|
||||
|
||||
# Update post_data with media info
|
||||
for info in media_info:
|
||||
media_file = MediaFile(
|
||||
filename=info["filename"],
|
||||
type=info["type"],
|
||||
caption=info.get("caption"),
|
||||
size=info.get("size"),
|
||||
is_too_large=info.get("is_too_large", False),
|
||||
# Download media
|
||||
if message.media:
|
||||
media_info = await self.client.download_media(
|
||||
message, bundle_dir, self.settings.max_file_size, timeout=30
|
||||
)
|
||||
post_data.media_files.append(media_file)
|
||||
|
||||
if info.get("is_too_large"):
|
||||
self.stats["media_skipped_large"] += 1
|
||||
self.stats["large_files"].append(
|
||||
{
|
||||
"message_id": message_id,
|
||||
"filename": info["filename"],
|
||||
"size": info.get("size"),
|
||||
}
|
||||
# Update post_data with media info
|
||||
for info in media_info:
|
||||
media_file = MediaFile(
|
||||
filename=info["filename"],
|
||||
type=info["type"],
|
||||
caption=info.get("caption"),
|
||||
size=info.get("size"),
|
||||
is_too_large=info.get("is_too_large", False),
|
||||
)
|
||||
else:
|
||||
self.stats["media_downloaded"] += 1
|
||||
post_data.media_files.append(media_file)
|
||||
|
||||
# Generate markdown
|
||||
markdown_content = self._generate_markdown(post_data)
|
||||
if info.get("is_too_large"):
|
||||
self.stats["media_skipped_large"] += 1
|
||||
self.stats["large_files"].append(
|
||||
{
|
||||
"message_id": message_id,
|
||||
"filename": info["filename"],
|
||||
"size": info.get("size"),
|
||||
}
|
||||
)
|
||||
else:
|
||||
self.stats["media_downloaded"] += 1
|
||||
|
||||
# Write index.md
|
||||
index_path = bundle_dir / "index.md"
|
||||
index_path.write_text(markdown_content, encoding="utf-8")
|
||||
# Generate markdown
|
||||
markdown_content = self._generate_markdown(post_data)
|
||||
|
||||
self.stats["posts_archived"] += 1
|
||||
logger.debug(f"Message {message_id} archived successfully")
|
||||
# Write index.md
|
||||
index_path = bundle_dir / "index.md"
|
||||
index_path.write_text(markdown_content, encoding="utf-8")
|
||||
|
||||
self.stats["posts_archived"] += 1
|
||||
|
||||
# Log progress every 50 posts
|
||||
if self.stats["posts_archived"] % 50 == 0:
|
||||
logger.info(f"Progress: {self.stats['posts_archived']} posts archived...")
|
||||
|
||||
logger.debug(f"Message {message_id} archived successfully")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to process message {message_id}: {e}")
|
||||
self.stats["posts_skipped"] += 1
|
||||
|
||||
async def _extract_post_data(self, message: Message) -> PostData:
|
||||
"""
|
||||
@@ -230,10 +289,18 @@ class ChannelArchiver:
|
||||
channel_title = getattr(channel, "title", channel_username)
|
||||
|
||||
# Convert message text to markdown
|
||||
text = message.message or ""
|
||||
if message.message:
|
||||
# Use Telethon's built-in markdown converter
|
||||
text = message.get_message_text(markdown=True)
|
||||
original_text = message.message or ""
|
||||
if message.message and message.entities:
|
||||
# Use Telethon's markdown parser with entities
|
||||
text = message.text # Plain text
|
||||
else:
|
||||
text = original_text
|
||||
|
||||
# Extract hashtags from text
|
||||
tags = extract_hashtags(text)
|
||||
|
||||
# Remove hashtags from text (clean up)
|
||||
clean_text = remove_hashtags(text)
|
||||
|
||||
# Handle reply-to
|
||||
reply_to = None
|
||||
@@ -258,18 +325,20 @@ class ChannelArchiver:
|
||||
return PostData(
|
||||
message_id=message.id,
|
||||
date=message.date,
|
||||
text=text,
|
||||
text=clean_text, # Use cleaned text (without hashtags)
|
||||
author=channel_title,
|
||||
channel_username=channel_username,
|
||||
reply_to=reply_to,
|
||||
repost_from=repost_from,
|
||||
repost_channel=repost_channel,
|
||||
views=getattr(message, "views", None),
|
||||
tags=tags, # Extracted hashtags
|
||||
)
|
||||
|
||||
def _generate_markdown(self, post: PostData) -> str:
|
||||
"""
|
||||
Generate markdown content with front-matter for Hugo.
|
||||
Uses Hugo shortcodes for media embedding.
|
||||
|
||||
Args:
|
||||
post: PostData object
|
||||
@@ -284,6 +353,10 @@ class ChannelArchiver:
|
||||
"author": post.author,
|
||||
}
|
||||
|
||||
# Add tags if present
|
||||
if post.tags:
|
||||
frontmatter["tags"] = post.tags
|
||||
|
||||
# Add optional fields
|
||||
if post.reply_to:
|
||||
# Relative link to the replied post's index.md
|
||||
@@ -309,21 +382,31 @@ class ChannelArchiver:
|
||||
# Build markdown content
|
||||
content = []
|
||||
|
||||
# Add media embeds for downloaded files
|
||||
# Add media embeds using Hugo shortcodes
|
||||
for mf in post.media_files:
|
||||
if not mf.is_too_large:
|
||||
if mf.type == "photo":
|
||||
content.append(f"")
|
||||
# Hugo figure shortcode
|
||||
if mf.caption:
|
||||
content.append(f'{{{{< figure src="{mf.filename}" alt="{mf.caption}" title="{mf.caption}" >}}}}')
|
||||
else:
|
||||
content.append(f'{{{{< figure src="{mf.filename}" >}}}}')
|
||||
elif mf.type == "video":
|
||||
content.append(f"@[video]({mf.filename})")
|
||||
# Hugo video shortcode
|
||||
content.append(f'{{{{< video src="{mf.filename}" >}}}}')
|
||||
elif mf.type == "audio":
|
||||
content.append(f"@[audio]({mf.filename})")
|
||||
# Hugo audio shortcode
|
||||
content.append(f'{{{{< audio src="{mf.filename}" >}}}}')
|
||||
elif mf.type == "document":
|
||||
content.append(f"📎 [{mf.filename}]({mf.filename})")
|
||||
# Download link for documents
|
||||
content.append(f'📎 [{mf.filename}]({mf.filename})')
|
||||
|
||||
# Add text content
|
||||
# Add text content (only if not already in media caption)
|
||||
if post.text:
|
||||
content.append(post.text)
|
||||
# Check if text is already used as caption in media
|
||||
captions = [mf.caption for mf in post.media_files if mf.caption]
|
||||
if post.text not in captions:
|
||||
content.append(post.text)
|
||||
|
||||
# Mark reposts visually
|
||||
if post.repost_from:
|
||||
|
||||
Reference in New Issue
Block a user