mirror of
https://gitverse.ru/kpa39l/chronicle.nixg.ru.git
synced 2026-09-29 18:05:07 +00:00
Add telegram-archiver microservice (MVP)
This commit is contained in:
@@ -0,0 +1,374 @@
|
||||
"""
|
||||
Core archiver logic.
|
||||
Handles message processing, markdown generation, and file organization.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import yaml
|
||||
from telethon.tl.types import Message
|
||||
|
||||
from app.models import PostData, MediaFile, MediaType
|
||||
from app.telethon_client import TelethonArchiver
|
||||
from config import Settings, get_settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class ChannelArchiver:
|
||||
"""
|
||||
Handles archiving of a single Telegram channel.
|
||||
Manages bundle creation, markdown generation, and media downloads.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
telethon_client: TelethonArchiver,
|
||||
settings: Optional[Settings] = None,
|
||||
):
|
||||
self.client = telethon_client
|
||||
self.settings = settings or get_settings()
|
||||
self.channel_path: Optional[Path] = None
|
||||
self.log_file: Optional[Path] = None
|
||||
|
||||
# Statistics
|
||||
self.stats = {
|
||||
"posts_archived": 0,
|
||||
"posts_skipped": 0,
|
||||
"media_downloaded": 0,
|
||||
"media_skipped_large": 0,
|
||||
"large_files": [],
|
||||
}
|
||||
|
||||
def _setup_logging(self, channel_username: str) -> None:
|
||||
"""Setup file logging for this channel archive session."""
|
||||
timestamp = datetime.now().strftime("%Y%m%d")
|
||||
log_filename = f"{timestamp}-{channel_username}.log"
|
||||
self.log_file = Path(__file__).parent.parent / log_filename
|
||||
|
||||
# Create file handler
|
||||
file_handler = logging.FileHandler(self.log_file, encoding="utf-8")
|
||||
file_handler.setLevel(self.settings.log_level)
|
||||
|
||||
# Create formatter
|
||||
formatter = logging.Formatter(
|
||||
"%(asctime)s - %(name)s - %(levelname)s - %(message)s"
|
||||
)
|
||||
file_handler.setFormatter(formatter)
|
||||
|
||||
# Add handler to logger
|
||||
logger.addHandler(file_handler)
|
||||
logger.info(f"=== Archive session started for {channel_username} ===")
|
||||
|
||||
def _get_channel_path(self, channel_username: str) -> Path:
|
||||
"""Get the output directory path for this channel."""
|
||||
# Remove @ prefix if present
|
||||
clean_username = channel_username.lstrip("@")
|
||||
base_path = self.settings.output_dir / clean_username
|
||||
base_path.mkdir(parents=True, exist_ok=True)
|
||||
return base_path
|
||||
|
||||
async def archive_channel(
|
||||
self,
|
||||
channel: str,
|
||||
limit: Optional[int] = None,
|
||||
from_message_id: Optional[int] = None,
|
||||
force: bool = False,
|
||||
) -> dict:
|
||||
"""
|
||||
Archive all messages from a channel.
|
||||
|
||||
Args:
|
||||
channel: Channel username or ID
|
||||
limit: Maximum number of messages to archive
|
||||
from_message_id: Start from this message ID
|
||||
force: Force re-download of already archived posts
|
||||
|
||||
Returns:
|
||||
Statistics dictionary
|
||||
"""
|
||||
start_time = datetime.now()
|
||||
|
||||
# Setup
|
||||
self._setup_logging(channel)
|
||||
self.channel_path = self._get_channel_path(channel)
|
||||
|
||||
logger.info(f"Starting archive of {channel}")
|
||||
logger.info(f"Output directory: {self.channel_path}")
|
||||
if limit:
|
||||
logger.info(f"Limit: {limit} messages")
|
||||
if from_message_id:
|
||||
logger.info(f"Starting from message ID: {from_message_id}")
|
||||
|
||||
# Get channel info
|
||||
entity = await self.client.get_channel_info(channel)
|
||||
channel_title = getattr(entity, "title", channel)
|
||||
|
||||
# Reset stats
|
||||
self.stats = {
|
||||
"posts_archived": 0,
|
||||
"posts_skipped": 0,
|
||||
"media_downloaded": 0,
|
||||
"media_skipped_large": 0,
|
||||
"large_files": [],
|
||||
}
|
||||
|
||||
# Process messages
|
||||
async for message in self.client.fetch_messages(
|
||||
channel, limit=limit, from_message_id=from_message_id
|
||||
):
|
||||
await self._process_message(message, force=force)
|
||||
|
||||
# Write large files report
|
||||
large_files_report = await self._write_large_files_report()
|
||||
|
||||
# Calculate duration
|
||||
duration = (datetime.now() - start_time).total_seconds()
|
||||
|
||||
logger.info(f"=== Archive completed in {duration:.2f}s ===")
|
||||
logger.info(f"Posts archived: {self.stats['posts_archived']}")
|
||||
logger.info(f"Posts skipped: {self.stats['posts_skipped']}")
|
||||
logger.info(f"Media downloaded: {self.stats['media_downloaded']}")
|
||||
logger.info(f"Media skipped (large): {self.stats['media_skipped_large']}")
|
||||
|
||||
return {
|
||||
"channel": channel,
|
||||
"channel_title": channel_title,
|
||||
"posts_archived": self.stats["posts_archived"],
|
||||
"posts_skipped": self.stats["posts_skipped"],
|
||||
"media_downloaded": self.stats["media_downloaded"],
|
||||
"media_skipped_large": self.stats["media_skipped_large"],
|
||||
"output_path": str(self.channel_path),
|
||||
"log_file": str(self.log_file),
|
||||
"large_files_report": large_files_report,
|
||||
"duration_seconds": duration,
|
||||
}
|
||||
|
||||
async def _process_message(
|
||||
self, message: Message, force: bool = False
|
||||
) -> None:
|
||||
"""
|
||||
Process a single message: create bundle, download media, generate markdown.
|
||||
|
||||
Args:
|
||||
message: Telegram message to process
|
||||
force: Force re-processing if bundle exists
|
||||
"""
|
||||
message_id = message.id
|
||||
bundle_dir = self.channel_path / str(message_id)
|
||||
|
||||
# Check if already archived (skip if exists and not force)
|
||||
if bundle_dir.exists() and not force:
|
||||
logger.debug(f"Skipping message {message_id}: already archived")
|
||||
self.stats["posts_skipped"] += 1
|
||||
return
|
||||
|
||||
logger.debug(f"Processing message {message_id}...")
|
||||
|
||||
# Create bundle directory
|
||||
bundle_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Extract post data
|
||||
post_data = await self._extract_post_data(message)
|
||||
|
||||
# Download media
|
||||
if message.media:
|
||||
media_info = await self.client.download_media(
|
||||
message, bundle_dir, self.settings.max_file_size
|
||||
)
|
||||
|
||||
# Update post_data with media info
|
||||
for info in media_info:
|
||||
media_file = MediaFile(
|
||||
filename=info["filename"],
|
||||
type=info["type"],
|
||||
caption=info.get("caption"),
|
||||
size=info.get("size"),
|
||||
is_too_large=info.get("is_too_large", False),
|
||||
)
|
||||
post_data.media_files.append(media_file)
|
||||
|
||||
if info.get("is_too_large"):
|
||||
self.stats["media_skipped_large"] += 1
|
||||
self.stats["large_files"].append(
|
||||
{
|
||||
"message_id": message_id,
|
||||
"filename": info["filename"],
|
||||
"size": info.get("size"),
|
||||
}
|
||||
)
|
||||
else:
|
||||
self.stats["media_downloaded"] += 1
|
||||
|
||||
# Generate markdown
|
||||
markdown_content = self._generate_markdown(post_data)
|
||||
|
||||
# Write index.md
|
||||
index_path = bundle_dir / "index.md"
|
||||
index_path.write_text(markdown_content, encoding="utf-8")
|
||||
|
||||
self.stats["posts_archived"] += 1
|
||||
logger.debug(f"Message {message_id} archived successfully")
|
||||
|
||||
async def _extract_post_data(self, message: Message) -> PostData:
|
||||
"""
|
||||
Extract structured data from a Telegram message.
|
||||
|
||||
Args:
|
||||
message: Telegram message
|
||||
|
||||
Returns:
|
||||
PostData object with all metadata
|
||||
"""
|
||||
# Get channel info
|
||||
channel = await message.get_chat()
|
||||
channel_username = getattr(channel, "username", "unknown")
|
||||
channel_title = getattr(channel, "title", channel_username)
|
||||
|
||||
# Convert message text to markdown
|
||||
text = message.message or ""
|
||||
if message.message:
|
||||
# Use Telethon's built-in markdown converter
|
||||
text = message.get_message_text(markdown=True)
|
||||
|
||||
# Handle reply-to
|
||||
reply_to = None
|
||||
if message.reply_to and message.reply_to.reply_to_msg_id:
|
||||
reply_to = message.reply_to.reply_to_msg_id
|
||||
|
||||
# Handle repost/forward
|
||||
repost_from = None
|
||||
repost_channel = None
|
||||
if message.fwd_from:
|
||||
repost_from = message.fwd_from.channel_post
|
||||
if message.fwd_from.from_id:
|
||||
# Try to get original channel info
|
||||
try:
|
||||
fwd_channel = await self.client.client.get_entity(
|
||||
message.fwd_from.from_id
|
||||
)
|
||||
repost_channel = getattr(fwd_channel, "title", None)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return PostData(
|
||||
message_id=message.id,
|
||||
date=message.date,
|
||||
text=text,
|
||||
author=channel_title,
|
||||
channel_username=channel_username,
|
||||
reply_to=reply_to,
|
||||
repost_from=repost_from,
|
||||
repost_channel=repost_channel,
|
||||
views=getattr(message, "views", None),
|
||||
)
|
||||
|
||||
def _generate_markdown(self, post: PostData) -> str:
|
||||
"""
|
||||
Generate markdown content with front-matter for Hugo.
|
||||
|
||||
Args:
|
||||
post: PostData object
|
||||
|
||||
Returns:
|
||||
Markdown string with YAML front-matter
|
||||
"""
|
||||
# Build front-matter
|
||||
frontmatter = {
|
||||
"message_id": post.message_id,
|
||||
"date": post.date.isoformat(),
|
||||
"author": post.author,
|
||||
}
|
||||
|
||||
# Add optional fields
|
||||
if post.reply_to:
|
||||
# Relative link to the replied post's index.md
|
||||
frontmatter["reply_to"] = f"../{post.reply_to}/index.md"
|
||||
|
||||
if post.repost_from:
|
||||
frontmatter["repost_from"] = post.repost_from
|
||||
if post.repost_channel:
|
||||
frontmatter["repost_channel"] = post.repost_channel
|
||||
|
||||
if post.media_files:
|
||||
frontmatter["media_files"] = [
|
||||
{
|
||||
"filename": mf.filename,
|
||||
"type": mf.type,
|
||||
"caption": mf.caption,
|
||||
"size": mf.size,
|
||||
"is_too_large": mf.is_too_large,
|
||||
}
|
||||
for mf in post.media_files
|
||||
]
|
||||
|
||||
# Build markdown content
|
||||
content = []
|
||||
|
||||
# Add media embeds for downloaded files
|
||||
for mf in post.media_files:
|
||||
if not mf.is_too_large:
|
||||
if mf.type == "photo":
|
||||
content.append(f"")
|
||||
elif mf.type == "video":
|
||||
content.append(f"@[video]({mf.filename})")
|
||||
elif mf.type == "audio":
|
||||
content.append(f"@[audio]({mf.filename})")
|
||||
elif mf.type == "document":
|
||||
content.append(f"📎 [{mf.filename}]({mf.filename})")
|
||||
|
||||
# Add text content
|
||||
if post.text:
|
||||
content.append(post.text)
|
||||
|
||||
# Mark reposts visually
|
||||
if post.repost_from:
|
||||
content.insert(0, f"> *Repost from {post.repost_channel or 'Unknown'}*")
|
||||
content.insert(1, "")
|
||||
|
||||
# Combine front-matter and content
|
||||
yaml_frontmatter = yaml.dump(
|
||||
frontmatter,
|
||||
allow_unicode=True,
|
||||
default_flow_style=False,
|
||||
sort_keys=False,
|
||||
)
|
||||
|
||||
return f"---\n{yaml_frontmatter}---\n\n{''.join(content)}"
|
||||
|
||||
async def _write_large_files_report(self) -> Optional[str]:
|
||||
"""
|
||||
Write report of files that were too large to download.
|
||||
|
||||
Returns:
|
||||
Path to report file or None if no large files
|
||||
"""
|
||||
if not self.stats["large_files"]:
|
||||
return None
|
||||
|
||||
report_path = self.channel_path / "2big2get.md"
|
||||
|
||||
content = ["# Files Too Large to Download\n\n"]
|
||||
content.append(
|
||||
f"Generated: {datetime.now().isoformat()}\n\n"
|
||||
)
|
||||
content.append(
|
||||
f"Total files skipped: {len(self.stats['large_files'])}\n\n"
|
||||
)
|
||||
content.append("| Message ID | Filename | Size (bytes) |\n")
|
||||
content.append("|------------|----------|-------------|\n")
|
||||
|
||||
for file_info in self.stats["large_files"]:
|
||||
content.append(
|
||||
f"| {file_info['message_id']} | {file_info['filename']} | "
|
||||
f"{file_info['size']} |\n"
|
||||
)
|
||||
|
||||
report_path.write_text("".join(content), encoding="utf-8")
|
||||
logger.info(f"Large files report written to {report_path}")
|
||||
|
||||
return str(report_path)
|
||||
Reference in New Issue
Block a user