Add lightweight web control panel

This commit is contained in:
2026-04-25 02:19:52 +02:00
parent 535b8195ef
commit f57e1d63eb
11 changed files with 1786 additions and 0 deletions
+227
View File
@@ -0,0 +1,227 @@
data/
.codex
.python-version
uv.lock
.uv-cache/
# Byte-compiled / optimized / DLL files
__pycache__/
*.py[codz]
*$py.class
# C extensions
*.so
# Distribution / packaging
.Python
build/
develop-eggs/
dist/
downloads/
eggs/
.eggs/
lib/
lib64/
parts/
sdist/
var/
wheels/
share/python-wheels/
*.egg-info/
.installed.cfg
*.egg
MANIFEST
# PyInstaller
# Usually these files are written by a python script from a template
# before PyInstaller builds the exe, so as to inject date/other infos into it.
*.manifest
*.spec
# Installer logs
pip-log.txt
pip-delete-this-directory.txt
# Unit test / coverage reports
htmlcov/
.tox/
.nox/
.coverage
.coverage.*
.cache
nosetests.xml
coverage.xml
*.cover
*.py.cover
*.lcov
.hypothesis/
.pytest_cache/
cover/
# Translations
*.mo
*.pot
# Django stuff:
*.log
local_settings.py
db.sqlite3
db.sqlite3-journal
# Flask stuff:
instance/
.webassets-cache
# Scrapy stuff:
.scrapy
# Sphinx documentation
docs/_build/
# PyBuilder
.pybuilder/
target/
# Jupyter Notebook
.ipynb_checkpoints
# IPython
profile_default/
ipython_config.py
# pyenv
# For a library or package, you might want to ignore these files since the code is
# intended to run in multiple environments; otherwise, check them in:
# .python-version
# pipenv
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
# However, in case of collaboration, if having platform-specific dependencies or dependencies
# having no cross-platform support, pipenv may install dependencies that don't work, or not
# install all needed dependencies.
# Pipfile.lock
# UV
# Similar to Pipfile.lock, it is generally recommended to include uv.lock in version control.
# This is especially recommended for binary packages to ensure reproducibility, and is more
# commonly ignored for libraries.
# uv.lock
# poetry
# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
# This is especially recommended for binary packages to ensure reproducibility, and is more
# commonly ignored for libraries.
# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
# poetry.lock
# poetry.toml
# pdm
# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
# pdm recommends including project-wide configuration in pdm.toml, but excluding .pdm-python.
# https://pdm-project.org/en/latest/usage/project/#working-with-version-control
# pdm.lock
# pdm.toml
.pdm-python
.pdm-build/
# pixi
# Similar to Pipfile.lock, it is generally recommended to include pixi.lock in version control.
# pixi.lock
# Pixi creates a virtual environment in the .pixi directory, just like venv module creates one
# in the .venv directory. It is recommended not to include this directory in version control.
.pixi/*
!.pixi/config.toml
# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
__pypackages__/
# Celery stuff
celerybeat-schedule*
celerybeat.pid
# Redis
*.rdb
*.aof
*.pid
# RabbitMQ
mnesia/
rabbitmq/
rabbitmq-data/
# ActiveMQ
activemq-data/
# SageMath parsed files
*.sage.py
# Environments
.env
.envrc
.venv
env/
venv/
ENV/
env.bak/
venv.bak/
# Spyder project settings
.spyderproject
.spyproject
# Rope project settings
.ropeproject
# mkdocs documentation
/site
# mypy
.mypy_cache/
.dmypy.json
dmypy.json
# Pyre type checker
.pyre/
# pytype static type analyzer
.pytype/
# Cython debug symbols
cython_debug/
# PyCharm
# JetBrains specific template is maintained in a separate JetBrains.gitignore that can
# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
# and can be added to the global gitignore or merged into this file. For a more nuclear
# option (not recommended) you can uncomment the following to ignore the entire idea folder.
# .idea/
# Abstra
# Abstra is an AI-powered process automation framework.
# Ignore directories containing user credentials, local state, and settings.
# Learn more at https://abstra.io/docs
.abstra/
# Visual Studio Code
# Visual Studio Code specific template is maintained in a separate VisualStudioCode.gitignore
# that can be found at https://github.com/github/gitignore/blob/main/Global/VisualStudioCode.gitignore
# and can be added to the global gitignore or merged into this file. However, if you prefer,
# you could uncomment the following to ignore the entire vscode folder
# .vscode/
# Temporary file for partial code execution
tempCodeRunnerFile.py
# Ruff stuff:
.ruff_cache/
# PyPI configuration file
.pypirc
# Marimo
marimo/_static/
marimo/_lsp/
__marimo__/
# Streamlit
.streamlit/secrets.toml
+18
View File
@@ -0,0 +1,18 @@
FROM python:3.11-slim
WORKDIR /app
RUN apt-get update && apt-get install -y --no-install-recommends \
gcc \
&& rm -rf /var/lib/apt/lists/*
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
COPY . .
EXPOSE 8080
VOLUME ["/app/data"]
VOLUME ["/app/session"]
CMD ["python", "main.py"]
+17
View File
@@ -0,0 +1,17 @@
services:
telegram-scraper:
build:
context: .
dockerfile: Dockerfile
container_name: telegram-scraper
user: 1000:1000
restart: unless-stopped
tty: true
stdin_open: true
ports:
- "8080:8080"
volumes:
- ./data:/app/data
- session:/app/session
volumes:
session:
+9
View File
@@ -0,0 +1,9 @@
from webui_server import run_server
def main():
run_server()
if __name__ == "__main__":
main()
+23
View File
@@ -0,0 +1,23 @@
[project]
name = "telegram-scraper"
version = "0.1.0"
description = "Telegram-scraper with webui"
readme = "README.md"
requires-python = ">=3.11"
dependencies = [
"aiohappyeyeballs==2.6.1",
"aiohttp==3.12.14",
"aiosignal==1.4.0",
"asyncio==3.4.3",
"attrs==25.3.0",
"frozenlist==1.7.0",
"idna==3.10",
"multidict==6.6.3",
"propcache==0.3.2",
"pyaes==1.6.1",
"pyasn1==0.6.1",
"qrcode==8.0",
"rsa==4.9.1",
"telethon==1.40.0",
"yarl==1.20.1",
]
+177
View File
@@ -0,0 +1,177 @@
const state = {
dashboard: null,
};
async function api(path, options = {}) {
const response = await fetch(path, {
headers: { "Content-Type": "application/json" },
...options,
});
const data = await response.json();
if (!response.ok) {
throw new Error(data.error || "Request failed");
}
return data;
}
function formatDate(value) {
if (!value) return "—";
return value.replace("T", " ").replace("+00:00", " UTC");
}
function setBusy(button, busy) {
if (!button) return;
button.disabled = busy;
}
function renderAuth(auth) {
document.getElementById("auth-status").textContent = auth.status;
document.getElementById("auth-details").textContent = auth.details || "";
}
function renderSummary(data) {
document.getElementById("channel-count").textContent = String(data.state.channel_count);
document.getElementById("forwarding-count").textContent = String(data.state.forwarding_rules.length);
const toggle = document.getElementById("scrape-media-toggle");
toggle.checked = Boolean(data.state.scrape_media);
document.getElementById("scrape-media-label").textContent = toggle.checked ? "ON" : "OFF";
}
function renderChannels(channels) {
const tbody = document.getElementById("channels-table");
tbody.innerHTML = "";
const template = document.getElementById("channel-row-template");
channels.forEach((channel) => {
const node = template.content.firstElementChild.cloneNode(true);
node.querySelector(".channel-name").textContent = channel.name;
node.querySelector(".channel-id").textContent = channel.channel_id;
node.querySelector(".message-count").textContent = String(channel.message_count);
node.querySelector(".media-count").textContent = String(channel.media_count);
node.querySelector(".last-date").textContent = channel.last_date || "—";
node.querySelector(".scrape-btn").addEventListener("click", async () => {
await api("/api/jobs/scrape", {
method: "POST",
body: JSON.stringify({ channel_id: channel.channel_id }),
});
await refreshDashboard();
});
node.querySelector(".export-view-btn").addEventListener("click", () => {
window.location.href = `/viewer?channel=${encodeURIComponent(channel.channel_id)}`;
});
node.querySelector(".media-btn").addEventListener("click", async () => {
await api("/api/jobs/rescrape-media", {
method: "POST",
body: JSON.stringify({ channel_id: channel.channel_id }),
});
await refreshDashboard();
});
node.querySelector(".remove-btn").addEventListener("click", async () => {
await api("/api/channels/remove", {
method: "POST",
body: JSON.stringify({ channel_id: channel.channel_id }),
});
await refreshDashboard();
});
tbody.appendChild(node);
});
}
function renderJobs(jobs) {
const root = document.getElementById("jobs-list");
root.innerHTML = "";
const template = document.getElementById("job-template");
if (!jobs.length) {
root.innerHTML = '<div class="muted">Пока задач нет.</div>';
return;
}
jobs.forEach((job) => {
const node = template.content.firstElementChild.cloneNode(true);
node.querySelector(".job-title").textContent = job.title;
node.querySelector(".job-status").textContent = job.status;
node.querySelector(".job-time").textContent = [job.started_at, job.finished_at].filter(Boolean).join(" -> ");
node.querySelector(".job-logs").textContent = job.logs || job.error || "Без логов";
root.appendChild(node);
});
}
async function refreshDashboard() {
const data = await api("/api/dashboard");
state.dashboard = data;
renderAuth(data.auth);
renderSummary(data);
renderChannels(data.channels);
renderJobs(data.jobs);
}
async function main() {
document.getElementById("scrape-all-btn").addEventListener("click", async (event) => {
setBusy(event.currentTarget, true);
try {
await api("/api/jobs/scrape", { method: "POST", body: JSON.stringify({}) });
await refreshDashboard();
} finally {
setBusy(event.currentTarget, false);
}
});
document.getElementById("export-all-btn").addEventListener("click", async (event) => {
setBusy(event.currentTarget, true);
try {
await api("/api/jobs/export", { method: "POST", body: JSON.stringify({}) });
await refreshDashboard();
} finally {
setBusy(event.currentTarget, false);
}
});
document.getElementById("refresh-dialogs-btn").addEventListener("click", async (event) => {
setBusy(event.currentTarget, true);
try {
await api("/api/jobs/refresh-dialogs", { method: "POST", body: JSON.stringify({}) });
await refreshDashboard();
} finally {
setBusy(event.currentTarget, false);
}
});
document.getElementById("refresh-jobs-btn").addEventListener("click", refreshDashboard);
document.getElementById("scrape-media-toggle").addEventListener("change", async (event) => {
const checked = event.currentTarget.checked;
document.getElementById("scrape-media-label").textContent = checked ? "ON" : "OFF";
await api("/api/settings/media", {
method: "POST",
body: JSON.stringify({ value: checked }),
});
await refreshDashboard();
});
document.getElementById("add-channel-form").addEventListener("submit", async (event) => {
event.preventDefault();
const channelId = document.getElementById("add-channel-id").value.trim();
const name = document.getElementById("add-channel-name").value.trim();
if (!channelId) return;
await api("/api/channels/add", {
method: "POST",
body: JSON.stringify({ channel_id: channelId, name }),
});
event.currentTarget.reset();
await refreshDashboard();
});
await refreshDashboard();
window.setInterval(refreshDashboard, 5000);
}
main().catch((error) => {
console.error(error);
alert(error.message);
});
+131
View File
@@ -0,0 +1,131 @@
<!DOCTYPE html>
<html lang="ru">
<head>
<meta charset="UTF-8" />
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<title>Telegram Scraper Control Panel</title>
<link rel="stylesheet" href="/static/style.css" />
</head>
<body>
<div class="app-shell">
<aside class="sidebar">
<div class="brand-block">
<div class="eyebrow">Telegram Scraper</div>
<h1>Control Panel</h1>
<p class="muted">Обычная админка для сервера: состояния, каналы, фоновые задачи.</p>
</div>
<nav class="nav-links">
<a class="nav-link active" href="/">Панель</a>
<a class="nav-link" href="/viewer">Просмотр сообщений</a>
</nav>
<section class="status-card">
<div class="section-title">Состояние Telegram</div>
<div id="auth-status" class="status-badge">Проверяем...</div>
<p id="auth-details" class="muted small"></p>
</section>
<section class="status-card">
<div class="section-title">Быстрые действия</div>
<button class="button primary" id="scrape-all-btn">Скрапить все каналы</button>
<button class="button" id="export-all-btn">Экспортировать всё</button>
<button class="button" id="refresh-dialogs-btn">Обновить список диалогов</button>
</section>
</aside>
<main class="content">
<section class="panel-grid">
<div class="panel stat-panel">
<div class="section-title">Каналы</div>
<div id="channel-count" class="stat-value">0</div>
</div>
<div class="panel stat-panel">
<div class="section-title">Media scraping</div>
<label class="toggle-row">
<span id="scrape-media-label">OFF</span>
<input id="scrape-media-toggle" type="checkbox" />
</label>
</div>
<div class="panel stat-panel">
<div class="section-title">Forwarding rules</div>
<div id="forwarding-count" class="stat-value">0</div>
</div>
</section>
<section class="panel">
<div class="panel-header">
<div>
<h2>Отслеживаемые каналы</h2>
<p class="muted">Берутся из <code>data/state.json</code>, статистика подтягивается из SQLite.</p>
</div>
<form id="add-channel-form" class="inline-form">
<input id="add-channel-id" name="channel_id" placeholder="ID или @username" required />
<input id="add-channel-name" name="name" placeholder="Псевдоним (необязательно)" />
<button class="button primary" type="submit">Добавить</button>
</form>
</div>
<div class="table-wrap">
<table>
<thead>
<tr>
<th>Канал</th>
<th>Сообщений</th>
<th>Медиа</th>
<th>Последнее сообщение</th>
<th>Действия</th>
</tr>
</thead>
<tbody id="channels-table"></tbody>
</table>
</div>
</section>
<section class="panel jobs-panel">
<div class="panel-header">
<div>
<h2>Фоновые задачи</h2>
<p class="muted">Очередь простая: задачи выполняются по одной, чтобы не драться за Telegram session.</p>
</div>
<button class="button" id="refresh-jobs-btn">Обновить</button>
</div>
<div id="jobs-list" class="jobs-list"></div>
</section>
</main>
</div>
<template id="channel-row-template">
<tr>
<td>
<div class="channel-name"></div>
<div class="muted small channel-id"></div>
</td>
<td class="message-count"></td>
<td class="media-count"></td>
<td class="last-date"></td>
<td>
<div class="action-row">
<button class="button button-small scrape-btn">Scrape</button>
<button class="button button-small export-view-btn">Viewer</button>
<button class="button button-small media-btn">Rescrape media</button>
<button class="button button-small danger remove-btn">Удалить</button>
</div>
</td>
</tr>
</template>
<template id="job-template">
<article class="job-card">
<div class="job-head">
<div class="job-title"></div>
<div class="job-status"></div>
</div>
<div class="muted small job-time"></div>
<pre class="job-logs"></pre>
</article>
</template>
<script src="/static/app.js"></script>
</body>
</html>
+365
View File
@@ -0,0 +1,365 @@
:root {
color-scheme: dark;
--bg: #0f1723;
--panel: #17212b;
--panel-alt: #101922;
--line: #253241;
--soft: #8ea2b5;
--text: #edf3f9;
--accent: #53a7ff;
--accent-strong: #2f8cff;
--danger: #ff6b6b;
--bubble: #182533;
--bubble-self: #1f3a4d;
}
* {
box-sizing: border-box;
}
body {
margin: 0;
font-family: Inter, Arial, sans-serif;
background: var(--bg);
color: var(--text);
}
a {
color: inherit;
text-decoration: none;
}
code,
pre,
input,
button {
font: inherit;
}
.app-shell,
.viewer-shell {
min-height: 100vh;
display: grid;
}
.app-shell {
grid-template-columns: 320px minmax(0, 1fr);
}
.sidebar,
.viewer-sidebar {
background: var(--panel-alt);
border-right: 1px solid var(--line);
padding: 24px;
}
.content,
.viewer-main {
padding: 24px;
}
.brand-block h1,
.viewer-sidebar-head h1,
.viewer-header h2,
.panel-header h2 {
margin: 0;
font-size: 24px;
}
.eyebrow {
color: var(--accent);
font-size: 12px;
text-transform: uppercase;
margin-bottom: 8px;
}
.muted {
color: var(--soft);
}
.small {
font-size: 12px;
}
.nav-links {
display: grid;
gap: 8px;
margin: 24px 0;
}
.nav-link,
.button {
border: 1px solid var(--line);
background: var(--panel);
color: var(--text);
padding: 10px 14px;
border-radius: 8px;
cursor: pointer;
}
.nav-link.active,
.button.primary {
background: var(--accent-strong);
border-color: var(--accent-strong);
}
.button.danger {
color: #ffd0d0;
}
.button.button-small {
padding: 8px 10px;
font-size: 13px;
}
.status-card,
.panel,
.stat-panel {
background: var(--panel);
border: 1px solid var(--line);
border-radius: 8px;
}
.status-card {
padding: 16px;
margin-top: 16px;
}
.section-title {
font-size: 12px;
text-transform: uppercase;
color: var(--soft);
margin-bottom: 12px;
}
.status-badge {
display: inline-flex;
align-items: center;
border-radius: 999px;
padding: 8px 12px;
background: #183248;
color: #a7d8ff;
margin-bottom: 10px;
}
.panel-grid {
display: grid;
gap: 16px;
grid-template-columns: repeat(3, minmax(0, 1fr));
margin-bottom: 16px;
}
.stat-panel {
padding: 18px;
}
.stat-value {
font-size: 32px;
font-weight: 700;
}
.toggle-row {
display: flex;
align-items: center;
justify-content: space-between;
gap: 12px;
}
.panel {
padding: 18px;
margin-bottom: 16px;
}
.panel-header,
.viewer-sidebar-head,
.viewer-header {
display: flex;
align-items: flex-start;
justify-content: space-between;
gap: 16px;
margin-bottom: 16px;
}
.inline-form {
display: flex;
gap: 8px;
flex-wrap: wrap;
}
input {
min-width: 160px;
padding: 10px 12px;
border-radius: 8px;
border: 1px solid var(--line);
background: #0d141c;
color: var(--text);
}
.table-wrap {
overflow-x: auto;
}
table {
width: 100%;
border-collapse: collapse;
}
th,
td {
text-align: left;
border-bottom: 1px solid var(--line);
padding: 14px 10px;
vertical-align: top;
}
tbody tr:hover {
background: rgba(255, 255, 255, 0.02);
}
.action-row {
display: flex;
gap: 8px;
flex-wrap: wrap;
}
.jobs-list {
display: grid;
gap: 12px;
}
.job-card {
border: 1px solid var(--line);
border-radius: 8px;
padding: 14px;
background: #121c25;
}
.job-head {
display: flex;
justify-content: space-between;
gap: 12px;
margin-bottom: 6px;
}
.job-status {
text-transform: uppercase;
font-size: 12px;
color: var(--soft);
}
.job-logs {
margin: 10px 0 0;
padding: 12px;
border-radius: 8px;
background: #0d141c;
max-height: 240px;
overflow: auto;
white-space: pre-wrap;
}
.viewer-shell {
grid-template-columns: 320px minmax(0, 1fr);
}
.viewer-channel-list {
display: grid;
gap: 8px;
}
.viewer-channel-item {
display: grid;
gap: 4px;
width: 100%;
text-align: left;
padding: 12px;
border-radius: 8px;
border: 1px solid var(--line);
background: var(--panel);
color: var(--text);
cursor: pointer;
}
.viewer-channel-item.active {
border-color: var(--accent);
background: #193246;
}
.viewer-channel-name {
font-weight: 600;
}
.viewer-channel-meta {
font-size: 12px;
color: var(--soft);
}
.messages-list {
display: grid;
gap: 12px;
align-content: start;
}
.message-card {
max-width: 760px;
border: 1px solid var(--line);
border-radius: 12px;
padding: 14px;
background: var(--bubble);
}
.message-meta {
display: flex;
justify-content: space-between;
gap: 12px;
margin-bottom: 10px;
}
.message-author {
color: #8fd3ff;
font-weight: 600;
}
.message-reply:empty,
.message-text:empty,
.message-footer:empty,
.message-media:empty {
display: none;
}
.message-text {
white-space: pre-wrap;
line-height: 1.45;
}
.message-media {
margin-top: 12px;
}
.message-media img,
.message-media video {
max-width: 100%;
border-radius: 10px;
border: 1px solid var(--line);
display: block;
}
.message-footer {
margin-top: 10px;
}
@media (max-width: 1100px) {
.app-shell,
.viewer-shell {
grid-template-columns: 1fr;
}
.sidebar,
.viewer-sidebar {
border-right: 0;
border-bottom: 1px solid var(--line);
}
.panel-grid {
grid-template-columns: 1fr;
}
}
+59
View File
@@ -0,0 +1,59 @@
<!DOCTYPE html>
<html lang="ru">
<head>
<meta charset="UTF-8" />
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<title>Telegram Scraper Viewer</title>
<link rel="stylesheet" href="/static/style.css" />
</head>
<body class="viewer-body">
<div class="viewer-shell">
<aside class="viewer-sidebar">
<div class="viewer-sidebar-head">
<div>
<div class="eyebrow">Export Viewer</div>
<h1>Сообщения</h1>
</div>
<a class="button button-small" href="/">Панель</a>
</div>
<div id="viewer-channel-list" class="viewer-channel-list"></div>
</aside>
<main class="viewer-main">
<header class="viewer-header">
<div>
<h2 id="viewer-title">Выберите канал</h2>
<p id="viewer-subtitle" class="muted">Показываем данные из локальной SQLite базы.</p>
</div>
<div class="viewer-actions">
<button class="button" id="load-older-btn" disabled>Загрузить старее</button>
</div>
</header>
<section id="messages-list" class="messages-list"></section>
</main>
</div>
<template id="viewer-channel-template">
<button class="viewer-channel-item">
<span class="viewer-channel-name"></span>
<span class="viewer-channel-meta"></span>
</button>
</template>
<template id="message-template">
<article class="message-card">
<div class="message-meta">
<span class="message-author"></span>
<span class="message-date"></span>
</div>
<div class="message-reply muted small"></div>
<div class="message-text"></div>
<div class="message-media"></div>
<div class="message-footer muted small"></div>
</article>
</template>
<script src="/static/viewer.js"></script>
</body>
</html>
+137
View File
@@ -0,0 +1,137 @@
const viewerState = {
channels: [],
channelId: null,
oldestMessageId: null,
};
async function api(path) {
const response = await fetch(path);
const data = await response.json();
if (!response.ok) {
throw new Error(data.error || "Request failed");
}
return data;
}
function textNodeWithBreaks(text) {
const fragment = document.createDocumentFragment();
const parts = (text || "").split("\n");
parts.forEach((part, index) => {
if (index > 0) fragment.appendChild(document.createElement("br"));
fragment.appendChild(document.createTextNode(part));
});
return fragment;
}
function renderChannelList() {
const root = document.getElementById("viewer-channel-list");
root.innerHTML = "";
const template = document.getElementById("viewer-channel-template");
viewerState.channels.forEach((channel) => {
const node = template.content.firstElementChild.cloneNode(true);
node.querySelector(".viewer-channel-name").textContent = channel.name;
node.querySelector(".viewer-channel-meta").textContent = `${channel.message_count} сообщений`;
if (channel.channel_id === viewerState.channelId) {
node.classList.add("active");
}
node.addEventListener("click", () => {
viewerState.oldestMessageId = null;
loadChannel(channel.channel_id);
});
root.appendChild(node);
});
}
function renderMessages(payload, append = false) {
const root = document.getElementById("messages-list");
const template = document.getElementById("message-template");
if (!append) root.innerHTML = "";
const channel = payload.channel;
document.getElementById("viewer-title").textContent = channel?.name || payload.channel_id;
document.getElementById("viewer-subtitle").textContent = channel
? `${channel.message_count} сообщений, последнее: ${channel.last_date || "—"}`
: "База данных для канала пока не найдена.";
payload.messages.forEach((message) => {
const node = template.content.firstElementChild.cloneNode(true);
node.dataset.messageId = String(message.message_id);
node.querySelector(".message-author").textContent = message.sender_name || "Unknown";
node.querySelector(".message-date").textContent = message.date || "";
const reply = node.querySelector(".message-reply");
if (message.reply_to) {
reply.textContent = `Ответ на сообщение #${message.reply_to}`;
}
node.querySelector(".message-text").appendChild(textNodeWithBreaks(message.text || ""));
const media = node.querySelector(".message-media");
if (message.media_url && message.media_kind === "image") {
const img = document.createElement("img");
img.src = message.media_url;
img.loading = "lazy";
media.appendChild(img);
} else if (message.media_url && message.media_kind === "video") {
const video = document.createElement("video");
video.src = message.media_url;
video.controls = true;
video.preload = "metadata";
media.appendChild(video);
} else if (message.media_url) {
const link = document.createElement("a");
link.href = message.media_url;
link.target = "_blank";
link.rel = "noreferrer";
link.textContent = "Открыть файл";
media.appendChild(link);
}
const footerParts = [];
if (message.views) footerParts.push(`views: ${message.views}`);
if (message.forwards) footerParts.push(`forwards: ${message.forwards}`);
if (message.reactions) footerParts.push(`reactions: ${message.reactions}`);
node.querySelector(".message-footer").textContent = footerParts.join(" | ");
if (append) {
root.prepend(node);
} else {
root.appendChild(node);
}
});
if (payload.messages.length) {
viewerState.oldestMessageId = payload.messages[0].message_id;
}
document.getElementById("load-older-btn").disabled = !payload.messages.length;
}
async function loadChannel(channelId, append = false) {
viewerState.channelId = channelId;
renderChannelList();
const before = append && viewerState.oldestMessageId ? `&before=${viewerState.oldestMessageId}` : "";
const payload = await api(`/api/channels/${encodeURIComponent(channelId)}/messages?limit=80${before}`);
renderMessages(payload, append);
}
async function main() {
viewerState.channels = await api("/api/channels");
const params = new URLSearchParams(window.location.search);
viewerState.channelId = params.get("channel") || viewerState.channels[0]?.channel_id || null;
renderChannelList();
if (viewerState.channelId) {
await loadChannel(viewerState.channelId);
}
document.getElementById("load-older-btn").addEventListener("click", async () => {
if (!viewerState.channelId || !viewerState.oldestMessageId) return;
await loadChannel(viewerState.channelId, true);
});
}
main().catch((error) => {
console.error(error);
alert(error.message);
});
+623
View File
@@ -0,0 +1,623 @@
import asyncio
import contextlib
import io
import json
import mimetypes
import os
import posixpath
import queue
import sqlite3
import threading
import time
import traceback
import urllib.parse
from dataclasses import dataclass, field
from datetime import datetime, timezone
from http import HTTPStatus
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
from typing import Any, Callable, Dict, List, Optional
BASE_DIR = Path(__file__).resolve().parent
DATA_DIR = BASE_DIR / "data"
WEBUI_DIR = BASE_DIR / "webui"
STATE_FILE = DATA_DIR / "state.json"
DEFAULT_HOST = os.environ.get("TELEGRAM_SCRAPER_HOST", "0.0.0.0")
DEFAULT_PORT = int(os.environ.get("TELEGRAM_SCRAPER_PORT", "8080"))
def utc_now_iso() -> str:
return datetime.now(timezone.utc).isoformat()
def load_state() -> Dict[str, Any]:
if STATE_FILE.exists():
with STATE_FILE.open("r", encoding="utf-8") as handle:
return json.load(handle)
return {
"api_id": None,
"api_hash": None,
"channels": {},
"channel_names": {},
"scrape_media": True,
"forwarding_rules": [],
}
def save_state(state: Dict[str, Any]) -> None:
DATA_DIR.mkdir(parents=True, exist_ok=True)
with STATE_FILE.open("w", encoding="utf-8") as handle:
json.dump(state, handle, ensure_ascii=False, indent=2)
def channel_db_path(channel_id: str) -> Path:
return DATA_DIR / channel_id / f"{channel_id}.db"
def guess_media_kind(media_path: Optional[str], media_type: Optional[str]) -> Optional[str]:
if not media_path:
return None
suffix = Path(media_path).suffix.lower()
if suffix in {".jpg", ".jpeg", ".png", ".webp", ".gif"}:
return "image"
if suffix in {".mp4", ".webm", ".mov", ".m4v"}:
return "video"
if suffix in {".mp3", ".ogg", ".wav", ".m4a"}:
return "audio"
if media_type == "MessageMediaPhoto":
return "image"
return "file"
def normalize_media_url(media_path: Optional[str]) -> Optional[str]:
if not media_path:
return None
path = Path(media_path)
if path.is_absolute():
try:
relative = path.resolve().relative_to(DATA_DIR.resolve())
except ValueError:
return None
else:
normalized = Path(media_path)
if normalized.parts and normalized.parts[0] == "data":
normalized = Path(*normalized.parts[1:])
relative = normalized
return "/media/" + urllib.parse.quote(str(relative).replace("\\", "/"))
def database_summary(channel_id: str) -> Dict[str, Any]:
db_path = channel_db_path(channel_id)
summary = {
"message_count": 0,
"last_date": None,
"first_date": None,
"media_count": 0,
"has_database": db_path.exists(),
}
if not db_path.exists():
return summary
conn = sqlite3.connect(str(db_path))
try:
cursor = conn.cursor()
cursor.execute(
"SELECT COUNT(*), MIN(date), MAX(date), "
"SUM(CASE WHEN media_type IS NOT NULL AND media_type != 'MessageMediaWebPage' THEN 1 ELSE 0 END) "
"FROM messages"
)
row = cursor.fetchone() or (0, None, None, 0)
summary.update(
{
"message_count": row[0] or 0,
"first_date": row[1],
"last_date": row[2],
"media_count": row[3] or 0,
}
)
return summary
finally:
conn.close()
def channel_display_name(state: Dict[str, Any], channel_id: str) -> str:
return (
state.get("channel_names", {}).get(channel_id)
or channel_id
)
def list_channels_snapshot() -> List[Dict[str, Any]]:
state = load_state()
channels: List[Dict[str, Any]] = []
for channel_id, last_message_id in state.get("channels", {}).items():
summary = database_summary(channel_id)
channels.append(
{
"channel_id": channel_id,
"name": channel_display_name(state, channel_id),
"last_message_id": last_message_id,
**summary,
}
)
channels.sort(
key=lambda item: (
item["last_date"] or "",
item["message_count"],
),
reverse=True,
)
return channels
def load_messages(channel_id: str, limit: int = 120, before_message_id: Optional[int] = None) -> List[Dict[str, Any]]:
db_path = channel_db_path(channel_id)
if not db_path.exists():
return []
conn = sqlite3.connect(str(db_path))
conn.row_factory = sqlite3.Row
try:
query = (
"SELECT message_id, date, sender_id, first_name, last_name, username, message, "
"media_type, media_path, reply_to, post_author, views, forwards, reactions "
"FROM messages "
)
params: List[Any] = []
if before_message_id is not None:
query += "WHERE message_id < ? "
params.append(before_message_id)
query += "ORDER BY message_id DESC LIMIT ?"
params.append(limit)
rows = conn.execute(query, params).fetchall()
finally:
conn.close()
messages: List[Dict[str, Any]] = []
for row in reversed(rows):
sender_name = " ".join(
part for part in [row["first_name"], row["last_name"]] if part
).strip()
if not sender_name:
sender_name = row["username"] or row["post_author"] or channel_id
messages.append(
{
"message_id": row["message_id"],
"date": row["date"],
"sender_id": row["sender_id"],
"sender_name": sender_name,
"username": row["username"],
"text": row["message"] or "",
"media_type": row["media_type"],
"media_path": row["media_path"],
"media_url": normalize_media_url(row["media_path"]),
"media_kind": guess_media_kind(row["media_path"], row["media_type"]),
"reply_to": row["reply_to"],
"post_author": row["post_author"],
"views": row["views"],
"forwards": row["forwards"],
"reactions": row["reactions"],
}
)
return messages
@dataclass
class Job:
job_id: str
job_type: str
title: str
payload: Dict[str, Any]
status: str = "queued"
created_at: str = field(default_factory=utc_now_iso)
started_at: Optional[str] = None
finished_at: Optional[str] = None
logs: str = ""
error: Optional[str] = None
def to_dict(self) -> Dict[str, Any]:
return {
"job_id": self.job_id,
"job_type": self.job_type,
"title": self.title,
"payload": self.payload,
"status": self.status,
"created_at": self.created_at,
"started_at": self.started_at,
"finished_at": self.finished_at,
"logs": self.logs,
"error": self.error,
}
class JobRunner:
def __init__(self) -> None:
self.jobs: Dict[str, Job] = {}
self.job_order: List[str] = []
self.queue: "queue.Queue[Job]" = queue.Queue()
self.lock = threading.Lock()
self.worker = threading.Thread(target=self._run, daemon=True)
self.worker.start()
def create_job(self, job_type: str, title: str, payload: Dict[str, Any]) -> Job:
job_id = f"job-{int(time.time() * 1000)}"
job = Job(job_id=job_id, job_type=job_type, title=title, payload=payload)
with self.lock:
self.jobs[job_id] = job
self.job_order.insert(0, job_id)
self.job_order = self.job_order[:20]
self.queue.put(job)
return job
def recent_jobs(self) -> List[Dict[str, Any]]:
with self.lock:
return [self.jobs[job_id].to_dict() for job_id in self.job_order]
def get_job(self, job_id: str) -> Optional[Dict[str, Any]]:
with self.lock:
job = self.jobs.get(job_id)
return job.to_dict() if job else None
def _run(self) -> None:
while True:
job = self.queue.get()
self._execute(job)
self.queue.task_done()
def _execute(self, job: Job) -> None:
with self.lock:
job.status = "running"
job.started_at = utc_now_iso()
buffer = io.StringIO()
try:
with contextlib.redirect_stdout(buffer), contextlib.redirect_stderr(buffer):
run_job(job.job_type, job.payload)
with self.lock:
job.status = "done"
except Exception as exc:
with self.lock:
job.status = "failed"
job.error = str(exc)
traceback.print_exc(file=buffer)
finally:
with self.lock:
job.logs = buffer.getvalue()
job.finished_at = utc_now_iso()
def import_scraper_class():
from telegram_scraper_with_forwarding import OptimizedTelegramScraper
return OptimizedTelegramScraper
def run_job(job_type: str, payload: Dict[str, Any]) -> None:
if job_type == "set_scrape_media":
state = load_state()
state["scrape_media"] = bool(payload["value"])
save_state(state)
print(f"Media scraping set to {state['scrape_media']}")
return
async def _async_job() -> None:
ScraperClass = import_scraper_class()
scraper = ScraperClass()
initialized = await scraper.initialize_client(interactive=False)
if not initialized:
raise RuntimeError("Telegram client is not ready. Check credentials and session.")
try:
if job_type == "scrape_channel":
channel_id = payload["channel_id"]
state = scraper.load_state()
offset = int(state.get("channels", {}).get(channel_id, 0) or 0)
await scraper.scrape_channel(channel_id, offset)
elif job_type == "scrape_all":
channels = list(scraper.state.get("channels", {}).keys())
for channel_id in channels:
offset = int(scraper.state.get("channels", {}).get(channel_id, 0) or 0)
await scraper.scrape_channel(channel_id, offset)
elif job_type == "export_all":
await scraper.export_data()
elif job_type == "rescrape_media":
await scraper.rescrape_media(payload["channel_id"])
elif job_type == "fix_missing_media":
await scraper.fix_missing_media(payload["channel_id"])
elif job_type == "refresh_dialogs":
await scraper.list_channels()
else:
raise RuntimeError(f"Unsupported job type: {job_type}")
finally:
scraper.close_db_connections()
if scraper.client:
await scraper.client.disconnect()
asyncio.run(_async_job())
def auth_status() -> Dict[str, Any]:
state = load_state()
status = {
"has_api_credentials": bool(state.get("api_id") and state.get("api_hash")),
"telethon_available": False,
"session_ready": False,
"status": "read_only",
"details": "",
}
try:
import_scraper_class()
status["telethon_available"] = True
except Exception as exc:
status["details"] = f"Python deps are missing: {exc}"
return status
session_file = BASE_DIR / "session" / "session.session"
status["session_ready"] = session_file.exists()
if status["has_api_credentials"] and status["session_ready"]:
status["status"] = "ready"
status["details"] = "Scraping actions should be available after dependencies are installed."
elif status["has_api_credentials"]:
status["status"] = "needs_auth"
status["details"] = "API keys found, but Telegram session file is missing."
else:
status["status"] = "needs_credentials"
status["details"] = "api_id/api_hash are missing in data/state.json."
return status
def dashboard_payload(job_runner: JobRunner) -> Dict[str, Any]:
state = load_state()
channels = list_channels_snapshot()
return {
"state": {
"scrape_media": bool(state.get("scrape_media", True)),
"channel_count": len(state.get("channels", {})),
"forwarding_rules": state.get("forwarding_rules", []),
},
"auth": auth_status(),
"channels": channels,
"jobs": job_runner.recent_jobs(),
}
class TelegramScraperRequestHandler(BaseHTTPRequestHandler):
server_version = "TelegramScraperWebUI/0.1"
@property
def app(self) -> "TelegramScraperWebServer":
return self.server # type: ignore[return-value]
def do_GET(self) -> None:
parsed = urllib.parse.urlparse(self.path)
path = parsed.path
query = urllib.parse.parse_qs(parsed.query)
if path == "/":
return self.serve_file(WEBUI_DIR / "index.html", "text/html; charset=utf-8")
if path == "/viewer":
return self.serve_file(WEBUI_DIR / "viewer.html", "text/html; charset=utf-8")
if path.startswith("/static/"):
relative = path.removeprefix("/static/")
return self.serve_static(relative)
if path.startswith("/media/"):
relative = path.removeprefix("/media/")
return self.serve_media(relative)
if path == "/api/dashboard":
return self.send_json(dashboard_payload(self.app.job_runner))
if path == "/api/jobs":
return self.send_json(self.app.job_runner.recent_jobs())
if path.startswith("/api/jobs/"):
job_id = path.rsplit("/", 1)[-1]
job = self.app.job_runner.get_job(job_id)
if not job:
return self.send_error_json(HTTPStatus.NOT_FOUND, "Job not found")
return self.send_json(job)
if path == "/api/channels":
return self.send_json(list_channels_snapshot())
if path.startswith("/api/channels/") and path.endswith("/messages"):
parts = path.split("/")
channel_id = parts[3]
limit = max(1, min(int(query.get("limit", ["120"])[0]), 300))
before = query.get("before")
before_message_id = int(before[0]) if before else None
payload = {
"channel_id": channel_id,
"messages": load_messages(channel_id, limit=limit, before_message_id=before_message_id),
"channel": next(
(item for item in list_channels_snapshot() if item["channel_id"] == channel_id),
None,
),
}
return self.send_json(payload)
return self.send_error_json(HTTPStatus.NOT_FOUND, "Not found")
def do_HEAD(self) -> None:
parsed = urllib.parse.urlparse(self.path)
path = parsed.path
if path == "/":
return self.serve_file(WEBUI_DIR / "index.html", "text/html; charset=utf-8", head_only=True)
if path == "/viewer":
return self.serve_file(WEBUI_DIR / "viewer.html", "text/html; charset=utf-8", head_only=True)
if path.startswith("/static/"):
relative = path.removeprefix("/static/")
return self.serve_static(relative, head_only=True)
if path.startswith("/media/"):
relative = path.removeprefix("/media/")
return self.serve_media(relative, head_only=True)
if path in {"/api/dashboard", "/api/jobs", "/api/channels"} or path.startswith("/api/jobs/") or (
path.startswith("/api/channels/") and path.endswith("/messages")
):
self.send_response(HTTPStatus.OK)
self.send_header("Content-Type", "application/json; charset=utf-8")
self.end_headers()
return
return self.send_error_json(HTTPStatus.NOT_FOUND, "Not found")
def do_POST(self) -> None:
parsed = urllib.parse.urlparse(self.path)
path = parsed.path
body = self.read_json_body()
if body is None:
return self.send_error_json(HTTPStatus.BAD_REQUEST, "Expected JSON body")
if path == "/api/settings/media":
value = bool(body.get("value"))
job = self.app.job_runner.create_job(
"set_scrape_media",
"Update media scraping setting",
{"value": value},
)
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
if path == "/api/channels/add":
channel_id = str(body.get("channel_id", "")).strip()
if not channel_id:
return self.send_error_json(HTTPStatus.BAD_REQUEST, "channel_id is required")
state = load_state()
if channel_id not in state["channels"]:
state["channels"][channel_id] = 0
if body.get("name"):
state.setdefault("channel_names", {})[channel_id] = str(body["name"]).strip()
save_state(state)
return self.send_json({"ok": True, "channel_id": channel_id})
if path == "/api/channels/remove":
channel_id = str(body.get("channel_id", "")).strip()
state = load_state()
existed = channel_id in state.get("channels", {})
state.get("channels", {}).pop(channel_id, None)
save_state(state)
return self.send_json({"ok": existed, "channel_id": channel_id})
if path == "/api/jobs/scrape":
channel_id = body.get("channel_id")
if channel_id:
job = self.app.job_runner.create_job(
"scrape_channel",
f"Scrape channel {channel_id}",
{"channel_id": str(channel_id)},
)
else:
job = self.app.job_runner.create_job(
"scrape_all",
"Scrape all tracked channels",
{},
)
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
if path == "/api/jobs/export":
job = self.app.job_runner.create_job(
"export_all",
"Export all tracked channels",
{},
)
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
if path == "/api/jobs/rescrape-media":
channel_id = str(body.get("channel_id", "")).strip()
if not channel_id:
return self.send_error_json(HTTPStatus.BAD_REQUEST, "channel_id is required")
job = self.app.job_runner.create_job(
"rescrape_media",
f"Rescrape media for {channel_id}",
{"channel_id": channel_id},
)
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
if path == "/api/jobs/fix-media":
channel_id = str(body.get("channel_id", "")).strip()
if not channel_id:
return self.send_error_json(HTTPStatus.BAD_REQUEST, "channel_id is required")
job = self.app.job_runner.create_job(
"fix_missing_media",
f"Fix missing media for {channel_id}",
{"channel_id": channel_id},
)
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
if path == "/api/jobs/refresh-dialogs":
job = self.app.job_runner.create_job(
"refresh_dialogs",
"Refresh Telegram dialogs list",
{},
)
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
return self.send_error_json(HTTPStatus.NOT_FOUND, "Not found")
def read_json_body(self) -> Optional[Dict[str, Any]]:
length = int(self.headers.get("Content-Length", "0"))
if length <= 0:
return {}
raw = self.rfile.read(length)
try:
return json.loads(raw.decode("utf-8"))
except json.JSONDecodeError:
return None
def serve_static(self, relative_path: str, head_only: bool = False) -> None:
file_path = (WEBUI_DIR / relative_path).resolve()
try:
file_path.relative_to(WEBUI_DIR.resolve())
except ValueError:
return self.send_error_json(HTTPStatus.FORBIDDEN, "Forbidden")
if not file_path.exists() or not file_path.is_file():
return self.send_error_json(HTTPStatus.NOT_FOUND, "File not found")
mime_type = mimetypes.guess_type(str(file_path))[0] or "application/octet-stream"
return self.serve_file(file_path, mime_type, head_only=head_only)
def serve_media(self, relative_path: str, head_only: bool = False) -> None:
clean = Path(posixpath.normpath(urllib.parse.unquote(relative_path)))
file_path = (DATA_DIR / clean).resolve()
try:
file_path.relative_to(DATA_DIR.resolve())
except ValueError:
return self.send_error_json(HTTPStatus.FORBIDDEN, "Forbidden")
if not file_path.exists() or not file_path.is_file():
return self.send_error_json(HTTPStatus.NOT_FOUND, "Media file not found")
mime_type = mimetypes.guess_type(str(file_path))[0] or "application/octet-stream"
return self.serve_file(file_path, mime_type, head_only=head_only)
def serve_file(self, file_path: Path, content_type: str, head_only: bool = False) -> None:
data = file_path.read_bytes()
self.send_response(HTTPStatus.OK)
self.send_header("Content-Type", content_type)
self.send_header("Content-Length", str(len(data)))
self.end_headers()
if not head_only:
self.wfile.write(data)
def send_json(self, payload: Any, status: HTTPStatus = HTTPStatus.OK) -> None:
raw = json.dumps(payload, ensure_ascii=False).encode("utf-8")
self.send_response(status)
self.send_header("Content-Type", "application/json; charset=utf-8")
self.send_header("Content-Length", str(len(raw)))
self.end_headers()
self.wfile.write(raw)
def send_error_json(self, status: HTTPStatus, message: str) -> None:
self.send_json({"error": message, "status": status}, status=status)
def log_message(self, format: str, *args: Any) -> None:
return
class TelegramScraperWebServer(ThreadingHTTPServer):
def __init__(self, server_address: tuple[str, int]):
super().__init__(server_address, TelegramScraperRequestHandler)
self.job_runner = JobRunner()
def run_server(host: str = DEFAULT_HOST, port: int = DEFAULT_PORT) -> None:
server = TelegramScraperWebServer((host, port))
print(f"Telegram Scraper Web UI listening on http://{host}:{port}")
print("Press Ctrl+C to stop.")
try:
server.serve_forever()
except KeyboardInterrupt:
pass
finally:
server.server_close()
if __name__ == "__main__":
run_server()