Add lightweight web control panel
This commit is contained in:
+227
@@ -0,0 +1,227 @@
|
||||
data/
|
||||
.codex
|
||||
.python-version
|
||||
uv.lock
|
||||
.uv-cache/
|
||||
|
||||
|
||||
# Byte-compiled / optimized / DLL files
|
||||
__pycache__/
|
||||
*.py[codz]
|
||||
*$py.class
|
||||
|
||||
# C extensions
|
||||
*.so
|
||||
|
||||
# Distribution / packaging
|
||||
.Python
|
||||
build/
|
||||
develop-eggs/
|
||||
dist/
|
||||
downloads/
|
||||
eggs/
|
||||
.eggs/
|
||||
lib/
|
||||
lib64/
|
||||
parts/
|
||||
sdist/
|
||||
var/
|
||||
wheels/
|
||||
share/python-wheels/
|
||||
*.egg-info/
|
||||
.installed.cfg
|
||||
*.egg
|
||||
MANIFEST
|
||||
|
||||
# PyInstaller
|
||||
# Usually these files are written by a python script from a template
|
||||
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
||||
*.manifest
|
||||
*.spec
|
||||
|
||||
# Installer logs
|
||||
pip-log.txt
|
||||
pip-delete-this-directory.txt
|
||||
|
||||
# Unit test / coverage reports
|
||||
htmlcov/
|
||||
.tox/
|
||||
.nox/
|
||||
.coverage
|
||||
.coverage.*
|
||||
.cache
|
||||
nosetests.xml
|
||||
coverage.xml
|
||||
*.cover
|
||||
*.py.cover
|
||||
*.lcov
|
||||
.hypothesis/
|
||||
.pytest_cache/
|
||||
cover/
|
||||
|
||||
# Translations
|
||||
*.mo
|
||||
*.pot
|
||||
|
||||
# Django stuff:
|
||||
*.log
|
||||
local_settings.py
|
||||
db.sqlite3
|
||||
db.sqlite3-journal
|
||||
|
||||
# Flask stuff:
|
||||
instance/
|
||||
.webassets-cache
|
||||
|
||||
# Scrapy stuff:
|
||||
.scrapy
|
||||
|
||||
# Sphinx documentation
|
||||
docs/_build/
|
||||
|
||||
# PyBuilder
|
||||
.pybuilder/
|
||||
target/
|
||||
|
||||
# Jupyter Notebook
|
||||
.ipynb_checkpoints
|
||||
|
||||
# IPython
|
||||
profile_default/
|
||||
ipython_config.py
|
||||
|
||||
# pyenv
|
||||
# For a library or package, you might want to ignore these files since the code is
|
||||
# intended to run in multiple environments; otherwise, check them in:
|
||||
# .python-version
|
||||
|
||||
# pipenv
|
||||
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
|
||||
# However, in case of collaboration, if having platform-specific dependencies or dependencies
|
||||
# having no cross-platform support, pipenv may install dependencies that don't work, or not
|
||||
# install all needed dependencies.
|
||||
# Pipfile.lock
|
||||
|
||||
# UV
|
||||
# Similar to Pipfile.lock, it is generally recommended to include uv.lock in version control.
|
||||
# This is especially recommended for binary packages to ensure reproducibility, and is more
|
||||
# commonly ignored for libraries.
|
||||
# uv.lock
|
||||
|
||||
# poetry
|
||||
# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
|
||||
# This is especially recommended for binary packages to ensure reproducibility, and is more
|
||||
# commonly ignored for libraries.
|
||||
# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
|
||||
# poetry.lock
|
||||
# poetry.toml
|
||||
|
||||
# pdm
|
||||
# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
|
||||
# pdm recommends including project-wide configuration in pdm.toml, but excluding .pdm-python.
|
||||
# https://pdm-project.org/en/latest/usage/project/#working-with-version-control
|
||||
# pdm.lock
|
||||
# pdm.toml
|
||||
.pdm-python
|
||||
.pdm-build/
|
||||
|
||||
# pixi
|
||||
# Similar to Pipfile.lock, it is generally recommended to include pixi.lock in version control.
|
||||
# pixi.lock
|
||||
# Pixi creates a virtual environment in the .pixi directory, just like venv module creates one
|
||||
# in the .venv directory. It is recommended not to include this directory in version control.
|
||||
.pixi/*
|
||||
!.pixi/config.toml
|
||||
|
||||
# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
|
||||
__pypackages__/
|
||||
|
||||
# Celery stuff
|
||||
celerybeat-schedule*
|
||||
celerybeat.pid
|
||||
|
||||
# Redis
|
||||
*.rdb
|
||||
*.aof
|
||||
*.pid
|
||||
|
||||
# RabbitMQ
|
||||
mnesia/
|
||||
rabbitmq/
|
||||
rabbitmq-data/
|
||||
|
||||
# ActiveMQ
|
||||
activemq-data/
|
||||
|
||||
# SageMath parsed files
|
||||
*.sage.py
|
||||
|
||||
# Environments
|
||||
.env
|
||||
.envrc
|
||||
.venv
|
||||
env/
|
||||
venv/
|
||||
ENV/
|
||||
env.bak/
|
||||
venv.bak/
|
||||
|
||||
# Spyder project settings
|
||||
.spyderproject
|
||||
.spyproject
|
||||
|
||||
# Rope project settings
|
||||
.ropeproject
|
||||
|
||||
# mkdocs documentation
|
||||
/site
|
||||
|
||||
# mypy
|
||||
.mypy_cache/
|
||||
.dmypy.json
|
||||
dmypy.json
|
||||
|
||||
# Pyre type checker
|
||||
.pyre/
|
||||
|
||||
# pytype static type analyzer
|
||||
.pytype/
|
||||
|
||||
# Cython debug symbols
|
||||
cython_debug/
|
||||
|
||||
# PyCharm
|
||||
# JetBrains specific template is maintained in a separate JetBrains.gitignore that can
|
||||
# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
|
||||
# and can be added to the global gitignore or merged into this file. For a more nuclear
|
||||
# option (not recommended) you can uncomment the following to ignore the entire idea folder.
|
||||
# .idea/
|
||||
|
||||
# Abstra
|
||||
# Abstra is an AI-powered process automation framework.
|
||||
# Ignore directories containing user credentials, local state, and settings.
|
||||
# Learn more at https://abstra.io/docs
|
||||
.abstra/
|
||||
|
||||
# Visual Studio Code
|
||||
# Visual Studio Code specific template is maintained in a separate VisualStudioCode.gitignore
|
||||
# that can be found at https://github.com/github/gitignore/blob/main/Global/VisualStudioCode.gitignore
|
||||
# and can be added to the global gitignore or merged into this file. However, if you prefer,
|
||||
# you could uncomment the following to ignore the entire vscode folder
|
||||
# .vscode/
|
||||
# Temporary file for partial code execution
|
||||
tempCodeRunnerFile.py
|
||||
|
||||
# Ruff stuff:
|
||||
.ruff_cache/
|
||||
|
||||
# PyPI configuration file
|
||||
.pypirc
|
||||
|
||||
# Marimo
|
||||
marimo/_static/
|
||||
marimo/_lsp/
|
||||
__marimo__/
|
||||
|
||||
# Streamlit
|
||||
.streamlit/secrets.toml
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
FROM python:3.11-slim
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
gcc \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY requirements.txt .
|
||||
RUN pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
COPY . .
|
||||
|
||||
EXPOSE 8080
|
||||
VOLUME ["/app/data"]
|
||||
VOLUME ["/app/session"]
|
||||
|
||||
CMD ["python", "main.py"]
|
||||
@@ -0,0 +1,17 @@
|
||||
services:
|
||||
telegram-scraper:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile
|
||||
container_name: telegram-scraper
|
||||
user: 1000:1000
|
||||
restart: unless-stopped
|
||||
tty: true
|
||||
stdin_open: true
|
||||
ports:
|
||||
- "8080:8080"
|
||||
volumes:
|
||||
- ./data:/app/data
|
||||
- session:/app/session
|
||||
volumes:
|
||||
session:
|
||||
@@ -0,0 +1,9 @@
|
||||
from webui_server import run_server
|
||||
|
||||
|
||||
def main():
|
||||
run_server()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,23 @@
|
||||
[project]
|
||||
name = "telegram-scraper"
|
||||
version = "0.1.0"
|
||||
description = "Telegram-scraper with webui"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.11"
|
||||
dependencies = [
|
||||
"aiohappyeyeballs==2.6.1",
|
||||
"aiohttp==3.12.14",
|
||||
"aiosignal==1.4.0",
|
||||
"asyncio==3.4.3",
|
||||
"attrs==25.3.0",
|
||||
"frozenlist==1.7.0",
|
||||
"idna==3.10",
|
||||
"multidict==6.6.3",
|
||||
"propcache==0.3.2",
|
||||
"pyaes==1.6.1",
|
||||
"pyasn1==0.6.1",
|
||||
"qrcode==8.0",
|
||||
"rsa==4.9.1",
|
||||
"telethon==1.40.0",
|
||||
"yarl==1.20.1",
|
||||
]
|
||||
+177
@@ -0,0 +1,177 @@
|
||||
const state = {
|
||||
dashboard: null,
|
||||
};
|
||||
|
||||
async function api(path, options = {}) {
|
||||
const response = await fetch(path, {
|
||||
headers: { "Content-Type": "application/json" },
|
||||
...options,
|
||||
});
|
||||
const data = await response.json();
|
||||
if (!response.ok) {
|
||||
throw new Error(data.error || "Request failed");
|
||||
}
|
||||
return data;
|
||||
}
|
||||
|
||||
function formatDate(value) {
|
||||
if (!value) return "—";
|
||||
return value.replace("T", " ").replace("+00:00", " UTC");
|
||||
}
|
||||
|
||||
function setBusy(button, busy) {
|
||||
if (!button) return;
|
||||
button.disabled = busy;
|
||||
}
|
||||
|
||||
function renderAuth(auth) {
|
||||
document.getElementById("auth-status").textContent = auth.status;
|
||||
document.getElementById("auth-details").textContent = auth.details || "";
|
||||
}
|
||||
|
||||
function renderSummary(data) {
|
||||
document.getElementById("channel-count").textContent = String(data.state.channel_count);
|
||||
document.getElementById("forwarding-count").textContent = String(data.state.forwarding_rules.length);
|
||||
const toggle = document.getElementById("scrape-media-toggle");
|
||||
toggle.checked = Boolean(data.state.scrape_media);
|
||||
document.getElementById("scrape-media-label").textContent = toggle.checked ? "ON" : "OFF";
|
||||
}
|
||||
|
||||
function renderChannels(channels) {
|
||||
const tbody = document.getElementById("channels-table");
|
||||
tbody.innerHTML = "";
|
||||
const template = document.getElementById("channel-row-template");
|
||||
|
||||
channels.forEach((channel) => {
|
||||
const node = template.content.firstElementChild.cloneNode(true);
|
||||
node.querySelector(".channel-name").textContent = channel.name;
|
||||
node.querySelector(".channel-id").textContent = channel.channel_id;
|
||||
node.querySelector(".message-count").textContent = String(channel.message_count);
|
||||
node.querySelector(".media-count").textContent = String(channel.media_count);
|
||||
node.querySelector(".last-date").textContent = channel.last_date || "—";
|
||||
|
||||
node.querySelector(".scrape-btn").addEventListener("click", async () => {
|
||||
await api("/api/jobs/scrape", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ channel_id: channel.channel_id }),
|
||||
});
|
||||
await refreshDashboard();
|
||||
});
|
||||
|
||||
node.querySelector(".export-view-btn").addEventListener("click", () => {
|
||||
window.location.href = `/viewer?channel=${encodeURIComponent(channel.channel_id)}`;
|
||||
});
|
||||
|
||||
node.querySelector(".media-btn").addEventListener("click", async () => {
|
||||
await api("/api/jobs/rescrape-media", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ channel_id: channel.channel_id }),
|
||||
});
|
||||
await refreshDashboard();
|
||||
});
|
||||
|
||||
node.querySelector(".remove-btn").addEventListener("click", async () => {
|
||||
await api("/api/channels/remove", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ channel_id: channel.channel_id }),
|
||||
});
|
||||
await refreshDashboard();
|
||||
});
|
||||
|
||||
tbody.appendChild(node);
|
||||
});
|
||||
}
|
||||
|
||||
function renderJobs(jobs) {
|
||||
const root = document.getElementById("jobs-list");
|
||||
root.innerHTML = "";
|
||||
const template = document.getElementById("job-template");
|
||||
|
||||
if (!jobs.length) {
|
||||
root.innerHTML = '<div class="muted">Пока задач нет.</div>';
|
||||
return;
|
||||
}
|
||||
|
||||
jobs.forEach((job) => {
|
||||
const node = template.content.firstElementChild.cloneNode(true);
|
||||
node.querySelector(".job-title").textContent = job.title;
|
||||
node.querySelector(".job-status").textContent = job.status;
|
||||
node.querySelector(".job-time").textContent = [job.started_at, job.finished_at].filter(Boolean).join(" -> ");
|
||||
node.querySelector(".job-logs").textContent = job.logs || job.error || "Без логов";
|
||||
root.appendChild(node);
|
||||
});
|
||||
}
|
||||
|
||||
async function refreshDashboard() {
|
||||
const data = await api("/api/dashboard");
|
||||
state.dashboard = data;
|
||||
renderAuth(data.auth);
|
||||
renderSummary(data);
|
||||
renderChannels(data.channels);
|
||||
renderJobs(data.jobs);
|
||||
}
|
||||
|
||||
async function main() {
|
||||
document.getElementById("scrape-all-btn").addEventListener("click", async (event) => {
|
||||
setBusy(event.currentTarget, true);
|
||||
try {
|
||||
await api("/api/jobs/scrape", { method: "POST", body: JSON.stringify({}) });
|
||||
await refreshDashboard();
|
||||
} finally {
|
||||
setBusy(event.currentTarget, false);
|
||||
}
|
||||
});
|
||||
|
||||
document.getElementById("export-all-btn").addEventListener("click", async (event) => {
|
||||
setBusy(event.currentTarget, true);
|
||||
try {
|
||||
await api("/api/jobs/export", { method: "POST", body: JSON.stringify({}) });
|
||||
await refreshDashboard();
|
||||
} finally {
|
||||
setBusy(event.currentTarget, false);
|
||||
}
|
||||
});
|
||||
|
||||
document.getElementById("refresh-dialogs-btn").addEventListener("click", async (event) => {
|
||||
setBusy(event.currentTarget, true);
|
||||
try {
|
||||
await api("/api/jobs/refresh-dialogs", { method: "POST", body: JSON.stringify({}) });
|
||||
await refreshDashboard();
|
||||
} finally {
|
||||
setBusy(event.currentTarget, false);
|
||||
}
|
||||
});
|
||||
|
||||
document.getElementById("refresh-jobs-btn").addEventListener("click", refreshDashboard);
|
||||
|
||||
document.getElementById("scrape-media-toggle").addEventListener("change", async (event) => {
|
||||
const checked = event.currentTarget.checked;
|
||||
document.getElementById("scrape-media-label").textContent = checked ? "ON" : "OFF";
|
||||
await api("/api/settings/media", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ value: checked }),
|
||||
});
|
||||
await refreshDashboard();
|
||||
});
|
||||
|
||||
document.getElementById("add-channel-form").addEventListener("submit", async (event) => {
|
||||
event.preventDefault();
|
||||
const channelId = document.getElementById("add-channel-id").value.trim();
|
||||
const name = document.getElementById("add-channel-name").value.trim();
|
||||
if (!channelId) return;
|
||||
await api("/api/channels/add", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ channel_id: channelId, name }),
|
||||
});
|
||||
event.currentTarget.reset();
|
||||
await refreshDashboard();
|
||||
});
|
||||
|
||||
await refreshDashboard();
|
||||
window.setInterval(refreshDashboard, 5000);
|
||||
}
|
||||
|
||||
main().catch((error) => {
|
||||
console.error(error);
|
||||
alert(error.message);
|
||||
});
|
||||
@@ -0,0 +1,131 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="ru">
|
||||
<head>
|
||||
<meta charset="UTF-8" />
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
||||
<title>Telegram Scraper Control Panel</title>
|
||||
<link rel="stylesheet" href="/static/style.css" />
|
||||
</head>
|
||||
<body>
|
||||
<div class="app-shell">
|
||||
<aside class="sidebar">
|
||||
<div class="brand-block">
|
||||
<div class="eyebrow">Telegram Scraper</div>
|
||||
<h1>Control Panel</h1>
|
||||
<p class="muted">Обычная админка для сервера: состояния, каналы, фоновые задачи.</p>
|
||||
</div>
|
||||
|
||||
<nav class="nav-links">
|
||||
<a class="nav-link active" href="/">Панель</a>
|
||||
<a class="nav-link" href="/viewer">Просмотр сообщений</a>
|
||||
</nav>
|
||||
|
||||
<section class="status-card">
|
||||
<div class="section-title">Состояние Telegram</div>
|
||||
<div id="auth-status" class="status-badge">Проверяем...</div>
|
||||
<p id="auth-details" class="muted small"></p>
|
||||
</section>
|
||||
|
||||
<section class="status-card">
|
||||
<div class="section-title">Быстрые действия</div>
|
||||
<button class="button primary" id="scrape-all-btn">Скрапить все каналы</button>
|
||||
<button class="button" id="export-all-btn">Экспортировать всё</button>
|
||||
<button class="button" id="refresh-dialogs-btn">Обновить список диалогов</button>
|
||||
</section>
|
||||
</aside>
|
||||
|
||||
<main class="content">
|
||||
<section class="panel-grid">
|
||||
<div class="panel stat-panel">
|
||||
<div class="section-title">Каналы</div>
|
||||
<div id="channel-count" class="stat-value">0</div>
|
||||
</div>
|
||||
<div class="panel stat-panel">
|
||||
<div class="section-title">Media scraping</div>
|
||||
<label class="toggle-row">
|
||||
<span id="scrape-media-label">OFF</span>
|
||||
<input id="scrape-media-toggle" type="checkbox" />
|
||||
</label>
|
||||
</div>
|
||||
<div class="panel stat-panel">
|
||||
<div class="section-title">Forwarding rules</div>
|
||||
<div id="forwarding-count" class="stat-value">0</div>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section class="panel">
|
||||
<div class="panel-header">
|
||||
<div>
|
||||
<h2>Отслеживаемые каналы</h2>
|
||||
<p class="muted">Берутся из <code>data/state.json</code>, статистика подтягивается из SQLite.</p>
|
||||
</div>
|
||||
<form id="add-channel-form" class="inline-form">
|
||||
<input id="add-channel-id" name="channel_id" placeholder="ID или @username" required />
|
||||
<input id="add-channel-name" name="name" placeholder="Псевдоним (необязательно)" />
|
||||
<button class="button primary" type="submit">Добавить</button>
|
||||
</form>
|
||||
</div>
|
||||
|
||||
<div class="table-wrap">
|
||||
<table>
|
||||
<thead>
|
||||
<tr>
|
||||
<th>Канал</th>
|
||||
<th>Сообщений</th>
|
||||
<th>Медиа</th>
|
||||
<th>Последнее сообщение</th>
|
||||
<th>Действия</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody id="channels-table"></tbody>
|
||||
</table>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section class="panel jobs-panel">
|
||||
<div class="panel-header">
|
||||
<div>
|
||||
<h2>Фоновые задачи</h2>
|
||||
<p class="muted">Очередь простая: задачи выполняются по одной, чтобы не драться за Telegram session.</p>
|
||||
</div>
|
||||
<button class="button" id="refresh-jobs-btn">Обновить</button>
|
||||
</div>
|
||||
<div id="jobs-list" class="jobs-list"></div>
|
||||
</section>
|
||||
</main>
|
||||
</div>
|
||||
|
||||
<template id="channel-row-template">
|
||||
<tr>
|
||||
<td>
|
||||
<div class="channel-name"></div>
|
||||
<div class="muted small channel-id"></div>
|
||||
</td>
|
||||
<td class="message-count"></td>
|
||||
<td class="media-count"></td>
|
||||
<td class="last-date"></td>
|
||||
<td>
|
||||
<div class="action-row">
|
||||
<button class="button button-small scrape-btn">Scrape</button>
|
||||
<button class="button button-small export-view-btn">Viewer</button>
|
||||
<button class="button button-small media-btn">Rescrape media</button>
|
||||
<button class="button button-small danger remove-btn">Удалить</button>
|
||||
</div>
|
||||
</td>
|
||||
</tr>
|
||||
</template>
|
||||
|
||||
<template id="job-template">
|
||||
<article class="job-card">
|
||||
<div class="job-head">
|
||||
<div class="job-title"></div>
|
||||
<div class="job-status"></div>
|
||||
</div>
|
||||
<div class="muted small job-time"></div>
|
||||
<pre class="job-logs"></pre>
|
||||
</article>
|
||||
</template>
|
||||
|
||||
<script src="/static/app.js"></script>
|
||||
</body>
|
||||
</html>
|
||||
+365
@@ -0,0 +1,365 @@
|
||||
:root {
|
||||
color-scheme: dark;
|
||||
--bg: #0f1723;
|
||||
--panel: #17212b;
|
||||
--panel-alt: #101922;
|
||||
--line: #253241;
|
||||
--soft: #8ea2b5;
|
||||
--text: #edf3f9;
|
||||
--accent: #53a7ff;
|
||||
--accent-strong: #2f8cff;
|
||||
--danger: #ff6b6b;
|
||||
--bubble: #182533;
|
||||
--bubble-self: #1f3a4d;
|
||||
}
|
||||
|
||||
* {
|
||||
box-sizing: border-box;
|
||||
}
|
||||
|
||||
body {
|
||||
margin: 0;
|
||||
font-family: Inter, Arial, sans-serif;
|
||||
background: var(--bg);
|
||||
color: var(--text);
|
||||
}
|
||||
|
||||
a {
|
||||
color: inherit;
|
||||
text-decoration: none;
|
||||
}
|
||||
|
||||
code,
|
||||
pre,
|
||||
input,
|
||||
button {
|
||||
font: inherit;
|
||||
}
|
||||
|
||||
.app-shell,
|
||||
.viewer-shell {
|
||||
min-height: 100vh;
|
||||
display: grid;
|
||||
}
|
||||
|
||||
.app-shell {
|
||||
grid-template-columns: 320px minmax(0, 1fr);
|
||||
}
|
||||
|
||||
.sidebar,
|
||||
.viewer-sidebar {
|
||||
background: var(--panel-alt);
|
||||
border-right: 1px solid var(--line);
|
||||
padding: 24px;
|
||||
}
|
||||
|
||||
.content,
|
||||
.viewer-main {
|
||||
padding: 24px;
|
||||
}
|
||||
|
||||
.brand-block h1,
|
||||
.viewer-sidebar-head h1,
|
||||
.viewer-header h2,
|
||||
.panel-header h2 {
|
||||
margin: 0;
|
||||
font-size: 24px;
|
||||
}
|
||||
|
||||
.eyebrow {
|
||||
color: var(--accent);
|
||||
font-size: 12px;
|
||||
text-transform: uppercase;
|
||||
margin-bottom: 8px;
|
||||
}
|
||||
|
||||
.muted {
|
||||
color: var(--soft);
|
||||
}
|
||||
|
||||
.small {
|
||||
font-size: 12px;
|
||||
}
|
||||
|
||||
.nav-links {
|
||||
display: grid;
|
||||
gap: 8px;
|
||||
margin: 24px 0;
|
||||
}
|
||||
|
||||
.nav-link,
|
||||
.button {
|
||||
border: 1px solid var(--line);
|
||||
background: var(--panel);
|
||||
color: var(--text);
|
||||
padding: 10px 14px;
|
||||
border-radius: 8px;
|
||||
cursor: pointer;
|
||||
}
|
||||
|
||||
.nav-link.active,
|
||||
.button.primary {
|
||||
background: var(--accent-strong);
|
||||
border-color: var(--accent-strong);
|
||||
}
|
||||
|
||||
.button.danger {
|
||||
color: #ffd0d0;
|
||||
}
|
||||
|
||||
.button.button-small {
|
||||
padding: 8px 10px;
|
||||
font-size: 13px;
|
||||
}
|
||||
|
||||
.status-card,
|
||||
.panel,
|
||||
.stat-panel {
|
||||
background: var(--panel);
|
||||
border: 1px solid var(--line);
|
||||
border-radius: 8px;
|
||||
}
|
||||
|
||||
.status-card {
|
||||
padding: 16px;
|
||||
margin-top: 16px;
|
||||
}
|
||||
|
||||
.section-title {
|
||||
font-size: 12px;
|
||||
text-transform: uppercase;
|
||||
color: var(--soft);
|
||||
margin-bottom: 12px;
|
||||
}
|
||||
|
||||
.status-badge {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
border-radius: 999px;
|
||||
padding: 8px 12px;
|
||||
background: #183248;
|
||||
color: #a7d8ff;
|
||||
margin-bottom: 10px;
|
||||
}
|
||||
|
||||
.panel-grid {
|
||||
display: grid;
|
||||
gap: 16px;
|
||||
grid-template-columns: repeat(3, minmax(0, 1fr));
|
||||
margin-bottom: 16px;
|
||||
}
|
||||
|
||||
.stat-panel {
|
||||
padding: 18px;
|
||||
}
|
||||
|
||||
.stat-value {
|
||||
font-size: 32px;
|
||||
font-weight: 700;
|
||||
}
|
||||
|
||||
.toggle-row {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: space-between;
|
||||
gap: 12px;
|
||||
}
|
||||
|
||||
.panel {
|
||||
padding: 18px;
|
||||
margin-bottom: 16px;
|
||||
}
|
||||
|
||||
.panel-header,
|
||||
.viewer-sidebar-head,
|
||||
.viewer-header {
|
||||
display: flex;
|
||||
align-items: flex-start;
|
||||
justify-content: space-between;
|
||||
gap: 16px;
|
||||
margin-bottom: 16px;
|
||||
}
|
||||
|
||||
.inline-form {
|
||||
display: flex;
|
||||
gap: 8px;
|
||||
flex-wrap: wrap;
|
||||
}
|
||||
|
||||
input {
|
||||
min-width: 160px;
|
||||
padding: 10px 12px;
|
||||
border-radius: 8px;
|
||||
border: 1px solid var(--line);
|
||||
background: #0d141c;
|
||||
color: var(--text);
|
||||
}
|
||||
|
||||
.table-wrap {
|
||||
overflow-x: auto;
|
||||
}
|
||||
|
||||
table {
|
||||
width: 100%;
|
||||
border-collapse: collapse;
|
||||
}
|
||||
|
||||
th,
|
||||
td {
|
||||
text-align: left;
|
||||
border-bottom: 1px solid var(--line);
|
||||
padding: 14px 10px;
|
||||
vertical-align: top;
|
||||
}
|
||||
|
||||
tbody tr:hover {
|
||||
background: rgba(255, 255, 255, 0.02);
|
||||
}
|
||||
|
||||
.action-row {
|
||||
display: flex;
|
||||
gap: 8px;
|
||||
flex-wrap: wrap;
|
||||
}
|
||||
|
||||
.jobs-list {
|
||||
display: grid;
|
||||
gap: 12px;
|
||||
}
|
||||
|
||||
.job-card {
|
||||
border: 1px solid var(--line);
|
||||
border-radius: 8px;
|
||||
padding: 14px;
|
||||
background: #121c25;
|
||||
}
|
||||
|
||||
.job-head {
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
gap: 12px;
|
||||
margin-bottom: 6px;
|
||||
}
|
||||
|
||||
.job-status {
|
||||
text-transform: uppercase;
|
||||
font-size: 12px;
|
||||
color: var(--soft);
|
||||
}
|
||||
|
||||
.job-logs {
|
||||
margin: 10px 0 0;
|
||||
padding: 12px;
|
||||
border-radius: 8px;
|
||||
background: #0d141c;
|
||||
max-height: 240px;
|
||||
overflow: auto;
|
||||
white-space: pre-wrap;
|
||||
}
|
||||
|
||||
.viewer-shell {
|
||||
grid-template-columns: 320px minmax(0, 1fr);
|
||||
}
|
||||
|
||||
.viewer-channel-list {
|
||||
display: grid;
|
||||
gap: 8px;
|
||||
}
|
||||
|
||||
.viewer-channel-item {
|
||||
display: grid;
|
||||
gap: 4px;
|
||||
width: 100%;
|
||||
text-align: left;
|
||||
padding: 12px;
|
||||
border-radius: 8px;
|
||||
border: 1px solid var(--line);
|
||||
background: var(--panel);
|
||||
color: var(--text);
|
||||
cursor: pointer;
|
||||
}
|
||||
|
||||
.viewer-channel-item.active {
|
||||
border-color: var(--accent);
|
||||
background: #193246;
|
||||
}
|
||||
|
||||
.viewer-channel-name {
|
||||
font-weight: 600;
|
||||
}
|
||||
|
||||
.viewer-channel-meta {
|
||||
font-size: 12px;
|
||||
color: var(--soft);
|
||||
}
|
||||
|
||||
.messages-list {
|
||||
display: grid;
|
||||
gap: 12px;
|
||||
align-content: start;
|
||||
}
|
||||
|
||||
.message-card {
|
||||
max-width: 760px;
|
||||
border: 1px solid var(--line);
|
||||
border-radius: 12px;
|
||||
padding: 14px;
|
||||
background: var(--bubble);
|
||||
}
|
||||
|
||||
.message-meta {
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
gap: 12px;
|
||||
margin-bottom: 10px;
|
||||
}
|
||||
|
||||
.message-author {
|
||||
color: #8fd3ff;
|
||||
font-weight: 600;
|
||||
}
|
||||
|
||||
.message-reply:empty,
|
||||
.message-text:empty,
|
||||
.message-footer:empty,
|
||||
.message-media:empty {
|
||||
display: none;
|
||||
}
|
||||
|
||||
.message-text {
|
||||
white-space: pre-wrap;
|
||||
line-height: 1.45;
|
||||
}
|
||||
|
||||
.message-media {
|
||||
margin-top: 12px;
|
||||
}
|
||||
|
||||
.message-media img,
|
||||
.message-media video {
|
||||
max-width: 100%;
|
||||
border-radius: 10px;
|
||||
border: 1px solid var(--line);
|
||||
display: block;
|
||||
}
|
||||
|
||||
.message-footer {
|
||||
margin-top: 10px;
|
||||
}
|
||||
|
||||
@media (max-width: 1100px) {
|
||||
.app-shell,
|
||||
.viewer-shell {
|
||||
grid-template-columns: 1fr;
|
||||
}
|
||||
|
||||
.sidebar,
|
||||
.viewer-sidebar {
|
||||
border-right: 0;
|
||||
border-bottom: 1px solid var(--line);
|
||||
}
|
||||
|
||||
.panel-grid {
|
||||
grid-template-columns: 1fr;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="ru">
|
||||
<head>
|
||||
<meta charset="UTF-8" />
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
||||
<title>Telegram Scraper Viewer</title>
|
||||
<link rel="stylesheet" href="/static/style.css" />
|
||||
</head>
|
||||
<body class="viewer-body">
|
||||
<div class="viewer-shell">
|
||||
<aside class="viewer-sidebar">
|
||||
<div class="viewer-sidebar-head">
|
||||
<div>
|
||||
<div class="eyebrow">Export Viewer</div>
|
||||
<h1>Сообщения</h1>
|
||||
</div>
|
||||
<a class="button button-small" href="/">Панель</a>
|
||||
</div>
|
||||
<div id="viewer-channel-list" class="viewer-channel-list"></div>
|
||||
</aside>
|
||||
|
||||
<main class="viewer-main">
|
||||
<header class="viewer-header">
|
||||
<div>
|
||||
<h2 id="viewer-title">Выберите канал</h2>
|
||||
<p id="viewer-subtitle" class="muted">Показываем данные из локальной SQLite базы.</p>
|
||||
</div>
|
||||
<div class="viewer-actions">
|
||||
<button class="button" id="load-older-btn" disabled>Загрузить старее</button>
|
||||
</div>
|
||||
</header>
|
||||
|
||||
<section id="messages-list" class="messages-list"></section>
|
||||
</main>
|
||||
</div>
|
||||
|
||||
<template id="viewer-channel-template">
|
||||
<button class="viewer-channel-item">
|
||||
<span class="viewer-channel-name"></span>
|
||||
<span class="viewer-channel-meta"></span>
|
||||
</button>
|
||||
</template>
|
||||
|
||||
<template id="message-template">
|
||||
<article class="message-card">
|
||||
<div class="message-meta">
|
||||
<span class="message-author"></span>
|
||||
<span class="message-date"></span>
|
||||
</div>
|
||||
<div class="message-reply muted small"></div>
|
||||
<div class="message-text"></div>
|
||||
<div class="message-media"></div>
|
||||
<div class="message-footer muted small"></div>
|
||||
</article>
|
||||
</template>
|
||||
|
||||
<script src="/static/viewer.js"></script>
|
||||
</body>
|
||||
</html>
|
||||
+137
@@ -0,0 +1,137 @@
|
||||
const viewerState = {
|
||||
channels: [],
|
||||
channelId: null,
|
||||
oldestMessageId: null,
|
||||
};
|
||||
|
||||
async function api(path) {
|
||||
const response = await fetch(path);
|
||||
const data = await response.json();
|
||||
if (!response.ok) {
|
||||
throw new Error(data.error || "Request failed");
|
||||
}
|
||||
return data;
|
||||
}
|
||||
|
||||
function textNodeWithBreaks(text) {
|
||||
const fragment = document.createDocumentFragment();
|
||||
const parts = (text || "").split("\n");
|
||||
parts.forEach((part, index) => {
|
||||
if (index > 0) fragment.appendChild(document.createElement("br"));
|
||||
fragment.appendChild(document.createTextNode(part));
|
||||
});
|
||||
return fragment;
|
||||
}
|
||||
|
||||
function renderChannelList() {
|
||||
const root = document.getElementById("viewer-channel-list");
|
||||
root.innerHTML = "";
|
||||
const template = document.getElementById("viewer-channel-template");
|
||||
viewerState.channels.forEach((channel) => {
|
||||
const node = template.content.firstElementChild.cloneNode(true);
|
||||
node.querySelector(".viewer-channel-name").textContent = channel.name;
|
||||
node.querySelector(".viewer-channel-meta").textContent = `${channel.message_count} сообщений`;
|
||||
if (channel.channel_id === viewerState.channelId) {
|
||||
node.classList.add("active");
|
||||
}
|
||||
node.addEventListener("click", () => {
|
||||
viewerState.oldestMessageId = null;
|
||||
loadChannel(channel.channel_id);
|
||||
});
|
||||
root.appendChild(node);
|
||||
});
|
||||
}
|
||||
|
||||
function renderMessages(payload, append = false) {
|
||||
const root = document.getElementById("messages-list");
|
||||
const template = document.getElementById("message-template");
|
||||
if (!append) root.innerHTML = "";
|
||||
|
||||
const channel = payload.channel;
|
||||
document.getElementById("viewer-title").textContent = channel?.name || payload.channel_id;
|
||||
document.getElementById("viewer-subtitle").textContent = channel
|
||||
? `${channel.message_count} сообщений, последнее: ${channel.last_date || "—"}`
|
||||
: "База данных для канала пока не найдена.";
|
||||
|
||||
payload.messages.forEach((message) => {
|
||||
const node = template.content.firstElementChild.cloneNode(true);
|
||||
node.dataset.messageId = String(message.message_id);
|
||||
node.querySelector(".message-author").textContent = message.sender_name || "Unknown";
|
||||
node.querySelector(".message-date").textContent = message.date || "";
|
||||
|
||||
const reply = node.querySelector(".message-reply");
|
||||
if (message.reply_to) {
|
||||
reply.textContent = `Ответ на сообщение #${message.reply_to}`;
|
||||
}
|
||||
|
||||
node.querySelector(".message-text").appendChild(textNodeWithBreaks(message.text || ""));
|
||||
|
||||
const media = node.querySelector(".message-media");
|
||||
if (message.media_url && message.media_kind === "image") {
|
||||
const img = document.createElement("img");
|
||||
img.src = message.media_url;
|
||||
img.loading = "lazy";
|
||||
media.appendChild(img);
|
||||
} else if (message.media_url && message.media_kind === "video") {
|
||||
const video = document.createElement("video");
|
||||
video.src = message.media_url;
|
||||
video.controls = true;
|
||||
video.preload = "metadata";
|
||||
media.appendChild(video);
|
||||
} else if (message.media_url) {
|
||||
const link = document.createElement("a");
|
||||
link.href = message.media_url;
|
||||
link.target = "_blank";
|
||||
link.rel = "noreferrer";
|
||||
link.textContent = "Открыть файл";
|
||||
media.appendChild(link);
|
||||
}
|
||||
|
||||
const footerParts = [];
|
||||
if (message.views) footerParts.push(`views: ${message.views}`);
|
||||
if (message.forwards) footerParts.push(`forwards: ${message.forwards}`);
|
||||
if (message.reactions) footerParts.push(`reactions: ${message.reactions}`);
|
||||
node.querySelector(".message-footer").textContent = footerParts.join(" | ");
|
||||
|
||||
if (append) {
|
||||
root.prepend(node);
|
||||
} else {
|
||||
root.appendChild(node);
|
||||
}
|
||||
});
|
||||
|
||||
if (payload.messages.length) {
|
||||
viewerState.oldestMessageId = payload.messages[0].message_id;
|
||||
}
|
||||
|
||||
document.getElementById("load-older-btn").disabled = !payload.messages.length;
|
||||
}
|
||||
|
||||
async function loadChannel(channelId, append = false) {
|
||||
viewerState.channelId = channelId;
|
||||
renderChannelList();
|
||||
const before = append && viewerState.oldestMessageId ? `&before=${viewerState.oldestMessageId}` : "";
|
||||
const payload = await api(`/api/channels/${encodeURIComponent(channelId)}/messages?limit=80${before}`);
|
||||
renderMessages(payload, append);
|
||||
}
|
||||
|
||||
async function main() {
|
||||
viewerState.channels = await api("/api/channels");
|
||||
const params = new URLSearchParams(window.location.search);
|
||||
viewerState.channelId = params.get("channel") || viewerState.channels[0]?.channel_id || null;
|
||||
renderChannelList();
|
||||
|
||||
if (viewerState.channelId) {
|
||||
await loadChannel(viewerState.channelId);
|
||||
}
|
||||
|
||||
document.getElementById("load-older-btn").addEventListener("click", async () => {
|
||||
if (!viewerState.channelId || !viewerState.oldestMessageId) return;
|
||||
await loadChannel(viewerState.channelId, true);
|
||||
});
|
||||
}
|
||||
|
||||
main().catch((error) => {
|
||||
console.error(error);
|
||||
alert(error.message);
|
||||
});
|
||||
+623
@@ -0,0 +1,623 @@
|
||||
import asyncio
|
||||
import contextlib
|
||||
import io
|
||||
import json
|
||||
import mimetypes
|
||||
import os
|
||||
import posixpath
|
||||
import queue
|
||||
import sqlite3
|
||||
import threading
|
||||
import time
|
||||
import traceback
|
||||
import urllib.parse
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from http import HTTPStatus
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable, Dict, List, Optional
|
||||
|
||||
|
||||
BASE_DIR = Path(__file__).resolve().parent
|
||||
DATA_DIR = BASE_DIR / "data"
|
||||
WEBUI_DIR = BASE_DIR / "webui"
|
||||
STATE_FILE = DATA_DIR / "state.json"
|
||||
DEFAULT_HOST = os.environ.get("TELEGRAM_SCRAPER_HOST", "0.0.0.0")
|
||||
DEFAULT_PORT = int(os.environ.get("TELEGRAM_SCRAPER_PORT", "8080"))
|
||||
|
||||
|
||||
def utc_now_iso() -> str:
|
||||
return datetime.now(timezone.utc).isoformat()
|
||||
|
||||
|
||||
def load_state() -> Dict[str, Any]:
|
||||
if STATE_FILE.exists():
|
||||
with STATE_FILE.open("r", encoding="utf-8") as handle:
|
||||
return json.load(handle)
|
||||
return {
|
||||
"api_id": None,
|
||||
"api_hash": None,
|
||||
"channels": {},
|
||||
"channel_names": {},
|
||||
"scrape_media": True,
|
||||
"forwarding_rules": [],
|
||||
}
|
||||
|
||||
|
||||
def save_state(state: Dict[str, Any]) -> None:
|
||||
DATA_DIR.mkdir(parents=True, exist_ok=True)
|
||||
with STATE_FILE.open("w", encoding="utf-8") as handle:
|
||||
json.dump(state, handle, ensure_ascii=False, indent=2)
|
||||
|
||||
|
||||
def channel_db_path(channel_id: str) -> Path:
|
||||
return DATA_DIR / channel_id / f"{channel_id}.db"
|
||||
|
||||
|
||||
def guess_media_kind(media_path: Optional[str], media_type: Optional[str]) -> Optional[str]:
|
||||
if not media_path:
|
||||
return None
|
||||
suffix = Path(media_path).suffix.lower()
|
||||
if suffix in {".jpg", ".jpeg", ".png", ".webp", ".gif"}:
|
||||
return "image"
|
||||
if suffix in {".mp4", ".webm", ".mov", ".m4v"}:
|
||||
return "video"
|
||||
if suffix in {".mp3", ".ogg", ".wav", ".m4a"}:
|
||||
return "audio"
|
||||
if media_type == "MessageMediaPhoto":
|
||||
return "image"
|
||||
return "file"
|
||||
|
||||
|
||||
def normalize_media_url(media_path: Optional[str]) -> Optional[str]:
|
||||
if not media_path:
|
||||
return None
|
||||
path = Path(media_path)
|
||||
if path.is_absolute():
|
||||
try:
|
||||
relative = path.resolve().relative_to(DATA_DIR.resolve())
|
||||
except ValueError:
|
||||
return None
|
||||
else:
|
||||
normalized = Path(media_path)
|
||||
if normalized.parts and normalized.parts[0] == "data":
|
||||
normalized = Path(*normalized.parts[1:])
|
||||
relative = normalized
|
||||
return "/media/" + urllib.parse.quote(str(relative).replace("\\", "/"))
|
||||
|
||||
|
||||
def database_summary(channel_id: str) -> Dict[str, Any]:
|
||||
db_path = channel_db_path(channel_id)
|
||||
summary = {
|
||||
"message_count": 0,
|
||||
"last_date": None,
|
||||
"first_date": None,
|
||||
"media_count": 0,
|
||||
"has_database": db_path.exists(),
|
||||
}
|
||||
if not db_path.exists():
|
||||
return summary
|
||||
conn = sqlite3.connect(str(db_path))
|
||||
try:
|
||||
cursor = conn.cursor()
|
||||
cursor.execute(
|
||||
"SELECT COUNT(*), MIN(date), MAX(date), "
|
||||
"SUM(CASE WHEN media_type IS NOT NULL AND media_type != 'MessageMediaWebPage' THEN 1 ELSE 0 END) "
|
||||
"FROM messages"
|
||||
)
|
||||
row = cursor.fetchone() or (0, None, None, 0)
|
||||
summary.update(
|
||||
{
|
||||
"message_count": row[0] or 0,
|
||||
"first_date": row[1],
|
||||
"last_date": row[2],
|
||||
"media_count": row[3] or 0,
|
||||
}
|
||||
)
|
||||
return summary
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def channel_display_name(state: Dict[str, Any], channel_id: str) -> str:
|
||||
return (
|
||||
state.get("channel_names", {}).get(channel_id)
|
||||
or channel_id
|
||||
)
|
||||
|
||||
|
||||
def list_channels_snapshot() -> List[Dict[str, Any]]:
|
||||
state = load_state()
|
||||
channels: List[Dict[str, Any]] = []
|
||||
for channel_id, last_message_id in state.get("channels", {}).items():
|
||||
summary = database_summary(channel_id)
|
||||
channels.append(
|
||||
{
|
||||
"channel_id": channel_id,
|
||||
"name": channel_display_name(state, channel_id),
|
||||
"last_message_id": last_message_id,
|
||||
**summary,
|
||||
}
|
||||
)
|
||||
channels.sort(
|
||||
key=lambda item: (
|
||||
item["last_date"] or "",
|
||||
item["message_count"],
|
||||
),
|
||||
reverse=True,
|
||||
)
|
||||
return channels
|
||||
|
||||
|
||||
def load_messages(channel_id: str, limit: int = 120, before_message_id: Optional[int] = None) -> List[Dict[str, Any]]:
|
||||
db_path = channel_db_path(channel_id)
|
||||
if not db_path.exists():
|
||||
return []
|
||||
conn = sqlite3.connect(str(db_path))
|
||||
conn.row_factory = sqlite3.Row
|
||||
try:
|
||||
query = (
|
||||
"SELECT message_id, date, sender_id, first_name, last_name, username, message, "
|
||||
"media_type, media_path, reply_to, post_author, views, forwards, reactions "
|
||||
"FROM messages "
|
||||
)
|
||||
params: List[Any] = []
|
||||
if before_message_id is not None:
|
||||
query += "WHERE message_id < ? "
|
||||
params.append(before_message_id)
|
||||
query += "ORDER BY message_id DESC LIMIT ?"
|
||||
params.append(limit)
|
||||
rows = conn.execute(query, params).fetchall()
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
messages: List[Dict[str, Any]] = []
|
||||
for row in reversed(rows):
|
||||
sender_name = " ".join(
|
||||
part for part in [row["first_name"], row["last_name"]] if part
|
||||
).strip()
|
||||
if not sender_name:
|
||||
sender_name = row["username"] or row["post_author"] or channel_id
|
||||
messages.append(
|
||||
{
|
||||
"message_id": row["message_id"],
|
||||
"date": row["date"],
|
||||
"sender_id": row["sender_id"],
|
||||
"sender_name": sender_name,
|
||||
"username": row["username"],
|
||||
"text": row["message"] or "",
|
||||
"media_type": row["media_type"],
|
||||
"media_path": row["media_path"],
|
||||
"media_url": normalize_media_url(row["media_path"]),
|
||||
"media_kind": guess_media_kind(row["media_path"], row["media_type"]),
|
||||
"reply_to": row["reply_to"],
|
||||
"post_author": row["post_author"],
|
||||
"views": row["views"],
|
||||
"forwards": row["forwards"],
|
||||
"reactions": row["reactions"],
|
||||
}
|
||||
)
|
||||
return messages
|
||||
|
||||
|
||||
@dataclass
|
||||
class Job:
|
||||
job_id: str
|
||||
job_type: str
|
||||
title: str
|
||||
payload: Dict[str, Any]
|
||||
status: str = "queued"
|
||||
created_at: str = field(default_factory=utc_now_iso)
|
||||
started_at: Optional[str] = None
|
||||
finished_at: Optional[str] = None
|
||||
logs: str = ""
|
||||
error: Optional[str] = None
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {
|
||||
"job_id": self.job_id,
|
||||
"job_type": self.job_type,
|
||||
"title": self.title,
|
||||
"payload": self.payload,
|
||||
"status": self.status,
|
||||
"created_at": self.created_at,
|
||||
"started_at": self.started_at,
|
||||
"finished_at": self.finished_at,
|
||||
"logs": self.logs,
|
||||
"error": self.error,
|
||||
}
|
||||
|
||||
|
||||
class JobRunner:
|
||||
def __init__(self) -> None:
|
||||
self.jobs: Dict[str, Job] = {}
|
||||
self.job_order: List[str] = []
|
||||
self.queue: "queue.Queue[Job]" = queue.Queue()
|
||||
self.lock = threading.Lock()
|
||||
self.worker = threading.Thread(target=self._run, daemon=True)
|
||||
self.worker.start()
|
||||
|
||||
def create_job(self, job_type: str, title: str, payload: Dict[str, Any]) -> Job:
|
||||
job_id = f"job-{int(time.time() * 1000)}"
|
||||
job = Job(job_id=job_id, job_type=job_type, title=title, payload=payload)
|
||||
with self.lock:
|
||||
self.jobs[job_id] = job
|
||||
self.job_order.insert(0, job_id)
|
||||
self.job_order = self.job_order[:20]
|
||||
self.queue.put(job)
|
||||
return job
|
||||
|
||||
def recent_jobs(self) -> List[Dict[str, Any]]:
|
||||
with self.lock:
|
||||
return [self.jobs[job_id].to_dict() for job_id in self.job_order]
|
||||
|
||||
def get_job(self, job_id: str) -> Optional[Dict[str, Any]]:
|
||||
with self.lock:
|
||||
job = self.jobs.get(job_id)
|
||||
return job.to_dict() if job else None
|
||||
|
||||
def _run(self) -> None:
|
||||
while True:
|
||||
job = self.queue.get()
|
||||
self._execute(job)
|
||||
self.queue.task_done()
|
||||
|
||||
def _execute(self, job: Job) -> None:
|
||||
with self.lock:
|
||||
job.status = "running"
|
||||
job.started_at = utc_now_iso()
|
||||
buffer = io.StringIO()
|
||||
try:
|
||||
with contextlib.redirect_stdout(buffer), contextlib.redirect_stderr(buffer):
|
||||
run_job(job.job_type, job.payload)
|
||||
with self.lock:
|
||||
job.status = "done"
|
||||
except Exception as exc:
|
||||
with self.lock:
|
||||
job.status = "failed"
|
||||
job.error = str(exc)
|
||||
traceback.print_exc(file=buffer)
|
||||
finally:
|
||||
with self.lock:
|
||||
job.logs = buffer.getvalue()
|
||||
job.finished_at = utc_now_iso()
|
||||
|
||||
|
||||
def import_scraper_class():
|
||||
from telegram_scraper_with_forwarding import OptimizedTelegramScraper
|
||||
|
||||
return OptimizedTelegramScraper
|
||||
|
||||
|
||||
def run_job(job_type: str, payload: Dict[str, Any]) -> None:
|
||||
if job_type == "set_scrape_media":
|
||||
state = load_state()
|
||||
state["scrape_media"] = bool(payload["value"])
|
||||
save_state(state)
|
||||
print(f"Media scraping set to {state['scrape_media']}")
|
||||
return
|
||||
|
||||
async def _async_job() -> None:
|
||||
ScraperClass = import_scraper_class()
|
||||
scraper = ScraperClass()
|
||||
initialized = await scraper.initialize_client(interactive=False)
|
||||
if not initialized:
|
||||
raise RuntimeError("Telegram client is not ready. Check credentials and session.")
|
||||
try:
|
||||
if job_type == "scrape_channel":
|
||||
channel_id = payload["channel_id"]
|
||||
state = scraper.load_state()
|
||||
offset = int(state.get("channels", {}).get(channel_id, 0) or 0)
|
||||
await scraper.scrape_channel(channel_id, offset)
|
||||
elif job_type == "scrape_all":
|
||||
channels = list(scraper.state.get("channels", {}).keys())
|
||||
for channel_id in channels:
|
||||
offset = int(scraper.state.get("channels", {}).get(channel_id, 0) or 0)
|
||||
await scraper.scrape_channel(channel_id, offset)
|
||||
elif job_type == "export_all":
|
||||
await scraper.export_data()
|
||||
elif job_type == "rescrape_media":
|
||||
await scraper.rescrape_media(payload["channel_id"])
|
||||
elif job_type == "fix_missing_media":
|
||||
await scraper.fix_missing_media(payload["channel_id"])
|
||||
elif job_type == "refresh_dialogs":
|
||||
await scraper.list_channels()
|
||||
else:
|
||||
raise RuntimeError(f"Unsupported job type: {job_type}")
|
||||
finally:
|
||||
scraper.close_db_connections()
|
||||
if scraper.client:
|
||||
await scraper.client.disconnect()
|
||||
|
||||
asyncio.run(_async_job())
|
||||
|
||||
|
||||
def auth_status() -> Dict[str, Any]:
|
||||
state = load_state()
|
||||
status = {
|
||||
"has_api_credentials": bool(state.get("api_id") and state.get("api_hash")),
|
||||
"telethon_available": False,
|
||||
"session_ready": False,
|
||||
"status": "read_only",
|
||||
"details": "",
|
||||
}
|
||||
try:
|
||||
import_scraper_class()
|
||||
status["telethon_available"] = True
|
||||
except Exception as exc:
|
||||
status["details"] = f"Python deps are missing: {exc}"
|
||||
return status
|
||||
|
||||
session_file = BASE_DIR / "session" / "session.session"
|
||||
status["session_ready"] = session_file.exists()
|
||||
if status["has_api_credentials"] and status["session_ready"]:
|
||||
status["status"] = "ready"
|
||||
status["details"] = "Scraping actions should be available after dependencies are installed."
|
||||
elif status["has_api_credentials"]:
|
||||
status["status"] = "needs_auth"
|
||||
status["details"] = "API keys found, but Telegram session file is missing."
|
||||
else:
|
||||
status["status"] = "needs_credentials"
|
||||
status["details"] = "api_id/api_hash are missing in data/state.json."
|
||||
return status
|
||||
|
||||
|
||||
def dashboard_payload(job_runner: JobRunner) -> Dict[str, Any]:
|
||||
state = load_state()
|
||||
channels = list_channels_snapshot()
|
||||
return {
|
||||
"state": {
|
||||
"scrape_media": bool(state.get("scrape_media", True)),
|
||||
"channel_count": len(state.get("channels", {})),
|
||||
"forwarding_rules": state.get("forwarding_rules", []),
|
||||
},
|
||||
"auth": auth_status(),
|
||||
"channels": channels,
|
||||
"jobs": job_runner.recent_jobs(),
|
||||
}
|
||||
|
||||
|
||||
class TelegramScraperRequestHandler(BaseHTTPRequestHandler):
|
||||
server_version = "TelegramScraperWebUI/0.1"
|
||||
|
||||
@property
|
||||
def app(self) -> "TelegramScraperWebServer":
|
||||
return self.server # type: ignore[return-value]
|
||||
|
||||
def do_GET(self) -> None:
|
||||
parsed = urllib.parse.urlparse(self.path)
|
||||
path = parsed.path
|
||||
query = urllib.parse.parse_qs(parsed.query)
|
||||
|
||||
if path == "/":
|
||||
return self.serve_file(WEBUI_DIR / "index.html", "text/html; charset=utf-8")
|
||||
if path == "/viewer":
|
||||
return self.serve_file(WEBUI_DIR / "viewer.html", "text/html; charset=utf-8")
|
||||
if path.startswith("/static/"):
|
||||
relative = path.removeprefix("/static/")
|
||||
return self.serve_static(relative)
|
||||
if path.startswith("/media/"):
|
||||
relative = path.removeprefix("/media/")
|
||||
return self.serve_media(relative)
|
||||
if path == "/api/dashboard":
|
||||
return self.send_json(dashboard_payload(self.app.job_runner))
|
||||
if path == "/api/jobs":
|
||||
return self.send_json(self.app.job_runner.recent_jobs())
|
||||
if path.startswith("/api/jobs/"):
|
||||
job_id = path.rsplit("/", 1)[-1]
|
||||
job = self.app.job_runner.get_job(job_id)
|
||||
if not job:
|
||||
return self.send_error_json(HTTPStatus.NOT_FOUND, "Job not found")
|
||||
return self.send_json(job)
|
||||
if path == "/api/channels":
|
||||
return self.send_json(list_channels_snapshot())
|
||||
if path.startswith("/api/channels/") and path.endswith("/messages"):
|
||||
parts = path.split("/")
|
||||
channel_id = parts[3]
|
||||
limit = max(1, min(int(query.get("limit", ["120"])[0]), 300))
|
||||
before = query.get("before")
|
||||
before_message_id = int(before[0]) if before else None
|
||||
payload = {
|
||||
"channel_id": channel_id,
|
||||
"messages": load_messages(channel_id, limit=limit, before_message_id=before_message_id),
|
||||
"channel": next(
|
||||
(item for item in list_channels_snapshot() if item["channel_id"] == channel_id),
|
||||
None,
|
||||
),
|
||||
}
|
||||
return self.send_json(payload)
|
||||
return self.send_error_json(HTTPStatus.NOT_FOUND, "Not found")
|
||||
|
||||
def do_HEAD(self) -> None:
|
||||
parsed = urllib.parse.urlparse(self.path)
|
||||
path = parsed.path
|
||||
|
||||
if path == "/":
|
||||
return self.serve_file(WEBUI_DIR / "index.html", "text/html; charset=utf-8", head_only=True)
|
||||
if path == "/viewer":
|
||||
return self.serve_file(WEBUI_DIR / "viewer.html", "text/html; charset=utf-8", head_only=True)
|
||||
if path.startswith("/static/"):
|
||||
relative = path.removeprefix("/static/")
|
||||
return self.serve_static(relative, head_only=True)
|
||||
if path.startswith("/media/"):
|
||||
relative = path.removeprefix("/media/")
|
||||
return self.serve_media(relative, head_only=True)
|
||||
if path in {"/api/dashboard", "/api/jobs", "/api/channels"} or path.startswith("/api/jobs/") or (
|
||||
path.startswith("/api/channels/") and path.endswith("/messages")
|
||||
):
|
||||
self.send_response(HTTPStatus.OK)
|
||||
self.send_header("Content-Type", "application/json; charset=utf-8")
|
||||
self.end_headers()
|
||||
return
|
||||
return self.send_error_json(HTTPStatus.NOT_FOUND, "Not found")
|
||||
|
||||
def do_POST(self) -> None:
|
||||
parsed = urllib.parse.urlparse(self.path)
|
||||
path = parsed.path
|
||||
body = self.read_json_body()
|
||||
if body is None:
|
||||
return self.send_error_json(HTTPStatus.BAD_REQUEST, "Expected JSON body")
|
||||
|
||||
if path == "/api/settings/media":
|
||||
value = bool(body.get("value"))
|
||||
job = self.app.job_runner.create_job(
|
||||
"set_scrape_media",
|
||||
"Update media scraping setting",
|
||||
{"value": value},
|
||||
)
|
||||
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
|
||||
|
||||
if path == "/api/channels/add":
|
||||
channel_id = str(body.get("channel_id", "")).strip()
|
||||
if not channel_id:
|
||||
return self.send_error_json(HTTPStatus.BAD_REQUEST, "channel_id is required")
|
||||
state = load_state()
|
||||
if channel_id not in state["channels"]:
|
||||
state["channels"][channel_id] = 0
|
||||
if body.get("name"):
|
||||
state.setdefault("channel_names", {})[channel_id] = str(body["name"]).strip()
|
||||
save_state(state)
|
||||
return self.send_json({"ok": True, "channel_id": channel_id})
|
||||
|
||||
if path == "/api/channels/remove":
|
||||
channel_id = str(body.get("channel_id", "")).strip()
|
||||
state = load_state()
|
||||
existed = channel_id in state.get("channels", {})
|
||||
state.get("channels", {}).pop(channel_id, None)
|
||||
save_state(state)
|
||||
return self.send_json({"ok": existed, "channel_id": channel_id})
|
||||
|
||||
if path == "/api/jobs/scrape":
|
||||
channel_id = body.get("channel_id")
|
||||
if channel_id:
|
||||
job = self.app.job_runner.create_job(
|
||||
"scrape_channel",
|
||||
f"Scrape channel {channel_id}",
|
||||
{"channel_id": str(channel_id)},
|
||||
)
|
||||
else:
|
||||
job = self.app.job_runner.create_job(
|
||||
"scrape_all",
|
||||
"Scrape all tracked channels",
|
||||
{},
|
||||
)
|
||||
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
|
||||
|
||||
if path == "/api/jobs/export":
|
||||
job = self.app.job_runner.create_job(
|
||||
"export_all",
|
||||
"Export all tracked channels",
|
||||
{},
|
||||
)
|
||||
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
|
||||
|
||||
if path == "/api/jobs/rescrape-media":
|
||||
channel_id = str(body.get("channel_id", "")).strip()
|
||||
if not channel_id:
|
||||
return self.send_error_json(HTTPStatus.BAD_REQUEST, "channel_id is required")
|
||||
job = self.app.job_runner.create_job(
|
||||
"rescrape_media",
|
||||
f"Rescrape media for {channel_id}",
|
||||
{"channel_id": channel_id},
|
||||
)
|
||||
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
|
||||
|
||||
if path == "/api/jobs/fix-media":
|
||||
channel_id = str(body.get("channel_id", "")).strip()
|
||||
if not channel_id:
|
||||
return self.send_error_json(HTTPStatus.BAD_REQUEST, "channel_id is required")
|
||||
job = self.app.job_runner.create_job(
|
||||
"fix_missing_media",
|
||||
f"Fix missing media for {channel_id}",
|
||||
{"channel_id": channel_id},
|
||||
)
|
||||
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
|
||||
|
||||
if path == "/api/jobs/refresh-dialogs":
|
||||
job = self.app.job_runner.create_job(
|
||||
"refresh_dialogs",
|
||||
"Refresh Telegram dialogs list",
|
||||
{},
|
||||
)
|
||||
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
|
||||
|
||||
return self.send_error_json(HTTPStatus.NOT_FOUND, "Not found")
|
||||
|
||||
def read_json_body(self) -> Optional[Dict[str, Any]]:
|
||||
length = int(self.headers.get("Content-Length", "0"))
|
||||
if length <= 0:
|
||||
return {}
|
||||
raw = self.rfile.read(length)
|
||||
try:
|
||||
return json.loads(raw.decode("utf-8"))
|
||||
except json.JSONDecodeError:
|
||||
return None
|
||||
|
||||
def serve_static(self, relative_path: str, head_only: bool = False) -> None:
|
||||
file_path = (WEBUI_DIR / relative_path).resolve()
|
||||
try:
|
||||
file_path.relative_to(WEBUI_DIR.resolve())
|
||||
except ValueError:
|
||||
return self.send_error_json(HTTPStatus.FORBIDDEN, "Forbidden")
|
||||
if not file_path.exists() or not file_path.is_file():
|
||||
return self.send_error_json(HTTPStatus.NOT_FOUND, "File not found")
|
||||
mime_type = mimetypes.guess_type(str(file_path))[0] or "application/octet-stream"
|
||||
return self.serve_file(file_path, mime_type, head_only=head_only)
|
||||
|
||||
def serve_media(self, relative_path: str, head_only: bool = False) -> None:
|
||||
clean = Path(posixpath.normpath(urllib.parse.unquote(relative_path)))
|
||||
file_path = (DATA_DIR / clean).resolve()
|
||||
try:
|
||||
file_path.relative_to(DATA_DIR.resolve())
|
||||
except ValueError:
|
||||
return self.send_error_json(HTTPStatus.FORBIDDEN, "Forbidden")
|
||||
if not file_path.exists() or not file_path.is_file():
|
||||
return self.send_error_json(HTTPStatus.NOT_FOUND, "Media file not found")
|
||||
mime_type = mimetypes.guess_type(str(file_path))[0] or "application/octet-stream"
|
||||
return self.serve_file(file_path, mime_type, head_only=head_only)
|
||||
|
||||
def serve_file(self, file_path: Path, content_type: str, head_only: bool = False) -> None:
|
||||
data = file_path.read_bytes()
|
||||
self.send_response(HTTPStatus.OK)
|
||||
self.send_header("Content-Type", content_type)
|
||||
self.send_header("Content-Length", str(len(data)))
|
||||
self.end_headers()
|
||||
if not head_only:
|
||||
self.wfile.write(data)
|
||||
|
||||
def send_json(self, payload: Any, status: HTTPStatus = HTTPStatus.OK) -> None:
|
||||
raw = json.dumps(payload, ensure_ascii=False).encode("utf-8")
|
||||
self.send_response(status)
|
||||
self.send_header("Content-Type", "application/json; charset=utf-8")
|
||||
self.send_header("Content-Length", str(len(raw)))
|
||||
self.end_headers()
|
||||
self.wfile.write(raw)
|
||||
|
||||
def send_error_json(self, status: HTTPStatus, message: str) -> None:
|
||||
self.send_json({"error": message, "status": status}, status=status)
|
||||
|
||||
def log_message(self, format: str, *args: Any) -> None:
|
||||
return
|
||||
|
||||
|
||||
class TelegramScraperWebServer(ThreadingHTTPServer):
|
||||
def __init__(self, server_address: tuple[str, int]):
|
||||
super().__init__(server_address, TelegramScraperRequestHandler)
|
||||
self.job_runner = JobRunner()
|
||||
|
||||
|
||||
def run_server(host: str = DEFAULT_HOST, port: int = DEFAULT_PORT) -> None:
|
||||
server = TelegramScraperWebServer((host, port))
|
||||
print(f"Telegram Scraper Web UI listening on http://{host}:{port}")
|
||||
print("Press Ctrl+C to stop.")
|
||||
try:
|
||||
server.serve_forever()
|
||||
except KeyboardInterrupt:
|
||||
pass
|
||||
finally:
|
||||
server.server_close()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
run_server()
|
||||
Reference in New Issue
Block a user