tgmirror 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tgmirror-0.1.1/.claude/skills/checkpoint-state/SKILL.md +90 -0
- tgmirror-0.1.1/.claude/skills/cli-wizard/SKILL.md +88 -0
- tgmirror-0.1.1/.claude/skills/doc-sync/SKILL.md +66 -0
- tgmirror-0.1.1/.claude/skills/filter-dsl/SKILL.md +54 -0
- tgmirror-0.1.1/.claude/skills/flood-safety/SKILL.md +64 -0
- tgmirror-0.1.1/.claude/skills/telethon-engine/SKILL.md +155 -0
- tgmirror-0.1.1/.dockerignore +13 -0
- tgmirror-0.1.1/.github/workflows/ci.yml +37 -0
- tgmirror-0.1.1/.github/workflows/release.yml +270 -0
- tgmirror-0.1.1/.gitignore +22 -0
- tgmirror-0.1.1/CLAUDE.md +47 -0
- tgmirror-0.1.1/Dockerfile +50 -0
- tgmirror-0.1.1/LICENSE +21 -0
- tgmirror-0.1.1/PKG-INFO +142 -0
- tgmirror-0.1.1/README.md +114 -0
- tgmirror-0.1.1/docs/00-tong-quan.md +50 -0
- tgmirror-0.1.1/docs/01-kien-truc.md +278 -0
- tgmirror-0.1.1/docs/02-cli-ux.md +264 -0
- tgmirror-0.1.1/docs/03-filters.md +101 -0
- tgmirror-0.1.1/docs/04-state-checkpoint.md +245 -0
- tgmirror-0.1.1/docs/05-chong-flood.md +99 -0
- tgmirror-0.1.1/docs/06-lo-trinh.md +957 -0
- tgmirror-0.1.1/pyinstaller.spec +44 -0
- tgmirror-0.1.1/pyproject.toml +70 -0
- tgmirror-0.1.1/scripts/_spike.py +38 -0
- tgmirror-0.1.1/scripts/mutation_check.py +591 -0
- tgmirror-0.1.1/scripts/pyinstaller_entry.py +10 -0
- tgmirror-0.1.1/scripts/spike_count.py +75 -0
- tgmirror-0.1.1/scripts/spike_reference.py +121 -0
- tgmirror-0.1.1/scripts/spike_throughput.py +463 -0
- tgmirror-0.1.1/src/tgmirror/__init__.py +8 -0
- tgmirror-0.1.1/src/tgmirror/__main__.py +3 -0
- tgmirror-0.1.1/src/tgmirror/cli/__init__.py +0 -0
- tgmirror-0.1.1/src/tgmirror/cli/app.py +134 -0
- tgmirror-0.1.1/src/tgmirror/cli/commands/__init__.py +0 -0
- tgmirror-0.1.1/src/tgmirror/cli/commands/appdata.py +67 -0
- tgmirror-0.1.1/src/tgmirror/cli/commands/auth.py +165 -0
- tgmirror-0.1.1/src/tgmirror/cli/commands/backup.py +431 -0
- tgmirror-0.1.1/src/tgmirror/cli/commands/channels.py +48 -0
- tgmirror-0.1.1/src/tgmirror/cli/commands/clone.py +674 -0
- tgmirror-0.1.1/src/tgmirror/cli/commands/config.py +98 -0
- tgmirror-0.1.1/src/tgmirror/cli/commands/control.py +52 -0
- tgmirror-0.1.1/src/tgmirror/cli/commands/doctor.py +120 -0
- tgmirror-0.1.1/src/tgmirror/cli/commands/history.py +206 -0
- tgmirror-0.1.1/src/tgmirror/cli/commands/restore.py +514 -0
- tgmirror-0.1.1/src/tgmirror/cli/commands/retry.py +139 -0
- tgmirror-0.1.1/src/tgmirror/cli/commands/run.py +307 -0
- tgmirror-0.1.1/src/tgmirror/cli/commands/status.py +298 -0
- tgmirror-0.1.1/src/tgmirror/cli/commands/topics.py +43 -0
- tgmirror-0.1.1/src/tgmirror/cli/errors.py +266 -0
- tgmirror-0.1.1/src/tgmirror/cli/filter_options.py +163 -0
- tgmirror-0.1.1/src/tgmirror/cli/interrupt.py +47 -0
- tgmirror-0.1.1/src/tgmirror/cli/keys.py +233 -0
- tgmirror-0.1.1/src/tgmirror/cli/runtime.py +111 -0
- tgmirror-0.1.1/src/tgmirror/cli/wizard.py +320 -0
- tgmirror-0.1.1/src/tgmirror/core/__init__.py +0 -0
- tgmirror-0.1.1/src/tgmirror/core/auth.py +136 -0
- tgmirror-0.1.1/src/tgmirror/core/config.py +339 -0
- tgmirror-0.1.1/src/tgmirror/core/errors.py +196 -0
- tgmirror-0.1.1/src/tgmirror/core/gateway.py +457 -0
- tgmirror-0.1.1/src/tgmirror/core/limiter.py +193 -0
- tgmirror-0.1.1/src/tgmirror/core/paths.py +65 -0
- tgmirror-0.1.1/src/tgmirror/core/pool.py +210 -0
- tgmirror-0.1.1/src/tgmirror/core/secrets.py +144 -0
- tgmirror-0.1.1/src/tgmirror/core/session_lock.py +73 -0
- tgmirror-0.1.1/src/tgmirror/core/telethon_gateway.py +1914 -0
- tgmirror-0.1.1/src/tgmirror/engine/__init__.py +0 -0
- tgmirror-0.1.1/src/tgmirror/engine/backup.py +395 -0
- tgmirror-0.1.1/src/tgmirror/engine/backup_reader.py +152 -0
- tgmirror-0.1.1/src/tgmirror/engine/backupdir.py +273 -0
- tgmirror-0.1.1/src/tgmirror/engine/batcher.py +107 -0
- tgmirror-0.1.1/src/tgmirror/engine/copy.py +29 -0
- tgmirror-0.1.1/src/tgmirror/engine/endpoints.py +203 -0
- tgmirror-0.1.1/src/tgmirror/engine/flood.py +306 -0
- tgmirror-0.1.1/src/tgmirror/engine/planner.py +123 -0
- tgmirror-0.1.1/src/tgmirror/engine/preview.py +61 -0
- tgmirror-0.1.1/src/tgmirror/engine/reconcile.py +55 -0
- tgmirror-0.1.1/src/tgmirror/engine/reupload.py +390 -0
- tgmirror-0.1.1/src/tgmirror/engine/runner.py +796 -0
- tgmirror-0.1.1/src/tgmirror/engine/runs.py +277 -0
- tgmirror-0.1.1/src/tgmirror/engine/status.py +237 -0
- tgmirror-0.1.1/src/tgmirror/engine/strategy.py +102 -0
- tgmirror-0.1.1/src/tgmirror/engine/topics.py +86 -0
- tgmirror-0.1.1/src/tgmirror/engine/transfer.py +66 -0
- tgmirror-0.1.1/src/tgmirror/filters/__init__.py +1 -0
- tgmirror-0.1.1/src/tgmirror/filters/matcher.py +137 -0
- tgmirror-0.1.1/src/tgmirror/filters/model.py +303 -0
- tgmirror-0.1.1/src/tgmirror/filters/parser.py +126 -0
- tgmirror-0.1.1/src/tgmirror/filters/pushdown.py +67 -0
- tgmirror-0.1.1/src/tgmirror/store/__init__.py +1 -0
- tgmirror-0.1.1/src/tgmirror/store/appdata.py +171 -0
- tgmirror-0.1.1/src/tgmirror/store/backups.py +107 -0
- tgmirror-0.1.1/src/tgmirror/store/db.py +838 -0
- tgmirror-0.1.1/src/tgmirror/store/floodlog.py +57 -0
- tgmirror-0.1.1/src/tgmirror/store/limiterstate.py +27 -0
- tgmirror-0.1.1/src/tgmirror/store/msgmap.py +245 -0
- tgmirror-0.1.1/src/tgmirror/store/runs.py +295 -0
- tgmirror-0.1.1/src/tgmirror/store/schema.sql +85 -0
- tgmirror-0.1.1/src/tgmirror/store/topicmap.py +34 -0
- tgmirror-0.1.1/src/tgmirror/ui/__init__.py +0 -0
- tgmirror-0.1.1/src/tgmirror/ui/lines.py +99 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/__init__.py +0 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/app.py +225 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/backup_screen.py +141 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/context.py +35 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/debug.py +46 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/frame.py +35 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/prompter.py +425 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/run_screen.py +119 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/screen.py +54 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/screens/__init__.py +0 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/screens/account.py +65 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/screens/backup.py +54 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/screens/channels.py +51 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/screens/clone.py +52 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/screens/config.py +67 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/screens/history.py +159 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/screens/info.py +22 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/screens/login.py +32 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/screens/main_menu.py +124 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/screens/restore.py +51 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/screens/resume.py +146 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/screens/status_dashboard.py +202 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/screens/wizard.py +117 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/task_screen.py +100 -0
- tgmirror-0.1.1/src/tgmirror/ui/menu/widgets.py +95 -0
- tgmirror-0.1.1/src/tgmirror/ui/messages.py +1225 -0
- tgmirror-0.1.1/src/tgmirror/ui/progress.py +233 -0
- tgmirror-0.1.1/src/tgmirror/ui/prompts.py +100 -0
- tgmirror-0.1.1/src/tgmirror/ui/tables.py +73 -0
- tgmirror-0.1.1/src/tgmirror/ui/tui.py +160 -0
- tgmirror-0.1.1/tests/__init__.py +0 -0
- tgmirror-0.1.1/tests/conftest.py +73 -0
- tgmirror-0.1.1/tests/fakes.py +730 -0
- tgmirror-0.1.1/tests/integration/__init__.py +0 -0
- tgmirror-0.1.1/tests/integration/test_backup.py +417 -0
- tgmirror-0.1.1/tests/integration/test_pushdown_equivalence.py +154 -0
- tgmirror-0.1.1/tests/integration/test_restore.py +382 -0
- tgmirror-0.1.1/tests/integration/test_runner.py +913 -0
- tgmirror-0.1.1/tests/integration/test_runner_analysis.py +156 -0
- tgmirror-0.1.1/tests/integration/test_runner_filters.py +319 -0
- tgmirror-0.1.1/tests/integration/test_runner_flood.py +435 -0
- tgmirror-0.1.1/tests/integration/test_runner_retry.py +287 -0
- tgmirror-0.1.1/tests/integration/test_runner_reupload.py +838 -0
- tgmirror-0.1.1/tests/integration/test_runner_split.py +385 -0
- tgmirror-0.1.1/tests/integration/test_runner_topics.py +185 -0
- tgmirror-0.1.1/tests/unit/__init__.py +0 -0
- tgmirror-0.1.1/tests/unit/test_appdata.py +213 -0
- tgmirror-0.1.1/tests/unit/test_architecture.py +71 -0
- tgmirror-0.1.1/tests/unit/test_auth_flow.py +133 -0
- tgmirror-0.1.1/tests/unit/test_backup_reader.py +174 -0
- tgmirror-0.1.1/tests/unit/test_backup_screen.py +124 -0
- tgmirror-0.1.1/tests/unit/test_backup_tui.py +129 -0
- tgmirror-0.1.1/tests/unit/test_backupdir.py +302 -0
- tgmirror-0.1.1/tests/unit/test_cli.py +35 -0
- tgmirror-0.1.1/tests/unit/test_cli_app_menu.py +142 -0
- tgmirror-0.1.1/tests/unit/test_cli_appdata.py +157 -0
- tgmirror-0.1.1/tests/unit/test_cli_auth.py +234 -0
- tgmirror-0.1.1/tests/unit/test_cli_backup.py +330 -0
- tgmirror-0.1.1/tests/unit/test_cli_channels.py +118 -0
- tgmirror-0.1.1/tests/unit/test_cli_clone.py +628 -0
- tgmirror-0.1.1/tests/unit/test_cli_config.py +82 -0
- tgmirror-0.1.1/tests/unit/test_cli_doctor.py +199 -0
- tgmirror-0.1.1/tests/unit/test_cli_errors.py +76 -0
- tgmirror-0.1.1/tests/unit/test_cli_filters.py +588 -0
- tgmirror-0.1.1/tests/unit/test_cli_restore.py +325 -0
- tgmirror-0.1.1/tests/unit/test_cli_retry.py +372 -0
- tgmirror-0.1.1/tests/unit/test_cli_reupload.py +576 -0
- tgmirror-0.1.1/tests/unit/test_cli_run.py +1015 -0
- tgmirror-0.1.1/tests/unit/test_cli_topics.py +56 -0
- tgmirror-0.1.1/tests/unit/test_cli_warnings.py +54 -0
- tgmirror-0.1.1/tests/unit/test_config.py +99 -0
- tgmirror-0.1.1/tests/unit/test_config_limits.py +102 -0
- tgmirror-0.1.1/tests/unit/test_config_save.py +83 -0
- tgmirror-0.1.1/tests/unit/test_endpoints.py +232 -0
- tgmirror-0.1.1/tests/unit/test_fake_gateway.py +160 -0
- tgmirror-0.1.1/tests/unit/test_filter_matcher.py +293 -0
- tgmirror-0.1.1/tests/unit/test_filter_model.py +206 -0
- tgmirror-0.1.1/tests/unit/test_filter_parser.py +187 -0
- tgmirror-0.1.1/tests/unit/test_filter_pushdown.py +111 -0
- tgmirror-0.1.1/tests/unit/test_limiter.py +381 -0
- tgmirror-0.1.1/tests/unit/test_menu_backup_restore.py +308 -0
- tgmirror-0.1.1/tests/unit/test_menu_badge.py +141 -0
- tgmirror-0.1.1/tests/unit/test_menu_channels.py +56 -0
- tgmirror-0.1.1/tests/unit/test_menu_config.py +104 -0
- tgmirror-0.1.1/tests/unit/test_menu_keys.py +96 -0
- tgmirror-0.1.1/tests/unit/test_menu_prompter.py +317 -0
- tgmirror-0.1.1/tests/unit/test_menu_screens_esc.py +53 -0
- tgmirror-0.1.1/tests/unit/test_menu_widgets.py +84 -0
- tgmirror-0.1.1/tests/unit/test_menu_wizard.py +197 -0
- tgmirror-0.1.1/tests/unit/test_messages.py +56 -0
- tgmirror-0.1.1/tests/unit/test_paths.py +55 -0
- tgmirror-0.1.1/tests/unit/test_planner.py +371 -0
- tgmirror-0.1.1/tests/unit/test_pool.py +361 -0
- tgmirror-0.1.1/tests/unit/test_progress_interrupt.py +220 -0
- tgmirror-0.1.1/tests/unit/test_reconcile.py +79 -0
- tgmirror-0.1.1/tests/unit/test_resume_screen.py +72 -0
- tgmirror-0.1.1/tests/unit/test_reupload.py +557 -0
- tgmirror-0.1.1/tests/unit/test_run_screen.py +136 -0
- tgmirror-0.1.1/tests/unit/test_runtime.py +52 -0
- tgmirror-0.1.1/tests/unit/test_secrets.py +125 -0
- tgmirror-0.1.1/tests/unit/test_session_lock.py +45 -0
- tgmirror-0.1.1/tests/unit/test_status.py +412 -0
- tgmirror-0.1.1/tests/unit/test_store.py +1005 -0
- tgmirror-0.1.1/tests/unit/test_telethon_gateway.py +584 -0
- tgmirror-0.1.1/tests/unit/test_telethon_messages.py +515 -0
- tgmirror-0.1.1/tests/unit/test_telethon_pin.py +145 -0
- tgmirror-0.1.1/tests/unit/test_telethon_pool.py +686 -0
- tgmirror-0.1.1/tests/unit/test_telethon_restore.py +534 -0
- tgmirror-0.1.1/tests/unit/test_telethon_reupload.py +688 -0
- tgmirror-0.1.1/tests/unit/test_telethon_transfer.py +254 -0
- tgmirror-0.1.1/tests/unit/test_transfer.py +229 -0
- tgmirror-0.1.1/tests/unit/test_tui.py +247 -0
- tgmirror-0.1.1/tests/unit/test_unit.py +48 -0
- tgmirror-0.1.1/tests/unit/test_wizard.py +42 -0
- tgmirror-0.1.1/uv.lock +1034 -0
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: checkpoint-state
|
|
3
|
+
description: Durable state rules for tgmirror runs and mirrors — SQLite schema, write-ahead pending rows, one-transaction-per-batch commits, resume/reconcile after a crash, delta via the pair's cursor, and pause/stop control. Also covers the backup log (`backups` table, Phase 11a), whose actual progress lives in the backup directory, not in a transactional batch, and a restore (Phase 11b), which is an ordinary run whose *source* reads come from that directory instead of the gateway. Use when touching store/*, engine/runner.py, engine/runs.py, engine/backup.py, engine/backup_reader.py, the run/retry/history/backup/restore commands, or any code that changes clone, backup or restore progress.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Checkpoint & state
|
|
7
|
+
|
|
8
|
+
Source doc: `docs/04-state-checkpoint.md` (schema, rationale). This skill is the rulebook for changing it safely.
|
|
9
|
+
|
|
10
|
+
## Model in one paragraph
|
|
11
|
+
|
|
12
|
+
One SQLite file (WAL) with two things and only one of them visible. A **run** (`runs`) is one execution of `clone`/`run`: status, this run's own counters (`stats_json`), the filter it used, `cursor_from`/`cursor_to`, `control` flag, heartbeat (`updated_at`), `fail_reason`/`resume_at`. Runs are the log (`tgmirror history`). A **mirror** (`mirrors`) is the checkpoint of a source→destination pair, never shown to the user: `cursor_src_id` (highest source id fully finished), the remembered `filters_json`, `options_json` (`dst_base_id`), `account`, `src_kind`/`dst_kind` (schema v2, phase 8 — no longer forced equal, `docs/01-kien-truc.md`). `msg_map` hangs off the mirror (`mirror_id`) and maps `src_msg_id → dst_msg_id` with status `pending|done|failed|skipped` (`skipped` = unsupported type such as game/invoice/quiz, or a poll left out because `--reset-polls` was not given, with `reason='unsupported:<kind>'` (and `dst_msg_id` = the placeholder text that stands for it, if `--placeholder` posted one; counted in `stats.skipped_unsupported`), or a failed message a retry found deleted at the source, `reason='gone_from_source'`; `retry` ignores it) and the `run_id` that last *settled* the row (`finish_batch`/`confirm_pending`/`mark_gone`; a write-ahead row keeps the old one). Why the checkpoint stays: delta needs the cursor, crash safety needs write-ahead rows and `dst_base_id`, and neither can be derived from a log. Filter-skipped messages are **not** stored — they only advance the cursor and the run's `skipped_filter` counter (credited per batch: `Batch.skipped`/`Batch.upto`, committed with the batch or via `advance_cursor(extra_stats=...)`; the batcher flushes a batch after 500 skips so long stretches still save progress and hear pause/stop). Forum pairs also keep `topic_map` (`mirror_id`, `src_topic_id → dst_topic_id`, filled lazily by `engine/topics.py::TopicResolver` the first time each topic is seen) and `msg_map.src_topic_id`; creating a destination topic and saving its `topic_map` row happen back to back with nothing in between (**not** one literal SQL transaction spanning the Telegram call — rule 5 forbids that; a crash in that narrow window just means the next run recreates the topic once more). `flood_log` (by `run_id`) and `limiter_state` support tuning and persistence (`limiter_state` holds the limiter's `delay`, local `day` and `sent_today` per `mirrors.account`; `commit_batch(limiter=...)` writes it in the batch's transaction, `save_limiter_state` after a flood). `options_json` is `RunOptions` (`store/runs.py`): `batch_size`, `dst_base_id` (the destination's newest message id when the pair was first cloned, recorded once on the mirror and copied to each run, so reconcile never scans an old destination) and `pushdown` (`false` = read everything, filter locally), plus the strategy B options (`caption`, `caption_text`, `reset_polls`, `ignore_unsupported`, `placeholder`, `protected_ack`: kept on the mirror and inherited by `run`/`retry`; older JSON reads as the defaults) and three keys that belong to one run and never to the mirror (`RunOptions.for_pair` drops them): `src_last_id` (the source's newest message id when the run began, the total that `status` measures progress and ETA against), `retry_of` (set by `tgmirror retry`: the run whose `failed` messages this run sends again), and `from_backup` (Phase 11b: the backup directory this run reads from instead of a live source — a run-only property because the same pair can go back to being driven by a live `clone`/`run` after a restore). Unknown keys are ignored. `filters_json` is the canonical `FilterSpec.to_json()` (`{}` = no filter). `Store` methods take the `run_id` doing the work and find its mirror themselves.
|
|
13
|
+
|
|
14
|
+
## Rules
|
|
15
|
+
|
|
16
|
+
1. **Write-ahead**: before calling Telegram for a batch, insert its messages as `pending` (with `batch_id`) and commit.
|
|
17
|
+
2. **One transaction per batch result**: update `msg_map` rows (`done`+`dst_msg_id` / `failed`+`reason`), `mirrors.cursor_src_id`, `runs.cursor_to`, `runs.stats_json`, `limiter_state`, `updated_at` together. Never split these.
|
|
18
|
+
3. `cursor_src_id` is monotonic and only advances past a batch with no `pending` rows left. Never move it backwards, except in `Store.start_run` — (a) when `fresh=True` (`--fresh`): after the busy check, in one transaction, every `msg_map` row of the pair is deleted, the cursor goes to 0 and the new `dst_base_id` is stored (`StartedRun.forgot` = done rows forgotten; `Store.count_copied` lets the CLI warn first) — and (b) when the filter differs from the remembered one (`clone` with another filter, or `--no-filter`): one transaction sets the new `filters_json` and cursor 0 and opens the run with `cursor_from = 0`; it relies on `msg_map` to skip `done` items, and `skipped_filter` starts over because counters are per run. It refuses (`RunBusy`) a pair another process is running.
|
|
19
|
+
3b. **Discarding `pending` rows never forgets a message that had failed.** A write-ahead row over a `failed` one keeps its `reason` (and `run_id`); `msgmap.delete_pending` (behind `discard_batch`/`discard_pending`) puts a row that still has a `reason` back to `failed` and deletes the rest. A failed message is below the cursor, so nothing else reads it again: deleting the row silently loses it. This bites only `retry`, and only when the batch is refused, cancelled or found unsent by reconcile.
|
|
20
|
+
4. The store API (`Store` in `store/db.py`) exposes intent-level methods (`start_run`, `begin_batch`, `commit_batch`, `discard_batch`, `confirm_pending`, `finish`, `set_control`, `set_status`, ...), not raw SQL, to the engine. Raw SQL stays inside `store/` (a test enforces it). Every `Store` method takes its `asyncio.Lock`: the runner and its heartbeat task share one connection.
|
|
21
|
+
5. Use `BEGIN IMMEDIATE` for writes; keep transactions short; never hold one across a Telegram call.
|
|
22
|
+
6. Migrations: `schema.sql` is version 1 (the runs + mirrors redesign was folded into it before the first release, with no migration from the old `jobs` tables), and `PRAGMA user_version` holds the version; every schema change appends a numbered script to `default_migrations()` in `store/db.py` (never edit an old one) and gets a test that upgrades a DB created by the previous version (see `test_upgrading_a_database_made_by_the_previous_version`).
|
|
23
|
+
7. Timestamps are UTC ISO-8601 strings; `limiter_state.day` uses local date (documented, because "daily cap" is a user-facing concept).
|
|
24
|
+
|
|
25
|
+
## Resume algorithm (must stay in this order)
|
|
26
|
+
|
|
27
|
+
1. `engine/runs.py::begin_run` → `Store.start_run` opens the run and takes the pair (one transaction: status `running` + heartbeat; refused with `RunBusy` while another run of the pair is `running`/`paused` with a heartbeat younger than 2 minutes, unless `--force-takeover`; a run with an older heartbeat is closed as `failed('interrupted')`). It also refuses, before connecting, a pair whose latest run is `waiting_flood` before `resume_at` or `failed('peer_flood')` within 24 h (`check_runnable`).
|
|
28
|
+
2. Reconcile any `pending` rows (`engine/reconcile.py` decides, `engine/runner.py` reads and writes):
|
|
29
|
+
- re-read the pending source messages (media kind, album structure) — for a restore (Phase 11b) this goes through `Runner._reader` (`reader_override` when set), never the gateway — and the destination tail, always through a second, live reader (`Runner._dst_reader`): non-service messages with id > `max(dst_msg_id of done rows, options.dst_base_id)`;
|
|
30
|
+
- count, media kinds and album structure all match ⇒ `confirm_pending`: mark `done` with the found dst ids and advance the cursor, one transaction;
|
|
31
|
+
- no new dst messages ⇒ delete the `pending` rows (the batch will be re-sent);
|
|
32
|
+
- anything else (including a pending message that vanished from the source) ⇒ warn, delete and re-send (favour "no gap" over "no duplicate").
|
|
33
|
+
3. Iterate `iter_messages(src, min_id=run.cursor_from)` (ascending) with the run's filters (`plan_read`: pushdown, or a full scan with `options.pushdown = false`).
|
|
34
|
+
4. Skip any unit that already has a `done` row; a batch with nothing left only advances the cursor (and books its filter skips).
|
|
35
|
+
|
|
36
|
+
## What a copy call leaves behind
|
|
37
|
+
|
|
38
|
+
- Returned normally: `dst_id` ⇒ `done`; `None` ⇒ `failed('not_copied')` (Telegram made no message; not "unknown").
|
|
39
|
+
- `PerMessage`: delete the batch's `pending`, resend unit by unit; a unit that still fails ⇒ `failed`.
|
|
40
|
+
- `FloodWait` within `max_auto_wait`: wait and repeat the same call, `pending` stays. `FloodWait` too long or 5 in a row, `PeerFlood`, `NoPermission`, `ForwardsRestricted`, other rejections, stop/Ctrl+C during a flood wait: nothing was created ⇒ delete the batch's `pending`, then end the run. A *pause* during a flood wait does not cut it short: it is honoured after the wait, at the batch boundary. The daily cap fires before the write-ahead (nothing pending): run `waiting_flood`, `fail_reason='daily_cap'`.
|
|
41
|
+
- Strategy B: one unit per batch, so a pending batch is one unit and reconcile compares it as usual. `prepare` runs **before** the write-ahead, so a refused fetch (`PerMessage`) leaves nothing pending: the unit is written `failed` at once. A unit left out on purpose (`DROP`) is recorded with `begin_batch` + `commit_batch` and no Telegram call between. A big single file goes up **before** the write-ahead (`upload_prepared`, no message exists, nothing pending; a kill there leaves nothing to reconcile), then `pace`, write-ahead, and `send_prepared` only posts; a `FloodWait` on the post repeats it with the file already uploaded, on the upload it repeats the upload. Albums and small files still upload inside `send_prepared`. A `FloodWait` while sending repeats the same call with the same downloaded files. The daily cap is checked before the upload (`check_cap`).
|
|
42
|
+
- `Transient` (connection cut after sending): outcome unknown ⇒ **keep** `pending`, run `failed('transient')`; the next run reconciles.
|
|
43
|
+
|
|
44
|
+
## Delta
|
|
45
|
+
|
|
46
|
+
Cloning the same pair again (`clone` or `run`) is a new run on the same mirror: the stored filter and cursor are used (no filter given ⇒ the remembered one), no new messages ⇒ exit 0 quickly, the counters are the new run's own. Another filter ⇒ cursor 0 and skip-`done` (rule 3). Edit/delete sync is out of scope for v1 — do not half-implement it.
|
|
47
|
+
|
|
48
|
+
## Retry
|
|
49
|
+
|
|
50
|
+
`tgmirror retry [n]` starts an ordinary run with `options.retry_of = n` (`RunRequest.retry_of`). Only the `Unit` source of the runner differs (`Runner._failed_units`): the `failed` ids of run `n` (`run_failures(n)`, snapshot after reconcile), read by id with `get_messages` in chunks of 100 (`planner.failed_units`: albums kept whole, only their failed members; `Gone` for ids deleted at the source → `Store.mark_gone`), then the same gate → `pace` → write-ahead → `guard.write` → `commit_batch`. The source cursor and the filter are untouched (`MAX` keeps the cursor; the read never sees the filter). Rows sent again get `run_id` = the retry, so a message that fails again belongs to it and `retry` without a number goes on; a stopped retry is continued by `retry n`, not `run` (`execute` prints the right hint). Never send a `done` row again.
|
|
51
|
+
|
|
52
|
+
## Backup (Phase 11a)
|
|
53
|
+
|
|
54
|
+
`tgmirror backup` (`store/backups.py`, `engine/backup.py`) is the one deliberate exception to write-ahead: its `backups` row (schema v3, no `mirrors`/`msg_map` underneath it) exists only for `history`, the heartbeat/control for `pause`/`stop` from another terminal, and what `flood_log.backup_id` points at (kept separate from `flood_log.run_id`, since `runs`/`backups` number their rows independently). The *authoritative* progress is the backup directory itself (`engine/backupdir.py`): a message is durably backed up once its media file(s) are written and its `messages.jsonl` line is appended after them, and appending a JSONL line is safe to repeat — so there is nothing to write ahead of. A crash can only ever leave the last JSONL line incomplete; `backupdir.iter_records`/`last_id` simply stop there, and the next run re-fetches that one message. `Store.advance_backup` still moves `backups.cursor_to`/`stats_json` forward after each unit, but purely for display — resuming reads `backupdir.last_id`, never this row. The filter is fixed on a directory's first run (`engine.backup.FiltersChanged`): unlike a mirror, a backup keeps no `msg_map`-equivalent record of what an old filter skipped, so it cannot safely restart with a new one.
|
|
55
|
+
|
|
56
|
+
`BackupWriter._analyze` (mirrors `Runner._analyze`) sets `stats_json["total_items"]` once near the top of every `run()` call via `Store.set_backup_total`, which *overwrites* that one key instead of `advance_backup`'s additive merge — also purely for display (the same un-paced-but-flood-aware `reader.count()` a run's analysis uses, now also available to a backup's `FloodOwner`), never read back to decide what to do next. Recomputed unconditionally on every call, not gated on "already analyzed": a backup row is a fresh insert per `begin_backup`, so there is no same-row resume case to protect against.
|
|
57
|
+
|
|
58
|
+
## Restore (Phase 11b)
|
|
59
|
+
|
|
60
|
+
`tgmirror restore` is **not** a new persistence mechanism: it is an ordinary run, using `mirrors`/`msg_map`/write-ahead/reconcile exactly as above. Two things differ:
|
|
61
|
+
|
|
62
|
+
- `RunOptions.from_backup` (a run-only key, see above) carries the backup directory; `engine/runs.py::begin_run` reads `backupdir.read_manifest`/`last_id` from it instead of calling `gateway.get_channel`/`last_message_id(src)`, and `check_source_from_backup` (D3) has only the flag/typed-statement path — no live channel is left to test `is_admin` against.
|
|
63
|
+
- `mirrors.src_id` is the *original* channel's id (`BackupManifest.src_id`), which may no longer exist. Since the unique index on `mirrors` is `(src_id, dst_id)` only, never `mode`, a restore and a live `clone`/`run` of that same original channel into the same destination share one mirror/`msg_map` — a message `done` by one is skipped by the other, because Telegram ids survive a backup unchanged.
|
|
64
|
+
|
|
65
|
+
`Runner(reader_override=...)`: when set, source reads (`planner.units`, `TopicResolver`, `fetch`/`prepare`, retry's `get_messages`) use it instead of `guard.reader(gateway)`; the destination-tail read inside reconcile always uses a second, live reader (`Runner._dst_reader`) regardless — see the resume algorithm above. `Prepared.files` stays `()` for anything read this way, so the engine's usual cleanup (`Pipeline.finish`, `send_unit_by_reference`) deletes nothing that belongs to the backup directory.
|
|
66
|
+
|
|
67
|
+
## Control
|
|
68
|
+
|
|
69
|
+
- `tgmirror pause|stop|run` (another terminal) only write `runs.control` (`pause`, `stop`, `none`), and only on a `running`/`paused` run. The runner polls it between batches (one cheap `SELECT`) and, for stop, during sleeps (`Interrupted`).
|
|
70
|
+
- The keys `p`/`r`/`q` and Ctrl+C reach the same runner through `RunControl` (`engine/runner.py`; thread-safe). A resume request also clears a `pause` left in the store.
|
|
71
|
+
- **Stop / Ctrl+C**: finish the current batch, commit, `finish(STOPPED)`, clear `control`. A second Ctrl+C exits immediately; resume must cope via reconcile.
|
|
72
|
+
- **Pause is in place** (`Runner._hold`): after the current batch the run becomes `paused` (the process, its heartbeat and the terminal stay alive, nothing is sent) until resumed or stopped. It is honoured at batch boundaries, never by cutting a sleep short. A `paused` run still holds its pair (`RunBusy` for a second run).
|
|
73
|
+
|
|
74
|
+
## Tests that must exist (FakeGateway)
|
|
75
|
+
|
|
76
|
+
Strategy B (`tests/integration/test_runner_reupload.py`): one send per unit and never a forward, an album stays one album, files kept until sent and gone afterwards (also after a stop, a kill and a refused fetch), the next unit downloaded while this one uploads and never more than the window allows, a stop noticed while a download is in flight, polls/games/quizzes per flag, a placeholder id kept on the skipped row, kill before/after the upload resumes without a duplicate. Existing since phase 2: `tests/integration/test_runner.py`, `tests/unit/test_store.py`, `tests/unit/test_planner.py`, `tests/unit/test_reconcile.py`. Keep them covering:
|
|
77
|
+
|
|
78
|
+
- Kill the runner before the copy call and before `commit_batch`, at every batch ⇒ resume yields no gaps and no duplicates; duplicates only in the documented ambiguous case.
|
|
79
|
+
- Album never split across a batch, including across a resume.
|
|
80
|
+
- `cursor_src_id` never decreases (except a changed filter in `start_run`); filter-only-skipped stretches still advance it and a stop gets through them; `skipped_filter` is counted once across runs (each run counts its own) and through the `PerMessage` split (`tests/integration/test_runner_filters.py`).
|
|
81
|
+
- Delta after `done` picks up only new messages; the same filter again is delta, another one reads from the start without copying twice.
|
|
82
|
+
- Two runs of one pair: the second refuses (fresh heartbeat); a dead run is logged `failed('interrupted')`; a paused run holds its pair.
|
|
83
|
+
- Pause in place: held after the batch in flight, resumed by `RunControl` or by the store flag, stop while held.
|
|
84
|
+
- Migration from every previous schema version.
|
|
85
|
+
- Retry (`tests/integration/test_runner_retry.py`): only the failed ids are sent, read by id, cursor and filter untouched; kill before/after the copy and a refused (long flood) or stopped retry lose no failed message; albums whole across the 100-id read boundary; deleted-at-source set aside; a retry of a retry.
|
|
86
|
+
- Restore (`tests/integration/test_restore.py`, `tests/unit/test_backup_reader.py`): kill/resume through reconcile with a backup-directory source (catches a reader mixed up between source and destination); a restore and a live clone of the same original source sharing one mirror without duplicating an id either side already sent.
|
|
87
|
+
|
|
88
|
+
## Doc sync
|
|
89
|
+
|
|
90
|
+
Before finishing, run the `doc-sync` skill. Any schema change must update the SQL block in `docs/04-state-checkpoint.md`, the model paragraph here, and add a numbered migration; changes to resume/reconcile order must update both this skill and the doc.
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: cli-wizard
|
|
3
|
+
description: Conventions for the tgmirror CLI — adding Typer commands, wizard (questionary) steps, Rich progress/TUI, the full-screen menu app (`ui/menu/`), exit codes, and keeping interactive and non-interactive paths identical. Covers `tgmirror backup`/`restore` (Phase 11a/11b) as the two commands whose wizard shape differs most from `clone`. Use when touching cli/*, ui/*, or adding/changing any `tgmirror` command, flag or menu screen.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# CLI & wizard
|
|
7
|
+
|
|
8
|
+
Source doc: `docs/02-cli-ux.md` (command table, wizard flow, exit codes, config). Keep it in sync with any change here.
|
|
9
|
+
|
|
10
|
+
## Structure
|
|
11
|
+
|
|
12
|
+
- `cli/app.py`: Typer app, registers commands from `cli/commands/*.py` (one module per command group). Its callback puts a `Runtime` in `ctx.obj`.
|
|
13
|
+
- `cli/runtime.py`: `Runtime` = paths, `connect(config)` (async context manager giving `Connection(auth, gateway)`), prompter, `interactive` (TTY, gates prompting), env, plus `keys` and `reporter` — two more TTY-gated injectables kept separate from `interactive` on purpose: tests set `interactive=True` to drive wizard prompts on a `CliRunner`, which is not a real terminal, so hotkeys and the Rich TUI must not switch on with it (`default_runtime()` wires `terminal_keys`/`terminal_reporter` only when a real terminal is attached; `keys` picks `LineReporter` vs `TuiReporter`, `cli/keys.py`/`ui/tui.py`). Commands use only this, so tests inject fakes with `CliRunner.invoke(app, args, obj=runtime)` (`make_runtime` fixture in `tests/conftest.py`). `authorized(rt)` is the connect-and-require-login helper.
|
|
14
|
+
- `cli/errors.py`: `run(rt, coro)` runs a command's coroutine (`asyncio.run`) and turns `TgMirrorError`s into one sentence + exit code (`describe`, `exit_code`); `UsageProblem(key, **params)` is a usage error (exit 2) that already names its message key. Add a new error type there together with its message.
|
|
15
|
+
- `cli/filter_options.py`: the filter flags as Typer option types, shared by `clone`; `collect(...)` turns them into a `FilterSpec` (or `None` when none was given) before anything else happens, so a bad filter is exit `2` with nothing written. Add a filter flag there once, not per command.
|
|
16
|
+
- `cli/wizard.py`: questionary prompts only. **Wizard functions collect values and return a spec; they contain no business logic and never talk to Telegram directly** (they receive already-fetched data, e.g. the channel list).
|
|
17
|
+
- `ui/`: `messages.py` (all user strings), `prompts.py` (async `Prompter` protocol with `say`/`text`/`path`/`secret`/`confirm`/`select`/`checkbox` + questionary implementation — `path` tab-completes a filesystem path like a shell (`questionary.path`), used for a directory (`backup`/`restore`, `only_directories=True`) or a filter YAML file; async because prompts happen between Telegram calls inside a running loop; `ScriptedPrompter` in `tests/fakes.py`), `tables.py` (Rich table / `--json`), `progress.py` (`LineReporter`: plain lines, no ANSI, throttled: progress as "x/y messages" when the run has a total, and a line per file of 8 MB or more while it downloads or uploads; also the shared `duration()` helper), `tui.py` (`TuiReporter`: the Rich `Live` view, phase 7, done 2026-09-23 — same `Reporter.notice`/`progress`/`transfer` calls as `LineReporter`, used instead of it when `Runtime.reporter` is `terminal_reporter`, see "TUI conventions" below).
|
|
18
|
+
- `cli/interrupt.py`: `stop_on_interrupt` (first Ctrl+C asks the runner to finish the batch and save, exit 130; the second quits at once; yields an `Interruption` whose `hit` tells Ctrl+C from the key `q`, which exits 0). `cli/keys.py`: the hotkeys `p` pause / `r` resume / `q` stop while a clone runs (`apply_key` maps a key onto `RunControl`; `terminal_keys` runs a reader thread with `msvcrt` on Windows or `termios` cbreak on POSIX; `Runtime.keys` injects it, tests use `no_keys` or a scripted provider). Plain keys, because the VS Code terminal swallows Ctrl+P/Ctrl+R. `cli/runtime.py` also has `opened_store(rt)` (SQLite state, migrated on first use). `cli/commands/run.py::execute` is shared by `clone`, `run`, `retry` and `restore` (a retry run gets its own start line and stop hint; `restore` passes `reader_override` so `Runner` reads its source from a backup directory instead of the gateway); `pause`, `stop`, `history` and `status` need no Telegram connection (`status` must work while a clone in another terminal holds the session), and `retry` connects only when there is something to retry. `run` with no number, a terminal, and more than one pair in history asks which to continue (`_pick_target`, `wizard.pick_run`, `Store.list_pairs`, done 2026-09-23); the flag/script equivalent is naming the run number, so the parity rule already holds without a separate `--pair` flag.
|
|
19
|
+
- Business logic lives in `engine/` and `store/`. A command is: parse args → build a spec → call one function → render the result.
|
|
20
|
+
|
|
21
|
+
## The parity rule
|
|
22
|
+
|
|
23
|
+
Every wizard outcome must be expressible with flags/YAML, and both paths call the same `begin_run(...)` (`engine/runs.py`) — or, for `tgmirror backup` (Phase 11a), `begin_backup(...)` (`engine/backup.py`) through `cli/commands/backup.py::BackupFlow`, the same shape as `CloneFlow` minus what backup has no equivalent of (a destination channel, a strategy/mode choice, `--fresh`). `tgmirror restore` (Phase 11b, `cli/commands/restore.py::RestoreFlow`) goes back through `begin_run` too (it is an ordinary run, just with `RunOptions.from_backup` set) — see its own paragraph below. When adding a wizard step:
|
|
24
|
+
|
|
25
|
+
1. Add the flag(s) (and YAML key if it belongs to filters) first.
|
|
26
|
+
2. Add the wizard step that fills the same field.
|
|
27
|
+
3. Add a test that clones via flags and via a scripted wizard (`questionary` can be fed with `pytest` monkeypatch / prompt-toolkit input pipes) and asserts equal specs.
|
|
28
|
+
|
|
29
|
+
`--fresh` (`clone` and `run`) forgets the pair's progress and copies everything again; it can duplicate what the destination holds, so when the pair has copied messages (`Store.count_copied`) the one confirmation says how many will be forgotten (`confirm_fresh` in `cli/commands/run.py`; `--yes` agrees; no terminal and no `--yes` ⇒ usage error `err.fresh_needs_yes`, exit 2, nothing forgotten). A pair with nothing copied asks nothing. The wizard asks "continue or start from scratch" (`wizard.pick_resume`) before the filter step, only for a pair with progress, under the same conditions as the filter step and only when `--fresh` was not given — same value as the flag. `run --fresh` must not be swallowed by the "resume a paused run" shortcut. The filter step asks only when the wizard was needed for the rest too (no `--src`, or no destination) and no filter flag or `--yes` was given; for a pair cloned before it also offers "keep the filter of the previous run" first (answer `None`, like giving no flag). No filter flag ⇒ the remembered filter (delta); a flag or `--filter-file` ⇒ that filter, and if it differs the source is read again from the start; `--no-filter` clears it (usage error together with filter flags). The preview (`--preview/--no-preview`, default: terminal + a filter chosen + no `--yes`) prints only; it never asks its own question.
|
|
30
|
+
|
|
31
|
+
`clone` runs at once: there is **one** confirmation, asked before anything is written (`_confirm_start`: "Clone A → B now?", the destination described as «title» when it will be created). `--yes` skips it; with no terminal it is not asked, except that creating a channel then needs `--yes` (exit 2). There is no "save the job" step and no `--run/--no-run`: a clone is a foreground command like any other, Ctrl+C ends it, `tgmirror run` continues it. `--yes` skips confirmations only; it never skips safety prompts that have their own explicit flag: re-uploading a source that restricts saving content needs `--yes-i-administer-this-channel` (any account) or the prompt (default no; only an admin account), and without a terminal and without the flag it is exit 2 (`clone._confirm_protected`); a non-admin account without the flag is refused (exit 4) — unless interactive, where `CloneFlow._confirm_unadministered` asks them to type that flag's name verbatim instead (refined 2026-09-25, see `telethon-engine`). Every run that relies on the statement prints `warn.responsibility` (see `telethon-engine`, decision D3 changed 2026-09-20). Strategy B has its own options: `--mode auto|copy|reupload`, `--caption keep|strip-links|append|none` + `--caption-text`, `--reset-polls`, `--ignore-unsupported` (game/invoice/unanswered quiz; `--placeholder` implies it and also posts a stub text). They contradict each other in known ways (`engine/runs.py::check_options`, exit 2 before anything is created): a caption change or a strategy B flag with `--mode copy`, `append` without text. Without `--ignore-unsupported`/`--placeholder` a run that meets a game/invoice/unanswered quiz stops there (`UnsupportedMedia`, exit 2) and never skips it silently; a poll without `--reset-polls` is dropped with a warning (`docs/02-cli-ux.md`, "Tin đặc thù"). Wizard step 4 (`wizard.pick_strategy`) always asks the mode (`auto`/`copy`/`reupload`, the values of `--mode`); `copy` ends it, `auto`/`reupload` add one yes/no question (default no) before the caption and flag questions, so an ordinary clone answers two quick prompts. It is asked only when the wizard asked for source and destination too and no option flag was given, and a protected source skips the mode question (reupload is the only way) and goes straight to the details. Phase 8: a forum source paired with a non-forum destination (`Plan.warnings` has `"topic_loss"`) gets one more question in this same step — "keep the topic as a hashtag?" (default yes, `--topic-as-hashtag`) under `auto` (any caption mode: those units are then sent again instead of forwarded, `strategy.router(topic_hashtag=...)`) or `reupload`, or a separate `_confirm_topic_loss` (default no, mirrors `confirm_fresh`'s shape) under `--mode copy`, which cannot add one. When nobody was asked (`--yes`, or flags on a terminal) and topics become hashtags, `warn.topic_as_hashtag` says so instead of `warn.topic_loss`. `CloneFlow.steps()` runs `_read_history` right after the destination, so a pair that cannot run now is refused before any strategy question; a forward-only mode on a protected source raises `ForwardsRestricted` in `_pick_strategy`, before anything is created, and `_may_download` skips the typed D3 statement when the mode flag could never use it. Wizard answers are re-asked one at a time (`wizard._ask_valid`): a refused channel title or filter value does not throw away the other answers. `run`/`retry` carry `topic_as_hashtag` over like every other run option — any new `RunOptions` field must be copied in `run.py`/`retry.py`'s `RunRequest` too.
|
|
32
|
+
|
|
33
|
+
`tgmirror backup` (Phase 11a) mirrors `clone`'s shape with a directory instead of a destination channel: `BackupFlow.steps()` = pick source, pick directory (`wizard.ask_backup_dir` → `wizard.resolve_dir_answer`, T3 Phase 15b: strips shell quotes and expands `~` — `Path.resolve()` alone does not — shared with `ask_restore_dir` and the menu's restore-recovery dialog, `ui/menu/screens/resume.py`; a re-asked `Prompter.path` prompt that tab-completes like a shell in the classic wizard's `QuestionaryPrompter`, via `questionary.path`/prompt_toolkit's `PathCompleter` — the full-screen menu's `MenuPrompter.path`/`PathQuestion` hand-draws the same `GreatUXPathCompleter`'s output instead, Phase 15a, "Full-screen menu" below), directory-plus-existing-check together in one retryable step (T3, Phase 15b: `_check_existing` used to be separate and unaskable — a directory already backing up a different source ended the whole wizard instead of asking for another one; a directory still cooling down from a FloodWait is *not* retried the same way, since a different directory would not answer that question — it still ends the wizard, as before), decision D3 (`_confirm_d3`, fails **immediately** here rather than waiting for `begin_backup` — non-admin without the flag/typed statement is `SourceRestricted` at once, admin declining is `Declined`, matching `plan_endpoints`'s fail-fast shape, not `CloneFlow`'s "ask everything then let `begin_run` refuse" shape), then the filter step (skipped silently, with `backup.filter_kept` said once, when the directory already has one — a backup cannot change its filter on resume, `engine/backup.py::FiltersChanged`), preview and one confirmation (`backup.confirm_start`, `--yes` skips it, never the D3 statement). `--pushdown/--no-pushdown` and `--preview/--no-preview` exist on `backup` for the same reasons they do on `clone`.
|
|
34
|
+
|
|
35
|
+
`tgmirror restore` (Phase 11b) is `backup` read backwards: `RestoreFlow.steps()` = pick directory (`wizard.ask_restore_dir`, the same `Prompter.path` prompt as `backup`'s; `read_manifest` returning `None` is `err.not_a_backup`, checked before anything else, and a directory a live `backup` is still cooling down on is refused the same way `BackupFlow._check_existing` refuses one), pick destination (existing, or "create new" asking for a title [+ about] — `RestoreFlow._ask_destination` calling `wizard.ask_new_channel`, same as `clone`'s destination step; only the flags path, `--dst-new`, still resolves straight to `NewChannelSpec(manifest.src_title, o.about)` with no name to type), decision D3 (`_confirm_d3` — **only** the flag/typed-statement path, no admin-plain-confirm branch: the original channel may be gone, so there is no `is_admin` left to check, unlike `clone`/`backup`), filter (always asked fresh — there is no "keep the filter" choice, because a restore's filter decides what gets read out of the directory, not what continues from a stored cursor), preview and one confirmation (`restore.confirm_start`; creating a new destination still needs `--yes` with no terminal, like `clone --dst-new`). No mode/strategy question (`RunRequest.mode="reupload"` always) and no `--reset-polls` flag (`RunOptions.reset_polls=True` always — a backup keeps no votes, so there is nothing else a poll could do). `tgmirror run`/`retry` continuing a restore build a fresh `BackupReader` from `RunOptions.from_backup` (`cli/commands/run.py::reader_override_for`) and pass it to `execute(..., reader_override=...)`, which threads it into `Runner`.
|
|
36
|
+
|
|
37
|
+
## Adding a command — checklist
|
|
38
|
+
|
|
39
|
+
- [ ] Name is a verb or noun consistent with the table in `docs/02-cli-ux.md`; run commands take an optional run number (`run [n]`, `history [n]`); there are no job ids or names.
|
|
40
|
+
- [ ] `--help` text is concrete, with one example.
|
|
41
|
+
- [ ] Non-interactive friendly: no prompt unless a TTY is attached and `--yes`/all needed flags aren't given; when there is no TTY and info is missing → exit code 2 with the missing flag named.
|
|
42
|
+
- [ ] Exit codes: `0` ok · `1` general error · `2` usage · `3` stopped by flood/peer_flood/daily cap · `4` missing permission · `130` interrupted.
|
|
43
|
+
- [ ] Machine-readable output via `--json` for `status`, `history`, `channels`, `topics`.
|
|
44
|
+
- [ ] Errors are human sentences with the next step ("Bạn không phải admin của kênh nguồn; ..."), not tracebacks. Tracebacks only with `--debug`.
|
|
45
|
+
- [ ] Never prints secrets (`api_hash`, session, phone, login code). Phone/code prompts use hidden input where applicable.
|
|
46
|
+
- [ ] Tests with Typer's `CliRunner` and `FakeGateway`.
|
|
47
|
+
|
|
48
|
+
## TUI conventions
|
|
49
|
+
|
|
50
|
+
- `ui/tui.py::TuiReporter` implements `Reporter` like `LineReporter`; which one runs is `Runtime.reporter`, picked in `default_runtime()` by whether a real terminal is attached — never by `interactive` (see "Structure" above).
|
|
51
|
+
- Rich `Live` layout, matching the mock in `docs/02-cli-ux.md` ("Foreground TUI"): header (run id, pair, mode, delay), a progress bar plus `handled/total (~n%)` and speed/ETA (reused from `engine.status.estimate`, the same function `tgmirror status` uses, so the two never disagree about one run), counts (done/failed/skipped by filter/floods), key hints.
|
|
52
|
+
- Keys: `p` pause, `r` resume, `q` stop — three keys, already wired to `RunControl` by `cli/keys.py` before the TUI existed (see "Structure"); the TUI only displays them in its footer instead of the separate `run.keys_hint` line, it never adds a key of its own. No key sets `delay`; it is shown, never editable.
|
|
53
|
+
- `delay` and the flood count are not part of `Reporter` (they belong to the limiter, which the reporter never sees): `TuiReporter` infers both from the `throttled`/`flood_waiting` notices it already receives, the same ones `LineReporter` prints as a line. A shown `delay` is only as fresh as the last flood; there is no live countdown while one is being sat out, only the one notice line above the panel (as `LineReporter` already does).
|
|
54
|
+
- Non-TTY (`plain_reporter`, the default): `LineReporter`, plain line logging every N seconds, no ANSI.
|
|
55
|
+
- `TuiReporter(limits, run, ...)` takes the run as it stands when the reporter is built (`ReporterFactory = Callable[[Limits, Run], AbstractContextManager[Reporter]]`) and seeds its panel with it immediately (`__enter__` refreshes right away) — do not go back to waiting for the first `progress()` call to fill it in. Bug found on a real `--mode reupload` run (2026-09-23): a unit that is one big file and keeps meeting FloodWait can occupy the *whole* run without a single batch ever committing, and until this seed existed the panel stayed blank the entire time, indistinguishable from the TUI never having started. `transfer()` also refreshes the panel now, for the same reason: a long download between two `notice()`s must not look frozen.
|
|
56
|
+
|
|
57
|
+
## Full-screen menu (`ui/menu/`, Chặng 1, 2 and 3 done 2026-09-24)
|
|
58
|
+
|
|
59
|
+
Bare `tgmirror` on a real terminal (`cli/app.py::_bare_invocation`, `Runtime.interactive`) launches `ui/menu/app.py::launch(rt)` instead of showing help — see `docs/02-cli-ux.md` ("Giao diện full-screen (menu)") and `docs/06-lo-trinh.md` ("Kế hoạch giao diện full-screen (menu)") for the full design and decision history. Not a daemon: the loop is exactly this process's lifetime.
|
|
60
|
+
|
|
61
|
+
- **One `Live(screen=True)`, `auto_refresh=False`.** `MenuApp` owns it and the `Layout` (header/body/footer, `ui/menu/frame.py`); it is the *only* thing that calls `.refresh()` (after every key/tick, `_redraw()`). Rich's own background refresh thread (`auto_refresh=True`, the default) would call `.refresh()` on an independent schedule from a different thread — found by hand (2026-09-24) to make the screen look frozen between keypresses; do not turn it back on.
|
|
62
|
+
- **`Screen` contract** (`ui/menu/screen.py`): `render()` (pure, testable like `TuiReporter.render()` — no terminal needed), `handle_key(key)`, `tick()` (called every `tick_interval` seconds when no key arrives — screens that must refresh on their own, e.g. a run in progress or the status dashboard, override this), and `on_enter()` (runs once right after the app pushes the screen; may fetch data or, like `ResumeScreen` with a single pair, decide immediately to push straight through to the next screen). All four may return a `ScreenResult` (`"stay"`/`"pop"`/`"quit"`/`("quit", code)`/`("push", screen)`); `MenuApp._apply()` applies whatever `on_enter()` returns exactly like a key's result, recursively — **never call `on_enter()` and discard its return value** (a real bug, found by hand 2026-09-24: the single-pair fast path did this, so the run screen was built but never actually pushed or started).
|
|
63
|
+
- **Esc must pop** on every list-driven screen. `SelectList.handle_key` only understands Up/Down/Enter and returns `None` for anything else, including Esc — a screen that forwards every key to its list and treats `None` as "stay" silently swallows Esc unless it checks `key == MenuKey.ESC` first (found by hand 2026-09-24 in `HistoryScreen`, `ResumeScreen`, `_ConfirmLogout`; `ChannelsScreen` already did this for its type-to-filter box — copy that pattern for any new list screen).
|
|
64
|
+
- **A screen that only owns a `TypeToFilter` has no Up/Down at all** — real bug, found by hand 2026-09-27: `ChannelsScreen` filtered but never moved a selection or scrolled, so a joined-channel list longer than the terminal simply overflowed the frame with no way to reach the rest. Any screen that lists more than a handful of items needs a `SelectList` too (Up/Down + wraparound), and, to keep a long one on screen, `ui/menu/widgets.py::Fit(head, body)` — sizes `body`'s `rows` to whatever height the frame's `Layout` region actually has at render time (`console.render_lines`/`ConsoleOptions.height`), the same helper `WizardScreen`'s question rendering already used before this fix pulled it out of `screens/wizard.py` into `widgets.py` to share.
|
|
65
|
+
- **`TaskScreen` (`ui/menu/task_screen.py`, generic over the task's result type) is the pattern for embedding a run or a backup** (`RunScreen`/`BackupScreen` are its only two subclasses, unified there in Phase 15b's L1 after starting as two near-identical screens): the same `RunControl`/`stop_on_interrupt` as `cli/commands/run.py::execute`, but no private `Live` (the reporter — `TuiReporter`/`BackupTuiReporter` — is only ever asked for `.render()`, never entered as a context manager) and no `typer.echo` (the alt-screen would swallow it — lines go into `self._lines`, rendered as `Text`). A subclass supplies only `_run()` (build the `Runner`/`BackupWriter` and await it — wrap it in its own `try`/`except` first if it wants extra debug logging around the task, as `RunScreen` does), `self._reporter` (set in `__init__`) and `_result_lines(final)` (the lines to append on success — `ui/lines.py` has the pure builders both this and the classic CLI call). `p`/`r`/`q` go through `cli/keys.py::apply_key` directly from the app's own key queue — never open a second `terminal_keys` reader thread. A finished task's outcome must be recorded into `_lines` exactly **once** (`TaskScreen`'s `self._reported` flag) — `tick()` runs on every idle timeout for as long as the screen stays on top, so anything appended there without a guard repeats forever (found by hand 2026-09-24). A new screen with the same shape (drive a background task, show `p`/`r`/`q`, report once) should subclass `TaskScreen` rather than copying this loop a third time.
|
|
66
|
+
- **`resume_flow`/`retry_flow_for`** (`cli/commands/run.py`/`retry.py`) are the classic commands' logic *before* `execute()`, factored out so the menu can call them directly inside its own event loop without `typer.Exit`/`asyncio.run` (which the classic `run()` wrapper in `cli/errors.py` still uses for the CLI path, unchanged). Any future screen that starts a command-like flow should follow this split rather than calling the Typer-registered function.
|
|
67
|
+
- **Debug log** (`ui/menu/debug.py`): off unless `TGMIRROR_MENU_DEBUG` names a file; never writes to stdout. Logs loop/redraw/screen-transition timing and `RunScreen`'s crash `message`/`trace` — added to chase a real race in `engine/reupload.py::Pipeline._next()` (found and fixed 2026-09-27, N5, Phase 15b: `poll_interval`'s timeout and the strategy-B download producer's last `put` could land in the same event-loop tick, so cancelling the not-yet-resumed `getter` discarded an item that was still sitting in the queue, raising a generic `RuntimeError` instead of ending cleanly) — kept on by the user's own choice (2026-09-24) as a permanent diagnostic now, not removed just because that bug is fixed. Never logs message content, filenames or secrets (CLAUDE.md rule 6).
|
|
68
|
+
- **Wizards in the frame** (`ui/menu/prompter.py::MenuPrompter`, `screens/wizard.py::WizardScreen`): a flow is a coroutine taking a `MenuPrompter` and returning a `ScreenResult` (usually `("replace", ...)`). Reuse the CLI's flow, never re-implement it: `clone` is `CloneFlow` (`cli/commands/clone.py`), login is `ensure_credentials` + `core.auth.login` with `LoginPromptsOn`. Flows print through an `echo`/`say` callable and raise `Declined` for a "no" — never `typer.echo`/`typer.Exit` (the CLI's `run()` maps `Declined` to `err.aborted`, exit 1). `MenuPrompter.handle_key` is `async` (T3, Phase 15b): every key but one goes straight to the open question's own (sync) `handle_key`, but Tab on a `PathQuestion` runs `complete_path`'s filesystem scan through `asyncio.to_thread` instead — that scan can block on a slow/network path, and calling it straight would have frozen the whole frame, not just this question, for as long as it took.
|
|
69
|
+
- **Esc goes back one question** only through `run_steps(prompter, steps)`: split a multi-question flow into steps that each *assign* their results (a re-run step replays its earlier answers). A step's questions may depend only on answers within it and on earlier steps. The CLI runs the same steps in order. A flow without steps (login) treats Esc as cancel.
|
|
70
|
+
- In a flow, check `prompter.asking` (not `question is not None`) to know a question is open — an answered one stays set until the flow task runs again.
|
|
71
|
+
- Typed characters never reach the debug log (`<char>`); `secret` is never drawn; a `MenuPrompter(private_text=True)` keeps typed text (a phone number) out of the answers list (CLAUDE.md rule 6).
|
|
72
|
+
- The app owns the Telegram connection (`MenuApp.open_connection`/`close_connection`) and opens logged out; screens that need Telegram are only offered with `app.account` set (`app.gateway` raises `NotLoggedIn` otherwise).
|
|
73
|
+
- **"Cấu hình" (`ui/menu/screens/config.py::ConfigScreen`, Chặng 3)**: lists paths (read-only) and every `[limits]` key; Enter on a key pushes a one-question `WizardScreen` (`prompter.text`, old value pre-filled) that calls `core.config.set_limit` directly — the same function `tgmirror config set` uses. No Telegram, so it is offered even logged out. A bad value raises `ConfigError`, which `WizardScreen._outcome()` already turns into an `InfoScreen` like any other flow error — no per-field retry loop needed for a single question.
|
|
74
|
+
- **Not yet in the menu**: nothing (Chặng 1–3 cover every item in "Kế hoạch giao diện full-screen (menu)"'s original brainstorm except the "running elsewhere" badge, tracked in `docs/06-lo-trinh.md`).
|
|
75
|
+
- **Testing a list-driven screen: select by value, never by counting keys.** `MainMenuScreen`'s item list depends on account/pair state and grows over time (Phase 15a added "Backup"/"Restore"); a test that pushes N `MenuKey.DOWN` to reach a fixed item hangs the moment the list changes shape, since the wrong item gets picked and the scripted queue runs dry while the app waits for another key (N1, Phase 15b — three tests hung this way, and on Windows `pytest-timeout`'s thread-based kill takes the whole process with it, losing the rest of the job's results). Instead build a throwaway instance of the screen, await `on_enter()`, read `_list.items` and find the target by its value (see `down_presses_to()` in `tests/unit/test_cli_app_menu.py`), then feed exactly that many `MenuKey.DOWN`.
|
|
76
|
+
- **Run the full suite before committing, not just the changed file.** N1 above was invisible to a scoped `pytest tests/unit/test_cli_app_menu.py::test_x` run during Phase 15a's own development, because the file wasn't touched; only `pytest` over everything (as `docs/06-lo-trinh.md`'s Phase 15b "Kế hoạch thực hiện" now requires at the end of every session) reruns tests that some *other* change quietly broke.
|
|
77
|
+
|
|
78
|
+
## Language and output
|
|
79
|
+
|
|
80
|
+
User-facing strings are Vietnamese by default with English fallbacks kept in one `ui/messages.py` (`t(key, **params)`, `TGMIRROR_LANG=en` switches); code, identifiers and comments are English. A key must exist in both tables (tests check keys, placeholders and that every key used in the source exists). Tests run with `TGMIRROR_LANG=en`.
|
|
81
|
+
|
|
82
|
+
- Real runs force stdout/stderr to UTF-8 (`_use_utf8_output` in `cli/app.py`): on Windows a redirected stream uses cp1252, which crashes Rich and garbles Vietnamese.
|
|
83
|
+
- Wrap user-provided text (channel titles) in `rich.text.Text` before putting it in a Rich table; a title like `[pro]` is otherwise read as markup.
|
|
84
|
+
- Tables must stay readable at 80 columns (the phase 1 `channels` table was unreadable before it was compacted); test with `CliRunner`, which is 80 wide.
|
|
85
|
+
|
|
86
|
+
## Doc sync
|
|
87
|
+
|
|
88
|
+
Before finishing, run the `doc-sync` skill. A new or changed command, flag, wizard step, exit code or config key updates `docs/02-cli-ux.md` (and the README usage block if user-visible), plus this skill's checklist if a convention changed.
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: doc-sync
|
|
3
|
+
description: Keeps tgmirror's docs (docs/*.md, CLAUDE.md) and skills (.claude/skills/*) in sync with reality. Run it at the end of EVERY task, before reporting done — code, refactor, spike, bug fix, config change or docs-only. Also use when a doc and the code disagree, when a spike answers an open question, or when deciding whether a new skill or doc is needed.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# doc-sync
|
|
7
|
+
|
|
8
|
+
Docs and skills are the project's memory. A stale doc is worse than none: the next session trusts it. Every task ends with this pass, even if the answer is "nothing to update".
|
|
9
|
+
|
|
10
|
+
## When to run
|
|
11
|
+
|
|
12
|
+
- **Always**, as the last step of a task, before the final message.
|
|
13
|
+
- Mid-task if you discover the docs are wrong, or a design decision has to change.
|
|
14
|
+
- Other skills point here (`## Doc sync` section at their end). Following them means running this skill, not skipping it.
|
|
15
|
+
|
|
16
|
+
## The pass (5 steps, keep it fast)
|
|
17
|
+
|
|
18
|
+
1. **List what changed.** Files touched, behaviours changed, new/renamed modules, new flags or config keys, changed defaults, new error types, resolved unknowns, new decisions.
|
|
19
|
+
2. **Map to targets** using the table below. One change often maps to several targets.
|
|
20
|
+
3. **Update in place**, minimal diff. Edit the sentence that became false; do not rewrite or reformat whole files.
|
|
21
|
+
4. **Check cross-references**: file names, section names, flag names, config keys, decision ids (D1…) still exist and are spelled the same across docs, skills and `CLAUDE.md`.
|
|
22
|
+
5. **Report** one line in the final message: `Doc-sync: updated <files>` or `Doc-sync: no change needed — <reason>`. Silence is not acceptable.
|
|
23
|
+
|
|
24
|
+
## Change → target map
|
|
25
|
+
|
|
26
|
+
| Changed | Update |
|
|
27
|
+
|---|---|
|
|
28
|
+
| Gateway API, Telethon usage, strategies A/B, error mapping | `docs/01-kien-truc.md`, skill `telethon-engine` |
|
|
29
|
+
| Limiter, defaults, FloodWait/PeerFlood handling, hygiene | `docs/05-chong-flood.md`, skill `flood-safety`; config defaults also in `docs/02-cli-ux.md` |
|
|
30
|
+
| SQLite schema, transactions, resume/reconcile, delta, control flags | `docs/04-state-checkpoint.md`, skill `checkpoint-state` |
|
|
31
|
+
| Filter model, predicates, pushdown, album semantics | `docs/03-filters.md`, skill `filter-dsl`; CLI flags also in `docs/02-cli-ux.md` |
|
|
32
|
+
| Commands, flags, wizard steps, TUI, exit codes, config keys | `docs/02-cli-ux.md`, skill `cli-wizard`, README usage block if user-visible |
|
|
33
|
+
| New/renamed/moved module or package layout | `docs/01-kien-truc.md` (package tree), `CLAUDE.md` (Layout) |
|
|
34
|
+
| A decision D1–D9 changes, or a new decision | `docs/00-tong-quan.md` table + dated entry in `docs/06-lo-trinh.md` decision log (+ affected skills). **Ask the user first** — decisions are theirs |
|
|
35
|
+
| A spike/unknown from `docs/06-lo-trinh.md` is resolved | Record the answer in `docs/06-lo-trinh.md` (tick it, add the finding), then fix every doc/skill that hedged on it (e.g. remove "verify" caveats in `telethon-engine`) |
|
|
36
|
+
| A phase starts or finishes | Status checklist in `docs/06-lo-trinh.md` |
|
|
37
|
+
| New hard rule / invariant / convention | `CLAUDE.md` (Hard rules or Conventions) and the owning skill |
|
|
38
|
+
| New user-visible feature | README (goals/usage), `docs/02-cli-ux.md` |
|
|
39
|
+
| Dependency added/removed/changed | `CLAUDE.md` Stack, `docs/00-tong-quan.md` if it is a decision |
|
|
40
|
+
|
|
41
|
+
If nothing on the table applies and no behaviour, structure, decision or unknown changed, "no change needed" is the correct outcome. Do not invent edits.
|
|
42
|
+
|
|
43
|
+
## Keeping skills healthy
|
|
44
|
+
|
|
45
|
+
- A skill states rules and checklists; **details and rationale live in `docs/`**. Link to the doc instead of copying paragraphs, so there is one place to update.
|
|
46
|
+
- Each SKILL.md `description` must still say accurately *when to use it*. Update it if the skill's scope changed.
|
|
47
|
+
- Remove rules that no longer hold; do not leave them "for history" (history is the decision log in `docs/06-lo-trinh.md` and version control).
|
|
48
|
+
- Keep skills short. If one grows past roughly 150 lines, split by topic or move detail to docs.
|
|
49
|
+
- **New skill** only when there is a recurring area or workflow with non-obvious rules that a future task would otherwise get wrong. Otherwise put the knowledge in `docs/`. When adding one: create `.claude/skills/<name>/SKILL.md` (frontmatter `name`, `description`), add its `## Doc sync` footer, and list it in `CLAUDE.md` → Skills and in the map above.
|
|
50
|
+
- **New doc** only if no existing numbered doc fits; use the next number and add it to the README docs table.
|
|
51
|
+
|
|
52
|
+
## Consistency rules
|
|
53
|
+
|
|
54
|
+
- Code is the truth for *what is*; docs are the truth for *what was decided*. If they disagree, determine which is wrong: a bug → fix code; an undocumented improvement → fix docs; a decision-level conflict → ask the user.
|
|
55
|
+
- Language: `docs/` in Vietnamese; `README.md`, `CLAUDE.md`, skills in English. Keep technical identifiers (flags, config keys, class names) verbatim in both.
|
|
56
|
+
- Never put secrets, real phone numbers, `api_hash`, session strings or real channel ids in docs, skills or examples.
|
|
57
|
+
- Numbers in `docs/05-chong-flood.md` are defaults, not Telegram facts. Any change to a default records *why* (which `flood_log` data or test).
|
|
58
|
+
- Dates in the decision log are absolute (`YYYY-MM-DD`).
|
|
59
|
+
|
|
60
|
+
## Checklist before saying "done"
|
|
61
|
+
|
|
62
|
+
- [ ] Did behaviour, structure, config, defaults, a decision, or an open question change? → targets updated.
|
|
63
|
+
- [ ] Docs and skills that mention it no longer contradict the code or each other.
|
|
64
|
+
- [ ] Cross-references (paths, flags, keys, D-ids, skill names) still valid.
|
|
65
|
+
- [ ] `docs/06-lo-trinh.md` status/spikes/decision log current.
|
|
66
|
+
- [ ] Final message contains the `Doc-sync:` line.
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: filter-dsl
|
|
3
|
+
description: How to add, change or debug message filters in tgmirror — the include/exclude/date/id filter model, predicates (media, hashtag, regex, size, ...), album semantics, server-side pushdown safety, YAML and CLI-flag parsing. Use when touching filters/*, adding a predicate, or changing what the wizard offers for filtering.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Filter DSL
|
|
7
|
+
|
|
8
|
+
Source doc: `docs/03-filters.md` (semantics, predicate table, pushdown table). Update it in the same change as the code.
|
|
9
|
+
|
|
10
|
+
## Semantics to preserve
|
|
11
|
+
|
|
12
|
+
A unit (single message or whole album) is cloned iff
|
|
13
|
+
`global_ok AND (include is empty OR any include-rule matches) AND NOT any exclude-rule matches`.
|
|
14
|
+
Inside a rule, predicates are ANDed; rules are ORed. Service messages are always skipped. `date`/`id` are global.
|
|
15
|
+
|
|
16
|
+
## Layers
|
|
17
|
+
|
|
18
|
+
| File | Job |
|
|
19
|
+
|---|---|
|
|
20
|
+
| `filters/model.py` | pydantic models: `FilterSpec`, `Rule`, ranges, `FilterError`. Validation + normalisation (lowercase hashtags, parse `2GB`, parse dates); `to_json`/`from_json` is the stored form and must round-trip |
|
|
21
|
+
| `filters/parser.py` | `FlagFilters` (what the flags and the wizard collect) → `from_flags`, `from_file` (YAML), `resolve` (mixing `--filter-file` with flags is `FilterMix`) |
|
|
22
|
+
| `filters/pushdown.py` | `plan_read(spec, cursor, pushdown=, content=) → ReadPlan(min_id, ServerFilter, complete_albums)` |
|
|
23
|
+
| `filters/matcher.py` | `Matcher(spec).matches(unit) -> bool`; no I/O (regex has a timeout → `FilterError`) |
|
|
24
|
+
| `engine/planner.py`, `batcher.py`, `preview.py` | apply the matcher to units (`Skip` for the dropped), complete albums after content pushdown, credit skips to batches, sample for the wizard preview |
|
|
25
|
+
|
|
26
|
+
## Adding a predicate — checklist
|
|
27
|
+
|
|
28
|
+
1. Add the field to `model.py` with validation and a normalised canonical form (this is what gets stored in `mirrors.filters_json` and `runs.filters_json`; keep it backwards compatible — old stored JSON must still load, and the *serialised* form must pass the *input* validators: sizes are stored as `"<bytes>B"` because bare numbers are rejected on input).
|
|
29
|
+
2. Implement in `matcher.py`, operating on our `SrcMessage` dataclass (never Telethon types).
|
|
30
|
+
3. Decide pushdown: only if it can **only narrow safely** (never drops a message that matches). If unsure, do not push down.
|
|
31
|
+
4. Parser: YAML key + CLI flag (`FlagFilters`, `cli/filter_options.py`, `docs/02-cli-ux.md` shorthand table) + wizard step (`cli/wizard.py::pick_filters`) if user-facing.
|
|
32
|
+
5. Tests: matcher truth table, album cases (`any`/`all`/`first`), parser round-trip (YAML ↔ flags ↔ stored JSON), and add the filter to `SPECS` in `tests/integration/test_pushdown_equivalence.py` (pushdown-vs-full-scan on random fake channels with albums).
|
|
33
|
+
6. Update `docs/03-filters.md` and, if the wizard changed, `docs/02-cli-ux.md`.
|
|
34
|
+
|
|
35
|
+
## Rules that are easy to get wrong
|
|
36
|
+
|
|
37
|
+
- **Always re-run the client matcher after pushdown.** Telegram `search` is word/prefix based, `filter` types are approximate. Pushdown is an optimisation, not the source of truth.
|
|
38
|
+
- **Pushdown must never split an album.** id bounds carry an `ALBUM_MARGIN`; `media`/`search` drop the non-matching members of an album, so the planner completes each album (`complete_albums`). Never push `contains` (substring vs word search), `document`, `sticker`, `webpage`. `--no-pushdown` (job option) is the escape hatch for checking a real account.
|
|
39
|
+
- A predicate about something the message lacks is false, for `max` too; all predicates of a rule look at the same message; `date` is `[from, to)` UTC and `id` inclusive, both judged on the unit's first message.
|
|
40
|
+
- Hashtags match `MessageEntityHashtag` entities, case-insensitive; do not rely on raw substring `#`. Convert entities to plain data in the gateway so the matcher stays pure.
|
|
41
|
+
- Album semantics: caption/hashtag is usually on one member. Default `album: any` for include; **exclude always uses `any`** (one excluded member excludes the album).
|
|
42
|
+
- Regex: compile once, reject patterns that are invalid at spec-validation time (before the clone starts), and guard against catastrophic backtracking (timeout or a safe engine) — filters may come from shared YAML files.
|
|
43
|
+
- Sizes/durations use explicit units; reject bare ambiguous numbers in size fields.
|
|
44
|
+
- Changing the filter of a pair never rewrites history: cloning it again with another filter (or `--no-filter`) reads from cursor 0 and skips `done` items through `msg_map` (`Store.start_run`); no filter flag reuses the remembered one.
|
|
45
|
+
- Filter-skipped messages are not written to `msg_map` (see `checkpoint-state`); they only bump `skipped_filter` and advance the cursor. This is different from *unsupported* messages (game, invoice, ...), which are recorded as `skipped`.
|
|
46
|
+
- `media` also has `geo contact game invoice`; `topic` (forum only) and `from_user` (group/forum only) are id-only lists (phase 8) — never a username or topic name, so `filters/model.py`/`matcher.py`/`parser.py` stay I/O-free; `tgmirror topics <src>` is where a user looks up an id (`docs/03-filters.md`).
|
|
47
|
+
|
|
48
|
+
## Preview (wizard step 5)
|
|
49
|
+
|
|
50
|
+
`engine/preview.py::sample` reads the first 100 messages of the range (id/date bounds only, no content narrowing, so the sample is not biased) through the matcher and shows matched/scanned plus a few captions; `new` prints it and, on a terminal, asks to save before anything is created. Preview reads count against the read limiter (see `flood-safety`).
|
|
51
|
+
|
|
52
|
+
## Doc sync
|
|
53
|
+
|
|
54
|
+
Before finishing, run the `doc-sync` skill. A new or changed predicate updates the predicate/pushdown tables in `docs/03-filters.md`, the CLI shorthand list there and in `docs/02-cli-ux.md`, and this skill if the layer responsibilities changed.
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: flood-safety
|
|
3
|
+
description: Rate limiting and anti-spam rules for tgmirror — the AIMD limiter, FloodWait/SlowMode/PeerFlood handling, daily caps, jitter and long pauses. Use when touching core/limiter.py, the runner loop, retry logic, batch sizing, or any code that makes Telegram write calls.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Flood safety
|
|
7
|
+
|
|
8
|
+
Source docs: `docs/05-chong-flood.md` (numbers, rationale) and `docs/01-kien-truc.md` (runner loop, error table). Telegram publishes no exact limits: treat every number as a conservative default that `flood_log` data may change. Never write "Telegram allows N per minute" in code comments or docs.
|
|
9
|
+
|
|
10
|
+
## Invariants
|
|
11
|
+
|
|
12
|
+
1. Every Telegram call a `run` (or a backup, Phase 11a) makes goes through the gateway, wrapped by `FloodGuard` (`engine/flood.py`): writes `await guard.pace(cost)` (= `limiter.acquire(cost, "write")`) before the write-ahead and go through `guard.write(...)`; reads go through `guard.reader(gateway)`. A backup only ever reads (no writes, no `daily_cap`): `engine/backup.py::BackupWriter` builds its `FloodGuard` with `owner=FloodOwner.of_backup(backup)` instead of `of_run(run)` — the only difference is what `flood_log`/`limiter_state` row it writes against (`FloodOwner` carries `run_id` xor `backup_id`, never both). A test (`test_architecture.py`) fails if `runner.py` touches the gateway any other way. The user-started one-shot commands (login, channels, `doctor`, `clone`'s `--preview`, `begin_backup`'s D3 re-check + `list_topics`) are the documented exception: a flood there is one sentence and exit code 3, no retry.
|
|
13
|
+
2. The client has `flood_sleep_threshold=0`. Every `FloodWaitError` must reach `limiter.on_flood(seconds)`. Never catch it and `await asyncio.sleep` locally.
|
|
14
|
+
3. A FloodWait retry re-sends the **same batch** (already `pending` in `msg_map`). Never rebuild the batch, never advance the cursor.
|
|
15
|
+
4. `PeerFloodError` ⇒ end the run (`failed`, `peer_flood`), exit code 3. No retry, no delay reset, no "try again in a minute".
|
|
16
|
+
5. Users may slow the tool down but never below `min_delay`, and never disable jitter or the daily cap without an explicit, documented flag that prints a warning.
|
|
17
|
+
6. Sleeping is interruptible: pause/stop/Ctrl+C must work during a flood wait (use `asyncio.wait` on a stop event, not a bare `sleep`).
|
|
18
|
+
|
|
19
|
+
**Where things live:** `core/limiter.py` (`Limiter`: AIMD, jitter, long pause, daily cap, read bucket, throttle; pure, injected `clock`/`sleep`), `engine/flood.py` (`FloodGuard`: logging, waiting, retrying, giving up), `store/limiterstate.py` (persistence; `commit_batch(limiter=...)` saves it with the batch). The runner's `_nap` raises `Interrupted` on pause/stop, so nothing sent after a request. Design and numbers: `docs/05-chong-flood.md`; what was chosen in phase 4: `docs/06-lo-trinh.md`.
|
|
20
|
+
|
|
21
|
+
## Limiter behaviour (spec)
|
|
22
|
+
|
|
23
|
+
- `delay` starts at `min_delay`; on flood `delay = min(delay*2, max_delay)`; after 20 consecutive successes `delay = max(delay*0.9, min_delay)`.
|
|
24
|
+
- Actual wait = `delay * uniform(1-jitter, 1+jitter)`.
|
|
25
|
+
- Every `long_pause_every` messages: sleep `uniform(*long_pause_range)`.
|
|
26
|
+
- Daily cap: when reached, the run ends in `waiting_flood` with `resume_at` = next local midnight and `reason=daily_cap`. This is not an error.
|
|
27
|
+
- ≥3 floods in 10 minutes ⇒ halve `batch_size` and raise `min_delay` temporarily (log it).
|
|
28
|
+
- State (`delay`, `sent_today`, day) is persisted in `limiter_state` so a restart does not forget lessons.
|
|
29
|
+
- Time comes from an injected `clock`. Tests use a fake clock; no real `sleep` in unit tests. `Limiter.acquire` may raise from `sleep` (`Interrupted`): it must record nothing before its sleep returns.
|
|
30
|
+
|
|
31
|
+
## FloodWait handling checklist
|
|
32
|
+
|
|
33
|
+
When adding or changing a code path that can raise flood errors:
|
|
34
|
+
|
|
35
|
+
- [ ] Wrapped by `FloodGuard.write` / `FloodGuard.reader` (do not hand-roll retry loops).
|
|
36
|
+
- [ ] Logs to `flood_log` (method, seconds, delay, batch_size).
|
|
37
|
+
- [ ] `seconds <= max_auto_wait` → notice, sleep `seconds + uniform(1,5)`, retry the same call (a cut read resumes after the last message it gave).
|
|
38
|
+
- [ ] `seconds > max_auto_wait`, or 5 floods in a row on one call → refused batch's `pending` deleted, run `waiting_flood`, `resume_at` set, exit 3 (keep waiting only with `--wait`, never past the 5-in-a-row stop).
|
|
39
|
+
- [ ] SlowMode handled the same way as FloodWait.
|
|
40
|
+
|
|
41
|
+
## Hygiene rules to preserve in code and docs
|
|
42
|
+
|
|
43
|
+
- Batch to reduce call count (forward up to `batch_size` ids per call).
|
|
44
|
+
- One account ⇒ one run at a time. Never parallelise sends across runs or pairs on the same session.
|
|
45
|
+
- Read calls are limited too: set `wait_time`, cache entities. Filters add reads (date-to-id lookups, one unfiltered window per album after content pushdown): they are counted as read requests (`requests` in `_GuardedReader`), so a new read path must add its request count there. `get_messages` (reading the failed messages by id for `retry`) is one paced read request per call (`_GuardedReader.get_messages`). `count` (a run's or a backup's analysis — `Runner._analyze`/`BackupWriter._analyze`) is deliberately *not* paced: one call that opens the run/backup, like the setup reads of `begin_run`/`begin_backup`, so the read bucket still starts with the first page of messages; its FloodWait is logged and sat out like any other (`_GuardedReader.count`).
|
|
46
|
+
- Strategy B (phase 6): `prepare` (re-read + download) is one paced read request through `guard.reader`; the send goes through `guard.write` (`send_prepared`, `send_text`) like any write and counts against `daily_cap` per message. Only the *read* half runs ahead (`Pipeline`, `[limits] prefetch`), never the sends: still one write at a time. The daily cap, not transfer speed, is what usually decides how long a big clone takes: `status` and the first lines of a run say how many days of rest it costs (`cap_days`). There is no size-based extra delay yet (no data; revisit with `flood_log`). A FloodWait while sending repeats the same `send_prepared` with the files already on disk.
|
|
47
|
+
- Sending by file id (`Strategy.REFERENCE`): `reader.fetch` is one paced read request, `send_by_reference` is one write through `guard.write` per unit (counts against `daily_cap`, paced like any write). The stale-reference retry adds one `fetch`. Nothing is transferred, so the pace and the cap are what limits it.
|
|
48
|
+
- File transfers (strategy B) go through a `RequestBudget` per direction (`core/pool.py`, `[limits] download_requests` default 4, `upload_requests` default 8, each starts at 2): it grows a step per 32 clean parts and is halved by any pushback. Download and upload used to share one budget (`max_requests`) until 2026-09-23, when a real `--mode reupload` run hit repeated transport 429s and a dead connection downloading at 8 in flight while uploading at 8 stayed fine in the same run — proof the two directions break at different points and need separate ceilings, not evidence against 8 in general (see `upload_requests` below, still 8). The transport-level `HTTP 429` is *not* a `FloodWaitError`: the pool turns it into `FloodWait(60, transport=True)` (logged as `transport_429`, sat out, the transfer repeated; a download resumes from the parts it has), never a retry within seconds. Bulk data has connections of its own, never the main one. Every time a budget shrinks it logs a warning (`tgmirror.core.pool`, printed by the CLI as `[cảnh báo] ...`): a transfer that suddenly crawls is usually a budget stuck near 1. Making fewer connections than asked for (`upload_connections`) is warned too, once (`tgmirror.core.telethon_gateway`): one upload connection runs at ~3 MB/s, two or more at 18-28, so never let that fallback be silent again. `upload_connections=8` / `upload_requests=8` (one request per connection), one file at a time with nothing downloading alongside, came back clean **twice in a row** (~24-28 MB/s) before the split, and going past either number broke both times (docs/06-lo-trinh.md, 2026-09-23) — do not raise it past 8 on a guess; that combination is the one that has actually been reproduced. `download_requests=4` has not had the same repeated-measurement treatment yet: it is a direct response to the one real failure, not a bisected optimum like upload's 8. Individual parts are not paced or counted against `daily_cap`; the budget is their limit. A `FloodWait` from a part ends the transfer and reaches `FloodGuard` like any other.
|
|
49
|
+
- Pace is the gap between two *posts*, not between two uploads (2026-09-20): a big single file goes up first through `guard.transfer("upload_prepared")` (not paced, not counted), the time it took is passed to `guard.pace(cost, credit)` and counts against the **delay** only (never a long pause), and `check_cap` runs before the upload. Do not give the copy, by-id or album paths a credit: the runner's `mono` clock is what tests pin to zero so no test earns credit by accident.
|
|
50
|
+
- Default order is chronological (D4). Any reordering option must be opt-in.
|
|
51
|
+
- `tgmirror doctor` and README must say: risk is reduced, not eliminated; use an established account.
|
|
52
|
+
|
|
53
|
+
## Tests that must exist
|
|
54
|
+
|
|
55
|
+
- Fake gateway raises `FloodWait(30)` once → limiter delay doubles, batch retried once, no duplicate `msg_map` rows.
|
|
56
|
+
- `FloodWait(max_auto_wait+1)` → run `waiting_flood`, `resume_at` set, exit code 3.
|
|
57
|
+
- `PeerFlood` → run `failed`, no retry.
|
|
58
|
+
- 20 successes → delay decays but never below `min_delay`.
|
|
59
|
+
- Stop event during a flood sleep → returns promptly, state saved.
|
|
60
|
+
- Daily cap reached → `waiting_flood`, resumes next day (fake clock).
|
|
61
|
+
|
|
62
|
+
## Doc sync
|
|
63
|
+
|
|
64
|
+
Before finishing, run the `doc-sync` skill. Typical updates from this area: changed defaults or new `[limits]` keys (`docs/05-chong-flood.md` **and** `docs/02-cli-ux.md` config block, with the reason for the change), new error handling rows in `docs/01-kien-truc.md`, and this skill's spec section if behaviour changed.
|