blun-king-cli 9.1.567 → 9.1.569
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/agent-spine-plugin/.claude-plugin/marketplace.json +1 -1
- package/agent-spine-plugin/.claude-plugin/plugin.json +1 -1
- package/agent-spine-plugin/.codex-plugin/plugin.json +2 -1
- package/agent-spine-plugin/CHANGELOG.md +1581 -0
- package/agent-spine-plugin/README.md +30 -4
- package/agent-spine-plugin/blun.plugin.json +3 -3
- package/agent-spine-plugin/docs/acceptance.md +61 -0
- package/agent-spine-plugin/docs/assignment-continuation.md +48 -0
- package/agent-spine-plugin/docs/host-integration.md +178 -0
- package/agent-spine-plugin/docs/preflight-recall.md +69 -0
- package/agent-spine-plugin/docs/preservation-contract.md +53 -0
- package/agent-spine-plugin/docs/quality-gates.md +50 -0
- package/agent-spine-plugin/docs/releasing.md +85 -0
- package/agent-spine-plugin/docs/session-timeline.md +251 -0
- package/agent-spine-plugin/docs/source-roots.md +113 -0
- package/agent-spine-plugin/docs/structured-completion.md +67 -0
- package/agent-spine-plugin/docs/world-model.md +94 -0
- package/agent-spine-plugin/hooks/codex.json +2 -2
- package/agent-spine-plugin/hooks/hooks.json +1 -1
- package/agent-spine-plugin/hooks/version.json +1 -1
- package/agent-spine-plugin/package.json +13 -3
- package/agent-spine-plugin/scripts/check-codex-install.js +226 -0
- package/agent-spine-plugin/scripts/check-hosts.js +14 -5
- package/agent-spine-plugin/scripts/check-install-hook.js +226 -0
- package/agent-spine-plugin/scripts/check-install-selfstarter.js +154 -0
- package/agent-spine-plugin/scripts/check-install.js +478 -0
- package/agent-spine-plugin/scripts/check-line-budget.js +58 -0
- package/agent-spine-plugin/scripts/check-syntax.js +29 -0
- package/agent-spine-plugin/scripts/github-actions.js +11 -0
- package/agent-spine-plugin/scripts/hermetic-process.js +183 -0
- package/agent-spine-plugin/scripts/release-check.js +145 -0
- package/agent-spine-plugin/scripts/run-acceptance.js +19 -0
- package/agent-spine-plugin/scripts/run-checks.js +47 -0
- package/agent-spine-plugin/scripts/run-tests-hermetic.js +89 -0
- package/agent-spine-plugin/skills/agent-spine/SKILL.md +16 -2
- package/agent-spine-plugin/spine-example/1-identity.md +12 -0
- package/agent-spine-plugin/spine-example/2-voice.md +6 -0
- package/agent-spine-plugin/spine-example/3-conduct.md +8 -0
- package/agent-spine-plugin/spine-example/4-history.md +4 -0
- package/agent-spine-plugin/src/cli-agent.js +296 -0
- package/agent-spine-plugin/src/cli-attention.js +95 -0
- package/agent-spine-plugin/src/cli-autonomy.js +36 -0
- package/agent-spine-plugin/src/cli-common.js +71 -0
- package/agent-spine-plugin/src/cli-continuity.js +116 -0
- package/agent-spine-plugin/src/cli-core.js +128 -0
- package/agent-spine-plugin/src/cli-diagnostics.js +291 -0
- package/agent-spine-plugin/src/cli-host.js +21 -0
- package/agent-spine-plugin/src/cli-learning.js +305 -0
- package/agent-spine-plugin/src/cli-premortem.js +16 -0
- package/agent-spine-plugin/src/cli-sharing.js +230 -0
- package/agent-spine-plugin/src/cli.js +40 -1350
- package/agent-spine-plugin/src/codex-reader-launcher.js +182 -0
- package/agent-spine-plugin/src/hook.js +185 -567
- package/agent-spine-plugin/src/index.js +16 -2
- package/agent-spine-plugin/src/lib/acceptance.js +113 -11
- package/agent-spine-plugin/src/lib/action-lesson-recall.js +53 -0
- package/agent-spine-plugin/src/lib/attention-context.js +167 -0
- package/agent-spine-plugin/src/lib/attention-events.js +113 -0
- package/agent-spine-plugin/src/lib/attention-privacy.js +72 -0
- package/agent-spine-plugin/src/lib/attention-schema.js +164 -0
- package/agent-spine-plugin/src/lib/attention-storage.js +93 -0
- package/agent-spine-plugin/src/lib/attention.js +9 -600
- package/agent-spine-plugin/src/lib/audit-premortem.js +223 -0
- package/agent-spine-plugin/src/lib/audit.js +56 -6
- package/agent-spine-plugin/src/lib/autonomy-policy.js +110 -0
- package/agent-spine-plugin/src/lib/autonomy-store.js +202 -0
- package/agent-spine-plugin/src/lib/autonomy.js +8 -0
- package/agent-spine-plugin/src/lib/briefing.js +67 -4
- package/agent-spine-plugin/src/lib/catalog-document-read.js +51 -0
- package/agent-spine-plugin/src/lib/catalog.js +1 -1
- package/agent-spine-plugin/src/lib/codex-installation.js +231 -0
- package/agent-spine-plugin/src/lib/codex-skill-installation.js +211 -0
- package/agent-spine-plugin/src/lib/context.js +3 -1
- package/agent-spine-plugin/src/lib/delivery-agent-usage.js +224 -0
- package/agent-spine-plugin/src/lib/delivery-assignment.js +220 -0
- package/agent-spine-plugin/src/lib/delivery-command-actions.js +453 -0
- package/agent-spine-plugin/src/lib/delivery-knowledge.js +78 -0
- package/agent-spine-plugin/src/lib/delivery-premortem-binding.js +107 -0
- package/agent-spine-plugin/src/lib/delivery-premortem-closure.js +171 -0
- package/agent-spine-plugin/src/lib/delivery-premortem-codec.js +45 -0
- package/agent-spine-plugin/src/lib/delivery-premortem-correction.js +101 -0
- package/agent-spine-plugin/src/lib/delivery-premortem-file.js +21 -0
- package/agent-spine-plugin/src/lib/delivery-premortem-index.js +493 -0
- package/agent-spine-plugin/src/lib/delivery-premortem-inspection.js +37 -0
- package/agent-spine-plugin/src/lib/delivery-premortem-recovery.js +120 -0
- package/agent-spine-plugin/src/lib/delivery-premortem-rejection.js +65 -0
- package/agent-spine-plugin/src/lib/delivery-premortem-results.js +28 -0
- package/agent-spine-plugin/src/lib/delivery-premortem-session-guard.js +46 -0
- package/agent-spine-plugin/src/lib/delivery-premortem-write-ledger.js +285 -0
- package/agent-spine-plugin/src/lib/delivery-premortem.js +499 -0
- package/agent-spine-plugin/src/lib/delivery-shell-heredoc.js +115 -0
- package/agent-spine-plugin/src/lib/delivery-shell-substitutions.js +111 -0
- package/agent-spine-plugin/src/lib/delivery-shell-wrapper.js +64 -0
- package/agent-spine-plugin/src/lib/delivery-target.js +73 -0
- package/agent-spine-plugin/src/lib/delivery-verification.js +445 -0
- package/agent-spine-plugin/src/lib/documents.js +27 -5
- package/agent-spine-plugin/src/lib/filesystem-retry.js +2 -0
- package/agent-spine-plugin/src/lib/gateway-common.js +68 -0
- package/agent-spine-plugin/src/lib/gateway-control.js +302 -0
- package/agent-spine-plugin/src/lib/gateway-delivery.js +103 -0
- package/agent-spine-plugin/src/lib/gateway-execution.js +350 -0
- package/agent-spine-plugin/src/lib/gateway-host-lifecycle.js +185 -0
- package/agent-spine-plugin/src/lib/gateway-inspection.js +85 -0
- package/agent-spine-plugin/src/lib/gateway-knowledge.js +129 -0
- package/agent-spine-plugin/src/lib/gateway-policy-provenance.js +197 -0
- package/agent-spine-plugin/src/lib/gateway-premortem-disposition.js +82 -0
- package/agent-spine-plugin/src/lib/gateway-premortem.js +363 -0
- package/agent-spine-plugin/src/lib/gateway-runs.js +300 -0
- package/agent-spine-plugin/src/lib/gateway-runtime-identity.js +31 -0
- package/agent-spine-plugin/src/lib/gateway-runtime-records.js +21 -0
- package/agent-spine-plugin/src/lib/gateway-runtime.js +18 -1623
- package/agent-spine-plugin/src/lib/gateway-state-transaction.js +356 -0
- package/agent-spine-plugin/src/lib/gateway-state.js +342 -0
- package/agent-spine-plugin/src/lib/hook-artifact-guards.js +356 -0
- package/agent-spine-plugin/src/lib/hook-audit.js +16 -2
- package/agent-spine-plugin/src/lib/hook-briefing-use.js +68 -0
- package/agent-spine-plugin/src/lib/hook-context.js +424 -0
- package/agent-spine-plugin/src/lib/hook-final-message.js +26 -0
- package/agent-spine-plugin/src/lib/hook-input.js +27 -0
- package/agent-spine-plugin/src/lib/hook-output.js +140 -0
- package/agent-spine-plugin/src/lib/hook-premortem.js +257 -0
- package/agent-spine-plugin/src/lib/hook-process-advisory.js +37 -0
- package/agent-spine-plugin/src/lib/hook-protection.js +95 -0
- package/agent-spine-plugin/src/lib/hook-stop-verification.js +84 -0
- package/agent-spine-plugin/src/lib/hook-timeline.js +91 -0
- package/agent-spine-plugin/src/lib/identifier-analysis.js +446 -0
- package/agent-spine-plugin/src/lib/indexed-memory.js +23 -6
- package/agent-spine-plugin/src/lib/knowledge-evidence.js +431 -0
- package/agent-spine-plugin/src/lib/learning-applications.js +441 -0
- package/agent-spine-plugin/src/lib/learning-candidates.js +274 -0
- package/agent-spine-plugin/src/lib/learning-context.js +130 -0
- package/agent-spine-plugin/src/lib/learning-delivery-contracts.js +310 -0
- package/agent-spine-plugin/src/lib/learning-evaluation-contracts.js +350 -0
- package/agent-spine-plugin/src/lib/learning-evaluation-registration.js +372 -0
- package/agent-spine-plugin/src/lib/learning-evaluation-revocation.js +292 -0
- package/agent-spine-plugin/src/lib/learning-evidence-contracts.js +321 -0
- package/agent-spine-plugin/src/lib/learning-findings.js +458 -0
- package/agent-spine-plugin/src/lib/learning-measurement-contracts.js +285 -0
- package/agent-spine-plugin/src/lib/learning-measurements.js +265 -0
- package/agent-spine-plugin/src/lib/learning-outcome-contracts.js +165 -0
- package/agent-spine-plugin/src/lib/learning-outcomes.js +343 -0
- package/agent-spine-plugin/src/lib/learning-reconciliation.js +290 -0
- package/agent-spine-plugin/src/lib/learning-retry-contracts.js +188 -0
- package/agent-spine-plugin/src/lib/learning-schema.js +229 -0
- package/agent-spine-plugin/src/lib/learning-scope-targets.js +424 -0
- package/agent-spine-plugin/src/lib/learning-state-upgrade.js +477 -0
- package/agent-spine-plugin/src/lib/learning-status-configuration.js +476 -0
- package/agent-spine-plugin/src/lib/learning-storage.js +203 -0
- package/agent-spine-plugin/src/lib/learning-trial-recovery.js +218 -0
- package/agent-spine-plugin/src/lib/learning-validation-contracts.js +375 -0
- package/agent-spine-plugin/src/lib/learning-validation-renewal.js +298 -0
- package/agent-spine-plugin/src/lib/learning-validation-runtime.js +234 -0
- package/agent-spine-plugin/src/lib/learning.js +36 -6923
- package/agent-spine-plugin/src/lib/lesson-recall-session.js +172 -0
- package/agent-spine-plugin/src/lib/mcp-autonomy-tools.js +37 -0
- package/agent-spine-plugin/src/lib/mcp-delivery-completion.js +124 -0
- package/agent-spine-plugin/src/lib/mcp-delivery-tools.js +45 -0
- package/agent-spine-plugin/src/lib/mcp-premortem.js +68 -0
- package/agent-spine-plugin/src/lib/mcp-runtime.js +269 -0
- package/agent-spine-plugin/src/lib/mcp-source-context.js +51 -0
- package/agent-spine-plugin/src/lib/mcp-timeline-tools.js +164 -0
- package/agent-spine-plugin/src/lib/mcp-world-tools.js +52 -0
- package/agent-spine-plugin/src/lib/owned-file-lock.js +30 -10
- package/agent-spine-plugin/src/lib/project-portfolio.js +176 -0
- package/agent-spine-plugin/src/lib/selfstarter-core.js +382 -0
- package/agent-spine-plugin/src/lib/selfstarter-jobs.js +149 -0
- package/agent-spine-plugin/src/lib/selfstarter-lease.js +204 -0
- package/agent-spine-plugin/src/lib/selfstarter-policy.js +90 -0
- package/agent-spine-plugin/src/lib/selfstarter-workspace.js +111 -0
- package/agent-spine-plugin/src/lib/selfstarter.js +11 -911
- package/agent-spine-plugin/src/lib/session-timeline-auth.js +316 -0
- package/agent-spine-plugin/src/lib/session-timeline-codex.js +58 -0
- package/agent-spine-plugin/src/lib/session-timeline-contract.js +48 -0
- package/agent-spine-plugin/src/lib/session-timeline-enrollment-source.js +41 -0
- package/agent-spine-plugin/src/lib/session-timeline-enrollment-storage.js +132 -0
- package/agent-spine-plugin/src/lib/session-timeline-enrollment-transport.js +17 -0
- package/agent-spine-plugin/src/lib/session-timeline-enrollment.js +500 -0
- package/agent-spine-plugin/src/lib/session-timeline-event-extract.js +157 -0
- package/agent-spine-plugin/src/lib/session-timeline-host-origin.js +74 -0
- package/agent-spine-plugin/src/lib/session-timeline-host-receipt.js +117 -0
- package/agent-spine-plugin/src/lib/session-timeline-invocation.js +201 -0
- package/agent-spine-plugin/src/lib/session-timeline-king.js +79 -0
- package/agent-spine-plugin/src/lib/session-timeline-prior.js +59 -0
- package/agent-spine-plugin/src/lib/session-timeline-provider.js +34 -0
- package/agent-spine-plugin/src/lib/session-timeline-query.js +55 -0
- package/agent-spine-plugin/src/lib/session-timeline-results.js +31 -0
- package/agent-spine-plugin/src/lib/session-timeline-root.js +11 -0
- package/agent-spine-plugin/src/lib/session-timeline-search.js +82 -0
- package/agent-spine-plugin/src/lib/session-timeline-sid-acl.js +217 -0
- package/agent-spine-plugin/src/lib/session-timeline-source.js +83 -0
- package/agent-spine-plugin/src/lib/session-timeline-state.js +45 -0
- package/agent-spine-plugin/src/lib/session-timeline-transport.js +50 -0
- package/agent-spine-plugin/src/lib/session-timeline-windows-acl.js +148 -0
- package/agent-spine-plugin/src/lib/session-timeline.js +453 -0
- package/agent-spine-plugin/src/lib/source-roots.js +69 -136
- package/agent-spine-plugin/src/lib/source-tree-scan.js +178 -0
- package/agent-spine-plugin/src/lib/task-knowledge-context.js +78 -0
- package/agent-spine-plugin/src/lib/timeline-tool-guard.js +202 -0
- package/agent-spine-plugin/src/lib/world-knowledge.js +249 -0
- package/agent-spine-plugin/src/lib/world-model.js +278 -0
- package/agent-spine-plugin/src/mcp.js +20 -161
- package/agent-spine-plugin/src/version.js +1 -1
- package/agent-spine-plugin/src/worker.js +22 -5
- package/bin/agent-resume-snapshot.cjs +2 -2
- package/bin/agentspine-king-goal-inbox.mjs +111 -0
- package/bin/agentspine-king-goal-intake.mjs +106 -0
- package/bin/baseline-skill-performance-policy.cjs +1 -16
- package/bin/core-bootstrap.js +2 -0
- package/bin/curiosity-scout-policy.cjs +5 -1
- package/bin/input-draft-persistence.cjs +2 -2
- package/bin/king-tui-function-contract.json +33 -0
- package/bin/launcher-restart-policy.cjs +150 -0
- package/bin/launcher-runtime.js +56 -44
- package/bin/managed-context-startup-policy.cjs +27 -0
- package/bin/managed-plugin-selection.cjs +116 -0
- package/bin/mistake-relevance-policy.cjs +1 -1
- package/bin/observer-hooks.cjs +14 -0
- package/bin/oversized-context-offload-policy.cjs +86 -0
- package/bin/plugin-bootstrap.js +7 -40
- package/bin/proactive-compaction-policy.cjs +1 -1
- package/bin/provider-model-refresh-deadline.cjs +53 -0
- package/bin/provider-model-refresh-policy.cjs +107 -0
- package/bin/release-artifact-freeze-policy.cjs +30 -0
- package/bin/repeated-user-message-projection.cjs +3 -126
- package/bin/research-page-result.cjs +74 -0
- package/bin/runtime-exit-ledger.cjs +1 -0
- package/bin/session-compaction-policy.cjs +84 -0
- package/bin/skill-listing-performance-policy.cjs +2 -2
- package/bin/standard-tools-bootstrap.js +0 -37
- package/bin/subagent-skill-policy.cjs +3 -1
- package/bin/telegram-approval-relay.cjs +12 -7
- package/bin/telegram-private-conversation-policy.cjs +3 -2
- package/bin/telegram-queue-handoff-policy.cjs +24 -0
- package/bin/thinking-activity-status-policy.cjs +1 -1
- package/bin/thinking-only-guard.cjs +16 -12
- package/bin/tool-call-loop-policy.cjs +0 -2
- package/bin/tool-result-offload-policy.cjs +11 -2
- package/bin/tui-functional-contract.cjs +55 -0
- package/bin/update-notice.js +18 -14
- package/bin/user-message-offload-policy.cjs +1 -1
- package/bin/user-prompt-hook-origin-policy.cjs +34 -0
- package/bin/windows-node-crash-dump.cjs +110 -0
- package/blun.mjs +2517 -1048
- package/codebase-index/codebase_index.py +4 -3
- package/package.json +3 -17
- package/standard-skills/research-evidence/SKILL.md +39 -0
- package/standard-skills/research-evidence/references/evidence-format.md +82 -0
- package/standard-skills/research-evidence/scripts/evidence-collection.cjs +254 -0
- package/standard-skills/research-evidence/scripts/score-report.cjs +112 -0
- package/standard-skills/web-lesen/SKILL.md +37 -22
- package/standard-skills/web-lesen/scripts/crawl_public.py +376 -0
- package/telegram-plugin/DELIVERY.md +36 -0
- package/telegram-plugin/bin/telegram-approval-relay.cjs +13 -7
- package/telegram-plugin/bin/telegram-launcher-status-queue.cjs +122 -0
- package/telegram-plugin/bin/telegram-private-conversation-policy.cjs +3 -2
- package/telegram-plugin/bin/telegram-reply-parts.cjs +149 -0
- package/telegram-plugin/dist/bridge.mjs +7 -56
- package/telegram-plugin/dist/mcp-server.mjs +33 -4
- package/agent-spine-plugin/skill/SKILL.md +0 -76
- package/bin/mnemo-connect-heartbeat.cjs +0 -204
- package/bin/mnemo-tool-agent-policy.cjs +0 -22
- package/telegram-plugin/bin/telegram-mnemo-capture.cjs +0 -297
|
@@ -0,0 +1,376 @@
|
|
|
1
|
+
"""Bounded, credential-free public HTTPS crawling with Crawl4AI extraction/BFS."""
|
|
2
|
+
import argparse
|
|
3
|
+
import asyncio
|
|
4
|
+
import contextlib
|
|
5
|
+
import hashlib
|
|
6
|
+
import ipaddress
|
|
7
|
+
import json
|
|
8
|
+
import os
|
|
9
|
+
import re
|
|
10
|
+
import socket
|
|
11
|
+
import stat
|
|
12
|
+
import sys
|
|
13
|
+
import time
|
|
14
|
+
from datetime import datetime, timezone
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from urllib.parse import parse_qsl, unquote, urljoin, urlsplit, urlunsplit
|
|
17
|
+
from urllib.robotparser import RobotFileParser
|
|
18
|
+
|
|
19
|
+
import aiohttp
|
|
20
|
+
|
|
21
|
+
AGENT = 'BLUN-PublicResearch/1.0'
|
|
22
|
+
BODY_LIMIT = 2 * 1024 * 1024
|
|
23
|
+
SENSITIVE_QUERY = re.compile(
|
|
24
|
+
r'^(?:(?:access|refresh|id|oauth)[_-]?token|token|api[_-]?key|key|client[_-]?secret|'
|
|
25
|
+
r'password|passwd|secret|signature|sig|auth|authorization|bearer|jwt|session(?:[_-]?id)?|'
|
|
26
|
+
r'code|credentials?|x-(?:amz|goog)-.*)$', re.I)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class CrawlError(Exception):
|
|
30
|
+
def __init__(self, code, status=None):
|
|
31
|
+
super().__init__(code)
|
|
32
|
+
self.code = code
|
|
33
|
+
self.status = status
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def require_plain_directory(directory):
|
|
37
|
+
directory = Path(directory)
|
|
38
|
+
if not directory.is_absolute() or '..' in directory.parts:
|
|
39
|
+
raise CrawlError('absolute-output-path-required')
|
|
40
|
+
for entry in reversed((directory, *directory.parents)):
|
|
41
|
+
info = entry.lstat()
|
|
42
|
+
if (stat.S_ISLNK(info.st_mode)
|
|
43
|
+
or getattr(info, 'st_file_attributes', 0) & getattr(stat, 'FILE_ATTRIBUTE_REPARSE_POINT', 0)):
|
|
44
|
+
raise CrawlError('linked-output-path')
|
|
45
|
+
if not stat.S_ISDIR(info.st_mode):
|
|
46
|
+
raise CrawlError('output-directory-required')
|
|
47
|
+
return directory
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def new_output_directory(home):
|
|
51
|
+
home = Path(home)
|
|
52
|
+
if not home.is_absolute() or not home.exists():
|
|
53
|
+
raise CrawlError('app-home-required')
|
|
54
|
+
base = require_plain_directory(home)
|
|
55
|
+
for name in ('artifacts', 'web-crawl'):
|
|
56
|
+
base = base / name
|
|
57
|
+
try:
|
|
58
|
+
base.mkdir()
|
|
59
|
+
except FileExistsError:
|
|
60
|
+
pass
|
|
61
|
+
require_plain_directory(base)
|
|
62
|
+
import tempfile
|
|
63
|
+
return require_plain_directory(Path(tempfile.mkdtemp(prefix='crawl-', dir=base)))
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def public_ip(value):
|
|
67
|
+
try:
|
|
68
|
+
address = ipaddress.ip_address(value)
|
|
69
|
+
if not address.is_global or address.is_multicast or address.is_reserved:
|
|
70
|
+
return False
|
|
71
|
+
if address.version == 6:
|
|
72
|
+
return address in ipaddress.ip_network('2000::/3') and not address.sixtofour and not address.teredo
|
|
73
|
+
return address not in ipaddress.ip_network('192.0.0.0/24')
|
|
74
|
+
except ValueError:
|
|
75
|
+
return False
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def normalize_url(value):
|
|
79
|
+
if not isinstance(value, str) or len(value) > 4096 or re.search(r'[\x00-\x20\\]', value):
|
|
80
|
+
raise CrawlError('invalid-url')
|
|
81
|
+
parsed = urlsplit(value)
|
|
82
|
+
if parsed.scheme != 'https' or parsed.username is not None or parsed.password is not None:
|
|
83
|
+
raise CrawlError('public-https-required')
|
|
84
|
+
if parsed.port not in (None, 443) or not parsed.hostname:
|
|
85
|
+
raise CrawlError('public-https-required')
|
|
86
|
+
host = parsed.hostname.encode('idna').decode('ascii').lower().rstrip('.')
|
|
87
|
+
try:
|
|
88
|
+
ipaddress.ip_address(host)
|
|
89
|
+
if not public_ip(host): raise CrawlError('private-address')
|
|
90
|
+
except ValueError:
|
|
91
|
+
if ('.' not in host or not re.fullmatch(r'[a-z0-9.-]+', host)
|
|
92
|
+
or host.endswith(('.local', '.localhost', '.internal', '.test', '.invalid'))):
|
|
93
|
+
raise CrawlError('public-host-required')
|
|
94
|
+
path = parsed.path or '/'
|
|
95
|
+
decoded = path
|
|
96
|
+
for _ in range(3):
|
|
97
|
+
decoded = unquote(decoded)
|
|
98
|
+
if '\\' in decoded or any(part in ('.', '..') for part in decoded.split('/')):
|
|
99
|
+
raise CrawlError('ambiguous-path')
|
|
100
|
+
if re.search(r'/(?:login|logout|signin|signout|oauth|auth|account)(?:/|$)', decoded, re.I):
|
|
101
|
+
raise CrawlError('login-path-not-crawled')
|
|
102
|
+
for key, _ in parse_qsl(parsed.query, keep_blank_values=True):
|
|
103
|
+
if any(SENSITIVE_QUERY.fullmatch(part) for part in re.split(r'[\[\].]+', key)):
|
|
104
|
+
raise CrawlError('credential-query-not-crawled')
|
|
105
|
+
authority = '[' + host + ']' if ':' in host else host
|
|
106
|
+
return urlunsplit(('https', authority, path, parsed.query, ''))
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
class Scope:
|
|
110
|
+
def __init__(self, start):
|
|
111
|
+
self.start = normalize_url(start)
|
|
112
|
+
parsed = urlsplit(self.start)
|
|
113
|
+
self.origin = urlunsplit((parsed.scheme, parsed.netloc, '', '', ''))
|
|
114
|
+
self.prefix = parsed.path if parsed.path.endswith('/') else parsed.path.rsplit('/', 1)[0] + '/'
|
|
115
|
+
|
|
116
|
+
def check(self, value, base=None, robots=False):
|
|
117
|
+
checked = normalize_url(urljoin(base or self.start, value))
|
|
118
|
+
parsed = urlsplit(checked)
|
|
119
|
+
if urlunsplit((parsed.scheme, parsed.netloc, '', '', '')) != self.origin:
|
|
120
|
+
raise CrawlError('outside-origin')
|
|
121
|
+
if not robots and not unquote(parsed.path).startswith(unquote(self.prefix)):
|
|
122
|
+
raise CrawlError('outside-path-scope')
|
|
123
|
+
return checked
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
class PublicResolver(aiohttp.abc.AbstractResolver):
|
|
127
|
+
async def resolve(self, host, port=0, family=socket.AF_UNSPEC):
|
|
128
|
+
addresses = await asyncio.get_running_loop().getaddrinfo(
|
|
129
|
+
host, port, type=socket.SOCK_STREAM, family=family)
|
|
130
|
+
if not addresses or any(not public_ip(row[4][0]) for row in addresses):
|
|
131
|
+
raise CrawlError('private-address')
|
|
132
|
+
# The connector receives these checked IPs directly, with the original
|
|
133
|
+
# hostname retained for certificate validation. There is no second DNS lookup.
|
|
134
|
+
return [dict(hostname=host, host=row[4][0], port=port, family=row[0],
|
|
135
|
+
proto=row[2], flags=socket.AI_NUMERICHOST) for row in addresses]
|
|
136
|
+
|
|
137
|
+
async def close(self):
|
|
138
|
+
pass
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
async def read_body(response, maximum):
|
|
142
|
+
if int(response.headers.get('Content-Length', '0')) > maximum:
|
|
143
|
+
raise CrawlError('body-limit')
|
|
144
|
+
chunks, size = [], 0
|
|
145
|
+
async for chunk in response.content.iter_chunked(16384):
|
|
146
|
+
size += len(chunk)
|
|
147
|
+
if size > maximum:
|
|
148
|
+
raise CrawlError('body-limit')
|
|
149
|
+
chunks.append(chunk)
|
|
150
|
+
return b''.join(chunks)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
class PublicStrategy:
|
|
154
|
+
def __init__(self, scope, max_pages):
|
|
155
|
+
self.scope, self.max_pages = scope, max_pages
|
|
156
|
+
self.attempts = 0
|
|
157
|
+
self.session = None
|
|
158
|
+
self.robots = None
|
|
159
|
+
self.lock = asyncio.Lock()
|
|
160
|
+
self.last_fetch = 0
|
|
161
|
+
self.inflight = set()
|
|
162
|
+
self.closed = False
|
|
163
|
+
self.errors = {}
|
|
164
|
+
|
|
165
|
+
async def __aenter__(self):
|
|
166
|
+
self.closed = False
|
|
167
|
+
self.session = aiohttp.ClientSession(
|
|
168
|
+
connector=aiohttp.TCPConnector(resolver=PublicResolver(), use_dns_cache=False,
|
|
169
|
+
force_close=True, limit=1),
|
|
170
|
+
cookie_jar=aiohttp.DummyCookieJar(), trust_env=False,
|
|
171
|
+
timeout=aiohttp.ClientTimeout(total=15, connect=5),
|
|
172
|
+
headers={'User-Agent': AGENT, 'Accept': 'text/html,application/xhtml+xml',
|
|
173
|
+
'Accept-Encoding': 'identity'}, auto_decompress=False)
|
|
174
|
+
return self
|
|
175
|
+
|
|
176
|
+
async def __aexit__(self, *args):
|
|
177
|
+
self.closed = True
|
|
178
|
+
owned = [task for task in self.inflight if task is not asyncio.current_task()]
|
|
179
|
+
for task in owned:
|
|
180
|
+
task.cancel()
|
|
181
|
+
if owned:
|
|
182
|
+
await asyncio.gather(*owned, return_exceptions=True)
|
|
183
|
+
if self.session:
|
|
184
|
+
await self.session.close()
|
|
185
|
+
|
|
186
|
+
async def fetch(self, url, robots=False):
|
|
187
|
+
current = self.scope.check(url, robots=robots)
|
|
188
|
+
for _ in range(4):
|
|
189
|
+
delay = (self.robots.crawl_delay(AGENT) or 0) if not robots and self.robots else 0
|
|
190
|
+
if delay > 10:
|
|
191
|
+
raise CrawlError('robots-delay-exceeds-budget')
|
|
192
|
+
await asyncio.sleep(max(0, max(0.4, delay) - (time.monotonic() - self.last_fetch)))
|
|
193
|
+
self.last_fetch = time.monotonic()
|
|
194
|
+
async with self.session.get(current, allow_redirects=False) as response:
|
|
195
|
+
if response.status in (301, 302, 303, 307, 308):
|
|
196
|
+
location = response.headers.get('Location')
|
|
197
|
+
if not location: raise CrawlError('redirect-without-location')
|
|
198
|
+
current = self.scope.check(location, base=current, robots=robots)
|
|
199
|
+
if not robots and self.robots and not self.robots.can_fetch(AGENT, current):
|
|
200
|
+
raise CrawlError('robots-disallowed')
|
|
201
|
+
continue
|
|
202
|
+
if robots and response.status in (404, 410):
|
|
203
|
+
return current, b'', 'text/plain'
|
|
204
|
+
if response.status != 200:
|
|
205
|
+
raise CrawlError('http-status-' + str(response.status), status=response.status)
|
|
206
|
+
if response.headers.get('Content-Encoding', 'identity').lower() != 'identity':
|
|
207
|
+
raise CrawlError('unsupported-content-encoding')
|
|
208
|
+
mime = response.headers.get('Content-Type', '').split(';')[0].lower().strip()
|
|
209
|
+
if mime not in (('text/plain',) if robots else ('text/html', 'application/xhtml+xml')):
|
|
210
|
+
raise CrawlError('unsupported-content-type')
|
|
211
|
+
return current, await read_body(response, 256 * 1024 if robots else BODY_LIMIT), mime
|
|
212
|
+
raise CrawlError('redirect-limit')
|
|
213
|
+
|
|
214
|
+
async def crawl(self, url, **kwargs):
|
|
215
|
+
if self.closed:
|
|
216
|
+
raise CrawlError('crawler-closed')
|
|
217
|
+
task = asyncio.current_task()
|
|
218
|
+
self.inflight.add(task)
|
|
219
|
+
try:
|
|
220
|
+
return await self._crawl(url)
|
|
221
|
+
except Exception as error:
|
|
222
|
+
code, status = 'crawl-error', None
|
|
223
|
+
if isinstance(error, CrawlError):
|
|
224
|
+
if isinstance(error.code, str) and re.fullmatch(r'[a-z0-9-]{1,64}', error.code):
|
|
225
|
+
code = error.code
|
|
226
|
+
status = error.status
|
|
227
|
+
elif isinstance(error, TimeoutError):
|
|
228
|
+
code = 'request-timeout'
|
|
229
|
+
elif isinstance(error, UnicodeError):
|
|
230
|
+
code = 'invalid-text-encoding'
|
|
231
|
+
elif isinstance(error, aiohttp.ClientError):
|
|
232
|
+
code = 'network-error'
|
|
233
|
+
self.errors[url] = dict(reason=code, status=status)
|
|
234
|
+
raise
|
|
235
|
+
finally:
|
|
236
|
+
self.inflight.discard(task)
|
|
237
|
+
|
|
238
|
+
async def _crawl(self, url):
|
|
239
|
+
from crawl4ai.models import AsyncCrawlResponse
|
|
240
|
+
checked = self.scope.check(url)
|
|
241
|
+
async with self.lock:
|
|
242
|
+
if self.attempts >= self.max_pages: raise CrawlError('page-limit')
|
|
243
|
+
self.attempts += 1
|
|
244
|
+
if self.robots is None:
|
|
245
|
+
_, raw, _ = await self.fetch(self.scope.origin + '/robots.txt', robots=True)
|
|
246
|
+
self.robots = RobotFileParser()
|
|
247
|
+
self.robots.parse(raw.decode('utf-8', errors='strict').splitlines())
|
|
248
|
+
if not self.robots.can_fetch(AGENT, checked): raise CrawlError('robots-disallowed')
|
|
249
|
+
final, raw, mime = await self.fetch(checked)
|
|
250
|
+
html = raw.decode('utf-8', errors='strict')
|
|
251
|
+
from lxml import html as lhtml
|
|
252
|
+
document = lhtml.fromstring(html)
|
|
253
|
+
if document.xpath('//form[@id="challenge-form" or @id="anomaly-form" or @id="challenge-stage"]'):
|
|
254
|
+
raise CrawlError('challenge-page-not-crawled', status=200)
|
|
255
|
+
if document.xpath('//input[translate(@type, "PASSWORD", "password")="password"]'):
|
|
256
|
+
raise CrawlError('login-page-not-crawled', status=200)
|
|
257
|
+
return AsyncCrawlResponse(html=html,
|
|
258
|
+
response_headers={'content-type': mime}, status_code=200,
|
|
259
|
+
redirected_url=final)
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
async def run_crawl(scope, directory, max_pages=6, depth=1, seconds=60, strategy=None):
|
|
263
|
+
if not 1 <= max_pages <= 12 or not 0 <= depth <= 2 or not 1 <= seconds <= 120:
|
|
264
|
+
raise CrawlError('invalid-budget')
|
|
265
|
+
directory = require_plain_directory(directory)
|
|
266
|
+
os.environ['CRAWL4_AI_BASE_DIRECTORY'] = str(directory)
|
|
267
|
+
from crawl4ai import AsyncWebCrawler, BrowserConfig, CacheMode, CrawlerRunConfig
|
|
268
|
+
from crawl4ai.deep_crawling import BFSDeepCrawlStrategy
|
|
269
|
+
from crawl4ai.deep_crawling.filters import FilterChain, URLFilter
|
|
270
|
+
from crawl4ai.async_logger import AsyncLogger
|
|
271
|
+
from crawl4ai.content_scraping_strategy import LXMLWebScrapingStrategy
|
|
272
|
+
from lxml import html as lhtml
|
|
273
|
+
|
|
274
|
+
class ArticleScraper(LXMLWebScrapingStrategy):
|
|
275
|
+
def scrap(self, url, html, **kwargs):
|
|
276
|
+
document = lhtml.fromstring(html)
|
|
277
|
+
for selector in ('main', 'article', '[role="main"]'):
|
|
278
|
+
if document.cssselect(selector):
|
|
279
|
+
kwargs['css_selector'] = selector
|
|
280
|
+
break
|
|
281
|
+
return super().scrap(url, html, **kwargs)
|
|
282
|
+
|
|
283
|
+
admitted = {scope.start}
|
|
284
|
+
class InScope(URLFilter):
|
|
285
|
+
def apply(self, url):
|
|
286
|
+
try:
|
|
287
|
+
checked = scope.check(url)
|
|
288
|
+
if checked not in admitted:
|
|
289
|
+
if len(admitted) >= max_pages:
|
|
290
|
+
return False
|
|
291
|
+
admitted.add(checked)
|
|
292
|
+
return True
|
|
293
|
+
except (CrawlError, ValueError):
|
|
294
|
+
return False
|
|
295
|
+
|
|
296
|
+
strategy = strategy or PublicStrategy(scope, max_pages)
|
|
297
|
+
# This Crawl4AI version stops before yielding its last allowed streaming page.
|
|
298
|
+
# One planner sentinel plus transport/consumer caps preserves that page safely.
|
|
299
|
+
plan = BFSDeepCrawlStrategy(max_depth=depth, include_external=False, max_pages=max_pages + 1,
|
|
300
|
+
filter_chain=FilterChain([InScope()]))
|
|
301
|
+
config = CrawlerRunConfig(deep_crawl_strategy=plan, cache_mode=CacheMode.DISABLED,
|
|
302
|
+
stream=True, max_retries=0, check_robots_txt=False, fallback_fetch_function=None,
|
|
303
|
+
page_timeout=15000, word_count_threshold=1, remove_forms=True, exclude_all_images=True,
|
|
304
|
+
excluded_tags=['nav', 'footer', 'script', 'style'], verbose=False,
|
|
305
|
+
scraping_strategy=ArticleScraper())
|
|
306
|
+
report = dict(schema='blun.public-crawl/v1', start=scope.start, scope=scope.origin + scope.prefix,
|
|
307
|
+
limits=dict(pages=max_pages, depth=depth, seconds=seconds, bytesPerPage=BODY_LIMIT),
|
|
308
|
+
siteComplete=False, pages=[], failures=[], outcome='bounded',
|
|
309
|
+
javascript=False, authenticated=False, contentAuthority='untrusted-source')
|
|
310
|
+
results = []
|
|
311
|
+
try:
|
|
312
|
+
async with asyncio.timeout(seconds):
|
|
313
|
+
async with AsyncWebCrawler(crawler_strategy=strategy,
|
|
314
|
+
config=BrowserConfig(verbose=False), base_directory=str(directory),
|
|
315
|
+
logger=AsyncLogger(verbose=False, log_file=None)) as crawler:
|
|
316
|
+
stream = await crawler.arun(scope.start, config=config)
|
|
317
|
+
async with contextlib.aclosing(stream):
|
|
318
|
+
async for result in stream:
|
|
319
|
+
results.append(result)
|
|
320
|
+
if len(results) >= max_pages:
|
|
321
|
+
break
|
|
322
|
+
except TimeoutError:
|
|
323
|
+
report['outcome'] = 'time-limit'
|
|
324
|
+
for result in results:
|
|
325
|
+
if not result.success or not result.markdown:
|
|
326
|
+
error = getattr(strategy, 'errors', {}).get(result.url, {})
|
|
327
|
+
report['failures'].append(dict(url=result.url,
|
|
328
|
+
status=error.get('status') or result.status_code,
|
|
329
|
+
reason=error.get('reason', 'page-unavailable')))
|
|
330
|
+
continue
|
|
331
|
+
content = result.markdown.raw_markdown if hasattr(result.markdown, 'raw_markdown') else str(result.markdown)
|
|
332
|
+
if not content.strip():
|
|
333
|
+
report['failures'].append(dict(url=result.url, reason='empty-markdown'))
|
|
334
|
+
continue
|
|
335
|
+
data = content.encode('utf-8')
|
|
336
|
+
if len(data) > BODY_LIMIT:
|
|
337
|
+
report['failures'].append(dict(url=result.url, reason='markdown-limit'))
|
|
338
|
+
continue
|
|
339
|
+
digest = hashlib.sha256(data).hexdigest()
|
|
340
|
+
filename = 'page-' + str(len(report['pages']) + 1) + '-' + digest[:12] + '.md'
|
|
341
|
+
require_plain_directory(directory)
|
|
342
|
+
with (directory / filename).open('x', encoding='utf-8', newline='\n') as output:
|
|
343
|
+
output.write(content)
|
|
344
|
+
report['pages'].append(dict(url=result.redirected_url or result.url, file=filename,
|
|
345
|
+
sha256=digest, bytes=len(data), retrievedAt=datetime.now(timezone.utc).isoformat(),
|
|
346
|
+
depth=(result.metadata or {}).get('depth', 0)))
|
|
347
|
+
if report['failures'] and report['outcome'] == 'bounded': report['outcome'] = 'partial'
|
|
348
|
+
if not report['pages'] and report['outcome'] == 'bounded': report['outcome'] = 'unavailable'
|
|
349
|
+
require_plain_directory(directory)
|
|
350
|
+
with (directory / 'manifest.json').open('x', encoding='utf-8', newline='\n') as output:
|
|
351
|
+
json.dump(report, output, ensure_ascii=True, indent=2)
|
|
352
|
+
return report
|
|
353
|
+
|
|
354
|
+
|
|
355
|
+
def main():
|
|
356
|
+
parser = argparse.ArgumentParser()
|
|
357
|
+
parser.add_argument('url')
|
|
358
|
+
parser.add_argument('--pages', type=int, default=6)
|
|
359
|
+
parser.add_argument('--depth', type=int, default=1)
|
|
360
|
+
parser.add_argument('--seconds', type=int, default=60)
|
|
361
|
+
args = parser.parse_args()
|
|
362
|
+
scope = Scope(args.url)
|
|
363
|
+
home = Path(os.environ.get('BLUN_HOME', str(Path.home() / '.blun')))
|
|
364
|
+
directory = new_output_directory(home)
|
|
365
|
+
with contextlib.redirect_stdout(sys.stderr):
|
|
366
|
+
report = asyncio.run(run_crawl(scope, directory, args.pages, args.depth, args.seconds))
|
|
367
|
+
print(json.dumps(dict(directory=str(directory), **report), ensure_ascii=True))
|
|
368
|
+
return 0 if report['pages'] and not report['failures'] and report['outcome'] == 'bounded' else 1
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
if __name__ == '__main__':
|
|
372
|
+
try:
|
|
373
|
+
raise SystemExit(main())
|
|
374
|
+
except (CrawlError, ValueError) as error:
|
|
375
|
+
print(json.dumps({'error': str(error)}))
|
|
376
|
+
raise SystemExit(1)
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
# File Reply Receipts
|
|
2
|
+
|
|
3
|
+
File-bearing `reply` calls retain acknowledged message IDs in the channel's
|
|
4
|
+
`reply-deliveries` directory. Text-only calls keep their existing behavior.
|
|
5
|
+
|
|
6
|
+
- Supply a unique `delivery_id` for one intended delivery and keep it unchanged
|
|
7
|
+
on retries. Completed retries return the original IDs without sending again.
|
|
8
|
+
- IDs bind the bot, recipient, reply target, rendered chunks, formatting, file
|
|
9
|
+
order, names, paths and actual bytes. Reusing an ID with another payload fails
|
|
10
|
+
before sending. A new ID is for a deliberate new delivery, not error recovery.
|
|
11
|
+
- Legacy calls without an ID resume an identical incomplete batch. After a
|
|
12
|
+
successful unkeyed batch, another unkeyed call is a new delivery. This is not
|
|
13
|
+
global file deduplication and cannot cover loss of the final MCP response.
|
|
14
|
+
Error and success results include the generated ID for subsequent retries.
|
|
15
|
+
- Each request is marked in-flight before contacting Telegram. An acknowledged
|
|
16
|
+
message ID is committed immediately. An explicit grammY Bot API 4xx rejection
|
|
17
|
+
makes only that part retryable. Network errors, process death during a send,
|
|
18
|
+
malformed success responses and failed acknowledgement commits remain
|
|
19
|
+
uncertain. Automatic retry stops; check Telegram before deciding to send anew.
|
|
20
|
+
- A best-effort aggregate outbox log is not the part receipt. Its failure must
|
|
21
|
+
not turn an already confirmed delivery into a retryable send error.
|
|
22
|
+
|
|
23
|
+
The store uses Node's built-in SQLite API, already required by the application,
|
|
24
|
+
with FULL synchronous commits. A separate SQLite writer transaction serializes
|
|
25
|
+
file deliveries for this channel and is released by the OS on process death.
|
|
26
|
+
A concurrent attempt fails busy without sending. No bot credentials, message
|
|
27
|
+
text or file contents are saved in the receipt database, only IDs, payload
|
|
28
|
+
digests and progress. Existing channel-state attachment restrictions still apply.
|
|
29
|
+
|
|
30
|
+
Limits are 50MB per regular file, 1024 total parts per call and 10000 retained
|
|
31
|
+
delivery receipts. The store never silently evicts proofs; capacity, corruption
|
|
32
|
+
or unsupported schema failures stop new sends. Preserve the database before
|
|
33
|
+
maintenance. Deleting receipts is not a safe way to retry uncertain deliveries.
|
|
34
|
+
|
|
35
|
+
This protects this MCP file-reply path. It does not make Telegram's network API
|
|
36
|
+
exactly-once, coordinate independent fallback senders, or prove live delivery.
|
|
@@ -168,13 +168,19 @@ function beginTelegramApprovalRelay(payload, options = {}) {
|
|
|
168
168
|
if (settled) return false;
|
|
169
169
|
settled = true;
|
|
170
170
|
if (interval !== undefined) clearInterval(interval);
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
171
|
+
const persistedResponse = response?.decision === 'edited' ? { decision: 'edited' } : response;
|
|
172
|
+
try {
|
|
173
|
+
atomicWriteJson(resolvedFile, {
|
|
174
|
+
v: RELAY_VERSION,
|
|
175
|
+
requestId: request.requestId,
|
|
176
|
+
resolvedAt: Date.now(),
|
|
177
|
+
response: persistedResponse,
|
|
178
|
+
});
|
|
179
|
+
} catch (error) {
|
|
180
|
+
try { options.onError?.(error); } catch {}
|
|
181
|
+
} finally {
|
|
182
|
+
resolvePromise(response);
|
|
183
|
+
}
|
|
178
184
|
return true;
|
|
179
185
|
};
|
|
180
186
|
const poll = () => {
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
const fs = require('node:fs');
|
|
4
|
+
const path = require('node:path');
|
|
5
|
+
|
|
6
|
+
const QUEUE_FILE = 'launcher-status-queue.jsonl';
|
|
7
|
+
const CHECKPOINT_FILE = 'launcher-status.checkpoint.json';
|
|
8
|
+
const OUTBOX_FILE = 'outbox.jsonl';
|
|
9
|
+
|
|
10
|
+
function readCheckpoint(filePath, fsImpl = fs) {
|
|
11
|
+
try {
|
|
12
|
+
const parsed = JSON.parse(fsImpl.readFileSync(filePath, 'utf8'));
|
|
13
|
+
return Number.isSafeInteger(parsed.offset) && parsed.offset >= 0 ? parsed.offset : 0;
|
|
14
|
+
} catch { return 0; }
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
function writeCheckpoint(filePath, offset, fsImpl = fs) {
|
|
18
|
+
const temporary = `${filePath}.${process.pid}.tmp`;
|
|
19
|
+
fsImpl.writeFileSync(temporary, `${JSON.stringify({
|
|
20
|
+
schema: 'blun.launcher-status-checkpoint/v1', offset,
|
|
21
|
+
})}\n`, { encoding: 'utf8', mode: 0o600 });
|
|
22
|
+
fsImpl.renameSync(temporary, filePath);
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
function resolveStatusChatId(access, configuredChatId) {
|
|
26
|
+
const groups = Object.keys(access?.groups || {});
|
|
27
|
+
if (configuredChatId !== undefined && configuredChatId !== null && String(configuredChatId).trim()) {
|
|
28
|
+
const requested = String(configuredChatId).trim();
|
|
29
|
+
return groups.includes(requested) ? requested : null;
|
|
30
|
+
}
|
|
31
|
+
return groups[0] || null;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
function hasReceipt(outboxPath, eventId, fsImpl = fs) {
|
|
35
|
+
try {
|
|
36
|
+
const content = fsImpl.readFileSync(outboxPath, 'utf8');
|
|
37
|
+
return content.split(/\r?\n/u).some((line) => {
|
|
38
|
+
if (!line.trim()) return false;
|
|
39
|
+
try {
|
|
40
|
+
const entry = JSON.parse(line);
|
|
41
|
+
return entry.kind === 'launcher-status' && entry.event_id === eventId;
|
|
42
|
+
} catch { return false; }
|
|
43
|
+
});
|
|
44
|
+
} catch { return false; }
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
async function drainLauncherStatusQueue(options) {
|
|
48
|
+
const fsImpl = options.fsImpl || fs;
|
|
49
|
+
const stateDir = path.resolve(options.stateDir);
|
|
50
|
+
const queuePath = path.join(stateDir, QUEUE_FILE);
|
|
51
|
+
const checkpointPath = path.join(stateDir, CHECKPOINT_FILE);
|
|
52
|
+
const outboxPath = path.join(stateDir, OUTBOX_FILE);
|
|
53
|
+
fsImpl.mkdirSync(stateDir, { recursive: true, mode: 0o700 });
|
|
54
|
+
let buffer;
|
|
55
|
+
try { buffer = fsImpl.readFileSync(queuePath); } catch (error) {
|
|
56
|
+
if (error?.code === 'ENOENT') return { delivered: 0, pending: 0 };
|
|
57
|
+
throw error;
|
|
58
|
+
}
|
|
59
|
+
let offset = Math.min(readCheckpoint(checkpointPath, fsImpl), buffer.length);
|
|
60
|
+
let delivered = 0;
|
|
61
|
+
while (offset < buffer.length) {
|
|
62
|
+
const newline = buffer.indexOf(0x0a, offset);
|
|
63
|
+
if (newline < 0) break;
|
|
64
|
+
const nextOffset = newline + 1;
|
|
65
|
+
const line = buffer.subarray(offset, newline).toString('utf8').trim();
|
|
66
|
+
if (!line) {
|
|
67
|
+
offset = nextOffset;
|
|
68
|
+
writeCheckpoint(checkpointPath, offset, fsImpl);
|
|
69
|
+
continue;
|
|
70
|
+
}
|
|
71
|
+
let event;
|
|
72
|
+
try { event = JSON.parse(line); } catch {
|
|
73
|
+
offset = nextOffset;
|
|
74
|
+
writeCheckpoint(checkpointPath, offset, fsImpl);
|
|
75
|
+
continue;
|
|
76
|
+
}
|
|
77
|
+
if (event?.schema !== 'blun.launcher-status/v1' || typeof event.id !== 'string'
|
|
78
|
+
|| typeof event.text !== 'string') {
|
|
79
|
+
offset = nextOffset;
|
|
80
|
+
writeCheckpoint(checkpointPath, offset, fsImpl);
|
|
81
|
+
continue;
|
|
82
|
+
}
|
|
83
|
+
if (hasReceipt(outboxPath, event.id, fsImpl)) {
|
|
84
|
+
offset = nextOffset;
|
|
85
|
+
writeCheckpoint(checkpointPath, offset, fsImpl);
|
|
86
|
+
continue;
|
|
87
|
+
}
|
|
88
|
+
const chatId = resolveStatusChatId(options.readAccess(), options.configuredChatId);
|
|
89
|
+
if (chatId === null) break;
|
|
90
|
+
const sent = await options.sendMessage(chatId, event.text);
|
|
91
|
+
fsImpl.appendFileSync(outboxPath, `${JSON.stringify({
|
|
92
|
+
ts: new Date().toISOString(), direction: 'out', kind: 'launcher-status',
|
|
93
|
+
chat_id: chatId, message_id: sent?.message_id, event_id: event.id, text: event.text,
|
|
94
|
+
})}\n`, { encoding: 'utf8', mode: 0o600 });
|
|
95
|
+
delivered += 1;
|
|
96
|
+
offset = nextOffset;
|
|
97
|
+
writeCheckpoint(checkpointPath, offset, fsImpl);
|
|
98
|
+
}
|
|
99
|
+
const pending = buffer.subarray(offset).toString('utf8').split(/\r?\n/u).filter((line) => line.trim()).length;
|
|
100
|
+
return { delivered, pending };
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
function startLauncherStatusWatcher(options) {
|
|
104
|
+
let busy = false;
|
|
105
|
+
let disposed = false;
|
|
106
|
+
const run = async () => {
|
|
107
|
+
if (busy || disposed) return;
|
|
108
|
+
busy = true;
|
|
109
|
+
try { await drainLauncherStatusQueue(options); } catch (error) { options.onError?.(error); }
|
|
110
|
+
finally { busy = false; }
|
|
111
|
+
};
|
|
112
|
+
const timer = setInterval(run, options.intervalMs || 1000);
|
|
113
|
+
timer.unref?.();
|
|
114
|
+
void run();
|
|
115
|
+
return { dispose() { disposed = true; clearInterval(timer); } };
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
module.exports = {
|
|
119
|
+
drainLauncherStatusQueue,
|
|
120
|
+
resolveStatusChatId,
|
|
121
|
+
startLauncherStatusWatcher,
|
|
122
|
+
};
|
|
@@ -6,7 +6,8 @@ const PRIVATE_CHAT_ID = /^[1-9]\d*$/u;
|
|
|
6
6
|
const WORK_PERMISSION_QUESTION = /^(?:kann|darf|soll) ich\b.{0,100}\b(?:arbeit(?:en)?|weiterarbeiten|weitermachen|fortfahren|weiterbauen|umbau|bau)\b/u;
|
|
7
7
|
const TRAILING_WORK_PERMISSION_QUESTION = /(?:^|\r?\n\s*\r?\n)(?:kann|darf|soll)\s+ich\b[^\r\n]{0,160}\b(?:arbeit(?:en)?|weiterarbeiten|weitermachen|fortfahren|weiterbauen|umbau|bau)\b[^\r\n]*$/iu;
|
|
8
8
|
const INTERNAL_CONTROL_MARKER = /\b(?:cron|loop|checkpoint|handoff|resume|session|sha|tool|werkzeug|hook|queue|lane|inbound|outbound|arbeitsstand|arbeitsauftrag|bau freigabe|freigabe dieser lane|turn|zug|status|private nachricht|telegram nachricht|reply|verlauf|basis|qa gate|wartezustand|f\d+|m\d+|w\d+)\b/gu;
|
|
9
|
-
const NO_USER_VALUE = /\b(?:keine neue aktion|keine neue information|keine neue nachricht|kein neuer auftrag|kein neuer zweck|kein neuer inbound bedarf|kein offener handlungsbedarf|nichts neues zu melden|nichts zu melden|
|
|
9
|
+
const NO_USER_VALUE = /\b(?:keine neue aktion|keine neue information|keine neue nachricht|kein neuer auftrag|kein neuer zweck|kein neuer inbound bedarf|kein offener handlungsbedarf|nichts neues zu melden|nichts zu melden|bleibt pausiert|warte auf (?:dein|sein|ihr|papas) (?:zeichen|go|freigabe|antwort)|ich warte|ich sende nichts|wiederholen wäre spam|keine weitere nachricht|bereits geantwortet|bereits gesendet|kein werkzeugaufruf nötig|stand unverändert|zug läuft seit|kein bau ohne|qa gate steht noch aus|wartezustand)\b/u;
|
|
10
|
+
const EMPTY_COMPLETION_STATUS = /^(?:(?:status|arbeitsstand|checkpoint)\s+)?(?:fertig\s+)?nichts weiteres(?:\s+(?:offen|zu melden|zu tun))?$/u;
|
|
10
11
|
|
|
11
12
|
const MEDIA_PROGRESS_MARKER = /\b(?:bildgenerierung|bildjob|generateimage|getmedia|media job|medienjob)\b/u;
|
|
12
13
|
const MEDIA_PENDING_STATE = /\b(?:auftrag wurde angenommen|job wurde angenommen|job angenommen|aufruf wurde abgesetzt|angestossen|angestoßen|processing|verarbeitung)\b/u;
|
|
@@ -169,7 +170,7 @@ function isPrivateInternalStatusReply(chatId, text) {
|
|
|
169
170
|
if (MEDIA_PROGRESS_MARKER.test(value) && MEDIA_PENDING_STATE.test(value) && MEDIA_NO_RESULT.test(value)) return true;
|
|
170
171
|
|
|
171
172
|
const internalMarkers = value.match(INTERNAL_CONTROL_MARKER) ?? [];
|
|
172
|
-
return internalMarkers.length > 0 && NO_USER_VALUE.test(value);
|
|
173
|
+
return internalMarkers.length > 0 && (NO_USER_VALUE.test(value) || EMPTY_COMPLETION_STATUS.test(value));
|
|
173
174
|
}
|
|
174
175
|
|
|
175
176
|
module.exports = {
|