@yoonion/mimi-seed-mcp 0.21.1 → 0.21.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +2 -1
  2. package/assets/agent-guide.md +1 -1
  3. package/dist/googleads/config.d.ts +5 -2
  4. package/dist/googleads/config.js +22 -9
  5. package/dist/googleads/errors.js +25 -17
  6. package/dist/googleads/setup-cli.js +1 -1
  7. package/dist/googleads/tools.d.ts +8 -1
  8. package/dist/googleads/tools.js +19 -1
  9. package/dist/lib/tool-registrar.js +2 -1
  10. package/dist/naver/robots.d.ts +44 -0
  11. package/dist/naver/robots.js +126 -0
  12. package/dist/naver/tools.d.ts +127 -0
  13. package/dist/naver/tools.js +612 -0
  14. package/dist/playstore/products.d.ts +1 -1
  15. package/dist/playstore/products.js +1 -1
  16. package/dist/registers/appstore/apps.d.ts +4 -0
  17. package/dist/registers/appstore/apps.js +43 -0
  18. package/dist/registers/appstore/customer-reviews.d.ts +4 -0
  19. package/dist/registers/appstore/customer-reviews.js +31 -0
  20. package/dist/registers/appstore/declarations.d.ts +4 -0
  21. package/dist/registers/appstore/declarations.js +163 -0
  22. package/dist/registers/appstore/media.d.ts +6 -0
  23. package/dist/registers/appstore/media.js +116 -0
  24. package/dist/registers/appstore/metadata.d.ts +11 -0
  25. package/dist/registers/appstore/metadata.js +120 -0
  26. package/dist/registers/appstore/products.d.ts +6 -0
  27. package/dist/registers/appstore/products.js +297 -0
  28. package/dist/registers/appstore/reports.d.ts +6 -0
  29. package/dist/registers/appstore/reports.js +77 -0
  30. package/dist/registers/appstore/review-submission.d.ts +6 -0
  31. package/dist/registers/appstore/review-submission.js +113 -0
  32. package/dist/registers/appstore/testflight.d.ts +6 -0
  33. package/dist/registers/appstore/testflight.js +193 -0
  34. package/dist/registers/appstore/versions.d.ts +8 -0
  35. package/dist/registers/appstore/versions.js +205 -0
  36. package/dist/registers/appstore.d.ts +8 -0
  37. package/dist/registers/appstore.js +37 -1274
  38. package/dist/registers/naver.d.ts +2 -0
  39. package/dist/registers/naver.js +21 -0
  40. package/dist/server.js +2 -0
  41. package/package.json +1 -1
  42. package/tool-manifest.json +19 -1
package/README.md CHANGED
@@ -128,7 +128,7 @@ export ANTHROPIC_API_KEY=sk-ant-...
128
128
  ---
129
129
 
130
130
  <!-- generated:readme-tools-heading:start — edit scripts/docs-spec.mjs, then npm run plugin:sync -->
131
- ## 제공 도구 (150+ 개 · 22개 영역)
131
+ ## 제공 도구 (150+ 개 · 23개 영역)
132
132
  <!-- generated:readme-tools-heading:end -->
133
133
 
134
134
  <!-- generated:readme-tools-table:start — edit scripts/docs-spec.mjs, then npm run plugin:sync -->
@@ -142,6 +142,7 @@ export ANTHROPIC_API_KEY=sk-ant-...
142
142
  | Jenkins (크리덴셜 + 잡 + 빌드) | 13 | `jenkins_create_credential` / `jenkins_upload_keystore` / `jenkins_create_job` / `jenkins_update_job` / `jenkins_trigger_build` |
143
143
  | GA4 (Google Analytics 4) | 8 | `ga4_create_property` / `ga4_create_data_stream` / `ga4_plan_bigquery_link` / `ga4_create_bigquery_link` / `ga4_run_report` |
144
144
  | Search Console | 6 | `gsc_inspect_url` / `gsc_search_analytics` / `gsc_submit_sitemap` |
145
+ | 네이버 서치어드바이저 | 2 | `naver_check_page` / `naver_indexnow_submit` |
145
146
  | Google Ads | 6 | `googleads_list_campaigns` / `googleads_get_uac_report` / `googleads_get_campaign_report` |
146
147
  | Facebook | 6 | `facebook_post_photo` / `facebook_post_multi_photo` / `facebook_list_pages` |
147
148
  | Google Cloud IAM | 5 | `iam_create_service_account` / `iam_create_key` / `iam_add_iam_policy_binding` |
@@ -78,7 +78,7 @@ row for the job; batching two rows in one `select:` call is fine.
78
78
  | App Store weekly growth insight | `select:appstore_get_weekly_insight,appstore_get_sales_report` |
79
79
  | AdMob | `select:admob_list_accounts,admob_list_apps,admob_create_app,admob_create_ad_unit,admob_list_ad_units,admob_get_today_earnings,admob_get_report` |
80
80
  | Google Ads (UAC) | `select:googleads_config_status,googleads_save_config,googleads_list_accessible_customers,googleads_list_campaigns,googleads_get_campaign_report,googleads_get_uac_report` |
81
- | Search Console | `select:gsc_list_sites,gsc_list_sitemaps,gsc_get_sitemap,gsc_submit_sitemap,gsc_inspect_url,gsc_search_analytics` |
81
+ | Search indexing (Search Console / Naver) | `select:gsc_list_sites,gsc_list_sitemaps,gsc_get_sitemap,gsc_submit_sitemap,gsc_inspect_url,gsc_search_analytics,naver_check_page,naver_indexnow_submit` |
82
82
  | Social posting (Facebook / Instagram / Threads) | `select:facebook_current_config,facebook_save_config,facebook_list_pages,facebook_get_page,facebook_post_photo,facebook_post_multi_photo,instagram_save_config,instagram_get_account,instagram_post_image,instagram_post_carousel,threads_current_config,threads_save_config,threads_refresh_token,threads_get_account,threads_post,threads_post_video,threads_post_carousel` |
83
83
  | TikTok Business video publish | `select:tiktok_business_auth_status,tiktok_business_get_account,tiktok_business_get_video_settings,tiktok_business_plan_video_post,tiktok_business_publish_video,tiktok_business_get_publish_status,tiktok_business_list_publish_audits` |
84
84
  | Jenkins credentials + jobs | `select:jenkins_status,jenkins_save_config,jenkins_list_credentials,jenkins_create_credential,jenkins_delete_credential,jenkins_upload_keystore,jenkins_upload_playstore_sa,jenkins_list_jobs,jenkins_get_job_config,jenkins_create_job,jenkins_update_job` |
@@ -3,8 +3,11 @@ export interface GoogleAdsConfig {
3
3
  customerId: string;
4
4
  loginCustomerId?: string;
5
5
  }
6
- /** 하이픈 제거 (API는 숫자만 허용) */
7
- export declare function normalizeCustomerId(id: string): string;
6
+ /**
7
+ * 하이픈 제거 (API는 숫자만 허용). 손으로 적은 설정 파일엔 따옴표 없는 숫자(`"customerId": 1234567890`)도
8
+ * 오므로 숫자도 받는다 — 안 그러면 `.replace` 가 던져 "설정 없음" 으로 읽힌다.
9
+ */
10
+ export declare function normalizeCustomerId(id: string | number): string;
8
11
  export declare function saveConfig(cfg: GoogleAdsConfig): void;
9
12
  export declare function loadConfig(): GoogleAdsConfig | null;
10
13
  export declare function requireConfig(): GoogleAdsConfig;
@@ -4,9 +4,12 @@ import os from 'node:os';
4
4
  import { writeCredentialJson } from '#core/atomic-write.js';
5
5
  const CONFIG_DIR = path.join(os.homedir(), '.mimi-seed');
6
6
  const CONFIG_PATH = path.join(CONFIG_DIR, 'google-ads.json');
7
- /** 하이픈 제거 (API는 숫자만 허용) */
7
+ /**
8
+ * 하이픈 제거 (API는 숫자만 허용). 손으로 적은 설정 파일엔 따옴표 없는 숫자(`"customerId": 1234567890`)도
9
+ * 오므로 숫자도 받는다 — 안 그러면 `.replace` 가 던져 "설정 없음" 으로 읽힌다.
10
+ */
8
11
  export function normalizeCustomerId(id) {
9
- return id.replace(/-/g, '');
12
+ return String(id).replace(/-/g, '');
10
13
  }
11
14
  export function saveConfig(cfg) {
12
15
  const normalized = {
@@ -16,6 +19,14 @@ export function saveConfig(cfg) {
16
19
  };
17
20
  writeCredentialJson(CONFIG_PATH, normalized);
18
21
  }
22
+ /** 손으로 적은 ID 는 문자열이거나 따옴표 없는 정수다. 그 밖(빈 값 · 객체 · 소수)은 설정이 아니다. */
23
+ function readCustomerId(value) {
24
+ if (typeof value === 'string' && value.trim())
25
+ return normalizeCustomerId(value.trim());
26
+ if (typeof value === 'number' && Number.isSafeInteger(value) && value > 0)
27
+ return normalizeCustomerId(value);
28
+ return null;
29
+ }
19
30
  export function loadConfig() {
20
31
  if (!fs.existsSync(CONFIG_PATH))
21
32
  return null;
@@ -23,13 +34,15 @@ export function loadConfig() {
23
34
  const cfg = JSON.parse(fs.readFileSync(CONFIG_PATH, 'utf-8'));
24
35
  // 읽기 시점에도 정규화 — 하이픈 포함 값이 손으로 적히거나 구버전이 쓴 경우에도
25
36
  // URL/login-customer-id 헤더가 항상 숫자만 포함하도록 보장 (write 경로 의존 제거).
26
- return {
27
- ...cfg,
28
- customerId: normalizeCustomerId(cfg.customerId),
29
- loginCustomerId: cfg.loginCustomerId
30
- ? normalizeCustomerId(cfg.loginCustomerId)
31
- : undefined,
32
- };
37
+ const customerId = readCustomerId(cfg.customerId);
38
+ if (typeof cfg.developerToken !== 'string' || !cfg.developerToken || !customerId)
39
+ return null;
40
+ // 비어 있지 않은 loginCustomerId 가 이상하면 조용히 버리지 않는다 — MCC 헤더 없이 호출하면 엉뚱한 권한 오류가 난다.
41
+ const hasLogin = cfg.loginCustomerId !== undefined && cfg.loginCustomerId !== null && cfg.loginCustomerId !== '';
42
+ const loginCustomerId = hasLogin ? readCustomerId(cfg.loginCustomerId) : undefined;
43
+ if (loginCustomerId === null)
44
+ return null;
45
+ return { ...cfg, developerToken: cfg.developerToken, customerId, loginCustomerId };
33
46
  }
34
47
  catch {
35
48
  return null;
@@ -1,29 +1,37 @@
1
1
  /** Preserve provider diagnostics without exposing authentication material. */
2
2
  export function googleAdsError(status, body, secrets) {
3
3
  let message = `Google Ads API ${status}`;
4
+ let parsed;
4
5
  try {
5
- const parsed = JSON.parse(body);
6
- const error = (Array.isArray(parsed) ? parsed[0] : parsed)?.error;
7
- if (typeof error?.message === 'string')
8
- message += `: ${error.message}`;
9
- for (const detail of Array.isArray(error?.details) ? error.details : []) {
10
- for (const item of Array.isArray(detail.errors) ? detail.errors : []) {
11
- const codes = item.errorCode && typeof item.errorCode === 'object'
12
- ? Object.entries(item.errorCode).map(([key, value]) => `${key}=${String(value)}`).join(', ')
13
- : '';
14
- message += `\n${codes}${typeof item.message === 'string' ? `: ${item.message}` : ''}`;
15
- }
16
- for (const item of Array.isArray(detail.fieldViolations) ? detail.fieldViolations : []) {
17
- message += `\n${String(item.field ?? '')}: ${String(item.description ?? '')}`;
18
- }
19
- if (typeof detail.requestId === 'string')
20
- message += `\nrequestId=${detail.requestId}`;
21
- }
6
+ parsed = JSON.parse(body);
22
7
  }
23
8
  catch {
24
9
  // HTML/proxy bodies can contain arbitrary data; do not echo them.
10
+ parsed = undefined;
25
11
  message += ': non-JSON response';
26
12
  }
13
+ // Only the parse decides "non-JSON": a JSON body with odd entries (`details: [null]`) is walked
14
+ // null-safely instead of throwing into the non-JSON branch and mislabelling a real provider error.
15
+ const error = (Array.isArray(parsed) ? parsed[0] : parsed)?.error;
16
+ if (typeof error?.message === 'string')
17
+ message += `: ${error.message}`;
18
+ for (const detail of Array.isArray(error?.details) ? error.details : []) {
19
+ for (const item of Array.isArray(detail?.errors) ? detail.errors : []) {
20
+ if (!item || typeof item !== 'object')
21
+ continue;
22
+ const codes = item?.errorCode && typeof item.errorCode === 'object'
23
+ ? Object.entries(item.errorCode).map(([key, value]) => `${key}=${String(value)}`).join(', ')
24
+ : '';
25
+ message += `\n${codes}${typeof item?.message === 'string' ? `: ${item.message}` : ''}`;
26
+ }
27
+ for (const item of Array.isArray(detail?.fieldViolations) ? detail.fieldViolations : []) {
28
+ if (!item || typeof item !== 'object')
29
+ continue;
30
+ message += `\n${String(item?.field ?? '')}: ${String(item?.description ?? '')}`;
31
+ }
32
+ if (typeof detail?.requestId === 'string')
33
+ message += `\nrequestId=${detail.requestId}`;
34
+ }
27
35
  for (const secret of secrets.filter(Boolean).sort((a, b) => b.length - a.length)) {
28
36
  message = message.split(secret).join('[REDACTED]');
29
37
  }
@@ -104,7 +104,7 @@ async function main() {
104
104
  console.log(M.probing);
105
105
  try {
106
106
  const customers = await listAccessibleCustomers(auth, cfg);
107
- console.log(M.probeOk(customers.length));
107
+ console.log(M.probeOk(customers.resourceNames.length));
108
108
  }
109
109
  catch (e) {
110
110
  const msg = e instanceof Error ? e.message : String(e);
@@ -5,7 +5,14 @@ export interface DateRange {
5
5
  startDate: string;
6
6
  endDate: string;
7
7
  }
8
- export declare function listAccessibleCustomers(auth: OAuth2Client, cfg: GoogleAdsConfig): Promise<any>;
8
+ /**
9
+ * 응답은 배열이 아니라 `{ resourceNames: ['customers/<id>', …] }` 다 — 계정이 없으면 protobuf JSON 이
10
+ * 필드를 생략하므로 빈 배열로 채운다. 반환 타입을 고정해 두지 않으면 호출부가 `.length` 를 읽어도
11
+ * 컴파일러가 못 잡는다(setup-cli 가 "계정 undefined개" 를 찍었다).
12
+ */
13
+ export declare function listAccessibleCustomers(auth: OAuth2Client, cfg: GoogleAdsConfig): Promise<{
14
+ resourceNames: string[];
15
+ }>;
9
16
  export declare function listCampaigns(auth: OAuth2Client, cfg: GoogleAdsConfig): Promise<{
10
17
  id: string | undefined;
11
18
  name: string | undefined;
@@ -98,6 +98,11 @@ function microsToCurrency(micros) {
98
98
  return value / 1_000_000;
99
99
  }
100
100
  // ─── 접근 가능한 고객 목록 (API 연결 확인용) ─────────────────────
101
+ /**
102
+ * 응답은 배열이 아니라 `{ resourceNames: ['customers/<id>', …] }` 다 — 계정이 없으면 protobuf JSON 이
103
+ * 필드를 생략하므로 빈 배열로 채운다. 반환 타입을 고정해 두지 않으면 호출부가 `.length` 를 읽어도
104
+ * 컴파일러가 못 잡는다(setup-cli 가 "계정 undefined개" 를 찍었다).
105
+ */
101
106
  export async function listAccessibleCustomers(auth, cfg) {
102
107
  const accessToken = await getAccessToken(auth);
103
108
  const url = `${BASE}/customers:listAccessibleCustomers`;
@@ -109,7 +114,20 @@ export async function listAccessibleCustomers(auth, cfg) {
109
114
  if (!res.ok) {
110
115
  throw googleAdsError(res.status, text, [accessToken, cfg.developerToken]);
111
116
  }
112
- return JSON.parse(text);
117
+ // SyntaxError 메시지는 본문 앞부분을 그대로 싣는다 — search() 와 같이 프록시 본문을 노출하지 않는다.
118
+ let json;
119
+ try {
120
+ json = JSON.parse(text);
121
+ }
122
+ catch {
123
+ throw new Error('Google Ads returned invalid JSON.');
124
+ }
125
+ // search() 와 같이 200 + error 본문도 거절한다 — 안 그러면 setup 이 "계정 0개" 로 저장해 버린다.
126
+ const names = isRecord(json) && !('error' in json) ? json.resourceNames ?? [] : null;
127
+ if (!Array.isArray(names) || names.some((name) => typeof name !== 'string')) {
128
+ throw new Error('Google Ads returned an invalid accessible-customers response.');
129
+ }
130
+ return { resourceNames: names };
113
131
  }
114
132
  // ─── 캠페인 목록 ─────────────────────────────────────────────
115
133
  export async function listCampaigns(auth, cfg) {
@@ -44,6 +44,7 @@ const BRAND_PREFIXES = [
44
44
  ['bigquery_', 'BigQuery'],
45
45
  ['ga4_', 'GA4'],
46
46
  ['gsc_', 'Search Console'],
47
+ ['naver_', 'Naver'],
47
48
  ['googleads_', 'Google Ads'],
48
49
  ['facebook_', 'Facebook'],
49
50
  ['instagram_', 'Instagram'],
@@ -55,7 +56,7 @@ const BRAND_PREFIXES = [
55
56
  ['android_', 'Android'],
56
57
  ];
57
58
  const WORD_CASE = {
58
- iap: 'IAP', ios: 'iOS', sa: 'SA', url: 'URL', uac: 'UAC', ai: 'AI', iam: 'IAM', ga4: 'GA4',
59
+ iap: 'IAP', ios: 'iOS', sa: 'SA', url: 'URL', uac: 'UAC', ai: 'AI', iam: 'IAM', ga4: 'GA4', indexnow: 'IndexNow',
59
60
  playstore: 'Play Store', appstore: 'App Store', bigquery: 'BigQuery', youtube: 'YouTube',
60
61
  };
61
62
  /** `playstore_submit_release` → `Play Store: Submit release`. */
@@ -0,0 +1,44 @@
1
+ /**
2
+ * robots.txt 해석 (RFC 9309) — 네이버 검색로봇(Yeti)이 한 경로를 긁을 수 있는지 판정한다.
3
+ *
4
+ * 순수 함수만 둔다(네트워크 없음). 판정 규칙:
5
+ * - 크롤러 토큰과 같은 `User-agent` 그룹이 있으면 그 그룹(여러 개면 합친다)만, 없으면 `*` 그룹을 쓴다.
6
+ * `Yeti` 그룹이 따로 있으면 `*` 의 규칙은 Yeti 에게 **적용되지 않는다** — 흔한 오해다.
7
+ * `User-agent: Yeti/1.1` 처럼 버전이 붙어도 제품 토큰(`[A-Za-z_-]+`)만 비교한다.
8
+ * - 규칙은 가장 긴 패턴이 이기고, 길이가 같으면 allow 가 이긴다. `*` 와 닫는 `$` 를 지원한다.
9
+ * - 값이 빈 `Disallow:` 는 규칙이 아니다(전부 허용).
10
+ * - 비교 전에 패턴과 경로를 같은 퍼센트 인코딩으로 맞춘다 — `Disallow: /블로그` 는 `/%EB%B8%94…` 를 막는다.
11
+ *
12
+ * 매칭은 정규식이 아니라 선형 와일드카드 매처다. robots.txt 는 **점검 대상 사이트가 쓴 입력**이라
13
+ * `/*a*a*a…b` 같은 패턴으로 정규식 백트래킹을 폭발시켜 stdio 서버 전체를 멈출 수 있다.
14
+ */
15
+ export declare const NAVER_CRAWLER_TOKEN = "yeti";
16
+ export interface RobotsRule {
17
+ allow: boolean;
18
+ pattern: string;
19
+ }
20
+ export interface RobotsGroup {
21
+ agents: string[];
22
+ rules: RobotsRule[];
23
+ }
24
+ export interface ParsedRobots {
25
+ groups: RobotsGroup[];
26
+ sitemaps: string[];
27
+ }
28
+ export interface RobotsVerdict {
29
+ allowed: boolean;
30
+ /** 판정에 쓴 그룹 — 'yeti' 전용, '*', 또는 해당 그룹 없음. */
31
+ group: 'crawler' | 'wildcard' | 'none';
32
+ /** 판정을 결정한 규칙. 없으면 기본 허용. */
33
+ matchedRule?: RobotsRule;
34
+ }
35
+ export declare function parseRobots(text: string): ParsedRobots;
36
+ /** 비ASCII 는 UTF-8 퍼센트 인코딩, 기존 `%xx` 는 대문자 16진수로 — 패턴과 경로를 같은 표기로 맞춘다. */
37
+ export declare function normalizeRobotsPath(value: string): string;
38
+ /**
39
+ * `*`(0자 이상)와 닫는 `$`(끝 고정)만 있는 와일드카드 매칭. 앞쪽은 항상 고정(접두어 매칭).
40
+ * 투 포인터 백트래킹이라 최악 O(패턴 × 경로) — 정규식처럼 지수적으로 터지지 않는다.
41
+ */
42
+ export declare function robotsPatternMatches(pattern: string, path: string): boolean;
43
+ /** `path` 는 경로 + 쿼리(`/a?b=1`). 크롤러 토큰은 소문자로 비교한다. */
44
+ export declare function isAllowed(robots: ParsedRobots, path: string, crawler?: string): RobotsVerdict;
@@ -0,0 +1,126 @@
1
+ /**
2
+ * robots.txt 해석 (RFC 9309) — 네이버 검색로봇(Yeti)이 한 경로를 긁을 수 있는지 판정한다.
3
+ *
4
+ * 순수 함수만 둔다(네트워크 없음). 판정 규칙:
5
+ * - 크롤러 토큰과 같은 `User-agent` 그룹이 있으면 그 그룹(여러 개면 합친다)만, 없으면 `*` 그룹을 쓴다.
6
+ * `Yeti` 그룹이 따로 있으면 `*` 의 규칙은 Yeti 에게 **적용되지 않는다** — 흔한 오해다.
7
+ * `User-agent: Yeti/1.1` 처럼 버전이 붙어도 제품 토큰(`[A-Za-z_-]+`)만 비교한다.
8
+ * - 규칙은 가장 긴 패턴이 이기고, 길이가 같으면 allow 가 이긴다. `*` 와 닫는 `$` 를 지원한다.
9
+ * - 값이 빈 `Disallow:` 는 규칙이 아니다(전부 허용).
10
+ * - 비교 전에 패턴과 경로를 같은 퍼센트 인코딩으로 맞춘다 — `Disallow: /블로그` 는 `/%EB%B8%94…` 를 막는다.
11
+ *
12
+ * 매칭은 정규식이 아니라 선형 와일드카드 매처다. robots.txt 는 **점검 대상 사이트가 쓴 입력**이라
13
+ * `/*a*a*a…b` 같은 패턴으로 정규식 백트래킹을 폭발시켜 stdio 서버 전체를 멈출 수 있다.
14
+ */
15
+ export const NAVER_CRAWLER_TOKEN = 'yeti';
16
+ /** `Yeti/1.1 (+…)` → `yeti`. `*` 는 그대로. */
17
+ function agentToken(value) {
18
+ if (value.startsWith('*'))
19
+ return '*';
20
+ return (/^[a-z_-]+/i.exec(value)?.[0] ?? value).toLowerCase();
21
+ }
22
+ export function parseRobots(text) {
23
+ const groups = [];
24
+ const sitemaps = [];
25
+ let current = null;
26
+ // 연속된 User-agent 줄은 한 그룹을 공유한다. 규칙이 나온 뒤의 User-agent 는 새 그룹.
27
+ let collectingAgents = false;
28
+ for (const raw of (text.charCodeAt(0) === 0xfeff ? text.slice(1) : text).split(/\r\n|\r|\n/)) {
29
+ const hash = raw.indexOf('#');
30
+ const line = (hash === -1 ? raw : raw.slice(0, hash)).trim();
31
+ const sep = line.indexOf(':');
32
+ if (sep <= 0)
33
+ continue;
34
+ const field = line.slice(0, sep).trim().toLowerCase();
35
+ const value = line.slice(sep + 1).trim();
36
+ if (field === 'sitemap') {
37
+ if (value)
38
+ sitemaps.push(value);
39
+ continue;
40
+ }
41
+ if (field === 'user-agent') {
42
+ if (!current || !collectingAgents) {
43
+ current = { agents: [], rules: [] };
44
+ groups.push(current);
45
+ }
46
+ current.agents.push(agentToken(value));
47
+ collectingAgents = true;
48
+ continue;
49
+ }
50
+ if (field === 'allow' || field === 'disallow') {
51
+ collectingAgents = false;
52
+ if (!current || !value)
53
+ continue; // 그룹 밖 규칙, 빈 Disallow 는 무시
54
+ current.rules.push({ allow: field === 'allow', pattern: value });
55
+ }
56
+ }
57
+ return { groups, sitemaps };
58
+ }
59
+ /** 비ASCII 는 UTF-8 퍼센트 인코딩, 기존 `%xx` 는 대문자 16진수로 — 패턴과 경로를 같은 표기로 맞춘다. */
60
+ export function normalizeRobotsPath(value) {
61
+ return value
62
+ .replace(/%[0-9a-f]{2}/gi, (m) => m.toUpperCase())
63
+ .replace(/[\u0080-￿]+/g, (chunk) => {
64
+ try {
65
+ return encodeURIComponent(chunk);
66
+ }
67
+ catch {
68
+ return chunk; // 짝 없는 서로게이트 — 그대로 둔다
69
+ }
70
+ });
71
+ }
72
+ /**
73
+ * `*`(0자 이상)와 닫는 `$`(끝 고정)만 있는 와일드카드 매칭. 앞쪽은 항상 고정(접두어 매칭).
74
+ * 투 포인터 백트래킹이라 최악 O(패턴 × 경로) — 정규식처럼 지수적으로 터지지 않는다.
75
+ */
76
+ export function robotsPatternMatches(pattern, path) {
77
+ const anchored = pattern.endsWith('$');
78
+ // `$` 가 없으면 접두어 매칭 = 끝에 `*` 가 붙은 것과 같다.
79
+ const p = anchored ? pattern.slice(0, -1) : `${pattern}*`;
80
+ let i = 0;
81
+ let j = 0;
82
+ let star = -1;
83
+ let mark = 0;
84
+ while (i < path.length) {
85
+ if (j < p.length && p[j] === '*') {
86
+ star = j++;
87
+ mark = i;
88
+ }
89
+ else if (j < p.length && p[j] === path[i]) {
90
+ i++;
91
+ j++;
92
+ }
93
+ else if (star !== -1) {
94
+ j = star + 1;
95
+ i = ++mark;
96
+ }
97
+ else {
98
+ return false;
99
+ }
100
+ }
101
+ while (j < p.length && p[j] === '*')
102
+ j++;
103
+ return j === p.length;
104
+ }
105
+ /** `path` 는 경로 + 쿼리(`/a?b=1`). 크롤러 토큰은 소문자로 비교한다. */
106
+ export function isAllowed(robots, path, crawler = NAVER_CRAWLER_TOKEN) {
107
+ const token = crawler.toLowerCase();
108
+ const own = robots.groups.filter((g) => g.agents.includes(token));
109
+ const wildcard = robots.groups.filter((g) => g.agents.includes('*'));
110
+ const [groups, group] = own.length > 0
111
+ ? [own, 'crawler']
112
+ : wildcard.length > 0 ? [wildcard, 'wildcard'] : [[], 'none'];
113
+ const target = normalizeRobotsPath(path);
114
+ let best;
115
+ let bestLength = -1;
116
+ for (const rule of groups.flatMap((g) => g.rules)) {
117
+ const pattern = normalizeRobotsPath(rule.pattern);
118
+ if (!robotsPatternMatches(pattern, target))
119
+ continue;
120
+ if (pattern.length > bestLength || (pattern.length === bestLength && rule.allow && !best?.allow)) {
121
+ best = rule;
122
+ bestLength = pattern.length;
123
+ }
124
+ }
125
+ return { allowed: best ? best.allow : true, group, ...(best ? { matchedRule: best } : {}) };
126
+ }
@@ -0,0 +1,127 @@
1
+ import { type RobotsVerdict } from './robots.js';
2
+ /**
3
+ * 네이버 서치어드바이저 — 검색 노출·색인 도구.
4
+ *
5
+ * 서치어드바이저에는 **공개 웹마스터 API 가 없다**(사이트 등록·사이트맵/RSS 제출·리포트는 콘솔 전용).
6
+ * 자동화할 수 있는 면은 두 가지뿐이라 그것만 다룬다:
7
+ * - IndexNow: 네이버가 참여 엔진이다. 새/수정/삭제 URL 을 즉시 알린다. 인증 = 사이트에 올린 키 파일.
8
+ * - 크롤러 관점 점검: 네이버 검색로봇(Yeti)이 보는 robots.txt·상태 코드·리다이렉트·메타 태그를 확인한다.
9
+ * 어느 쪽도 `~/.mimi-seed/` 자격증명을 쓰지 않는다.
10
+ *
11
+ * 점검 대상 사이트의 응답(HTML·robots.txt·헤더)은 **적대적 입력**으로 다룬다: 본문은 상한까지만 읽고,
12
+ * 파싱은 선형 스캔으로만 한다. 정규식 백트래킹이나 무한 버퍼링 한 번이면 stdio 서버 전체가 멈춘다.
13
+ */
14
+ /** 네이버가 공개한 검색로봇 User-Agent. UA 로 봇을 막는 서버를 잡아내려고 그대로 보낸다. */
15
+ export declare const YETI_USER_AGENT = "Mozilla/5.0 (compatible; Yeti/1.1; +https://naver.me/spd)";
16
+ /** indexnow.org/searchengines.json 에 등록된 네이버 엔드포인트. */
17
+ export declare const NAVER_INDEXNOW_ENDPOINT = "https://searchadvisor.naver.com/indexnow";
18
+ /** IndexNow 프로토콜: POST 1회 최대 URL 수. */
19
+ export declare const INDEXNOW_MAX_URLS = 10000;
20
+ /** IndexNow 프로토콜: 8–128자, 영문·숫자·대시. */
21
+ export declare const INDEXNOW_KEY_PATTERN: RegExp;
22
+ /** RFC 9309: 크롤러는 robots.txt 를 최소 500 KiB 까지 해석해야 한다. 그 이상은 버린다. */
23
+ export declare const ROBOTS_MAX_BYTES: number;
24
+ /** HTML 은 이만큼만 읽는다 — head·본문 추정에는 충분하고 메모리는 묶인다. */
25
+ export declare const HTML_MAX_BYTES: number;
26
+ /**
27
+ * 응답 본문을 `maxBytes` 까지만 읽고 나머지는 끊는다. 인코딩은 content-type 의 charset
28
+ * (EUC-KR 등 레거시 한국어 페이지 포함), 모르면 UTF-8. BOM 은 TextDecoder 가 벗긴다.
29
+ */
30
+ export declare function readCapped(res: Response, maxBytes: number): Promise<{
31
+ text: string;
32
+ truncated: boolean;
33
+ }>;
34
+ export interface PageHead {
35
+ title?: string;
36
+ description?: string;
37
+ canonical?: string;
38
+ /** `<base href>` — 상대 canonical 을 풀 때 쓴다. */
39
+ base?: string;
40
+ lang?: string;
41
+ og: {
42
+ title?: string;
43
+ description?: string;
44
+ image?: string;
45
+ url?: string;
46
+ };
47
+ /** `<meta name="robots">` 와 `<meta name="yeti">` 의 지시어(소문자). */
48
+ robots: string[];
49
+ naverSiteVerification: boolean;
50
+ /** script/style 을 뺀 원본 HTML 의 보이는 글자 수. */
51
+ textLength: number;
52
+ scriptCount: number;
53
+ }
54
+ export declare function decodeEntities(value: string): string;
55
+ /** 속성 문자열 파서. HTML 처럼 같은 이름은 **처음 것**이 이긴다. */
56
+ export declare function parseAttrs(src: string): Record<string, string>;
57
+ export declare function parseHead(html: string): PageHead;
58
+ export interface XRobots {
59
+ /** 이 크롤러에 확실히 적용되는 지시어. */
60
+ applies: string[];
61
+ /**
62
+ * 다른 봇 접두어(`googlebot: …`) 뒤에 콤마로 이어진 지시어. 같은 헤더의 연속이면 그 봇 전용이지만,
63
+ * 헤더가 여러 개면 fetch 가 `, ` 로 합쳐 버려 **접두어 없는 별도 헤더**였을 수도 있다 — 구분할 수 없다.
64
+ */
65
+ ambiguous: string[];
66
+ }
67
+ /**
68
+ * `X-Robots-Tag` 에서 이 크롤러에 적용되는 지시어를 뽑는다.
69
+ * UA 접두어가 없는 지시어는 모두에게, `yeti: …` 는 Yeti 에게, `googlebot: …` 은 건너뛴다.
70
+ */
71
+ export declare function xRobotsDirectives(header: string | null, crawler?: string): XRobots;
72
+ /** HTTP `Link: <…>; rel="canonical"` 헤더의 대상. */
73
+ export declare function linkHeaderCanonical(header: string | null): string | undefined;
74
+ export type IssueLevel = 'error' | 'warn' | 'info';
75
+ export interface PageIssue {
76
+ level: IssueLevel;
77
+ code: string;
78
+ message: string;
79
+ }
80
+ export interface RedirectHop {
81
+ url: string;
82
+ status: number;
83
+ location: string;
84
+ }
85
+ export interface PageCheck {
86
+ url: string;
87
+ finalUrl: string;
88
+ status: number;
89
+ contentType: string | null;
90
+ redirects: RedirectHop[];
91
+ robotsTxt: {
92
+ url: string;
93
+ status: number | null;
94
+ verdict: RobotsVerdict | null;
95
+ sitemaps: string[];
96
+ };
97
+ directives: {
98
+ meta: string[];
99
+ header: string[];
100
+ headerAmbiguous: string[];
101
+ };
102
+ head: PageHead | null;
103
+ issues: PageIssue[];
104
+ summary: Record<IssueLevel, number>;
105
+ }
106
+ /** 네이버 검색로봇(Yeti) 관점에서 URL 1개를 점검한다. 쓰기 없음. */
107
+ export declare function checkPage(input: string): Promise<PageCheck>;
108
+ export interface IndexNowInput {
109
+ urls: string[];
110
+ key: string;
111
+ /** 키 파일이 사이트 루트(`/<key>.txt`)가 아닐 때만. */
112
+ keyLocation?: string;
113
+ }
114
+ export interface IndexNowResult {
115
+ endpoint: string;
116
+ status: number;
117
+ host: string;
118
+ keyLocation: string;
119
+ submitted: number;
120
+ }
121
+ /**
122
+ * 네이버 IndexNow 로 URL 변경을 알린다.
123
+ *
124
+ * 보내기 전에 키 파일을 직접 받아 key 와 같은지 확인한다 — 틀린 키는 엔드포인트가 403 만 돌려주고,
125
+ * 같은 호스트에 틀린 키로 반복 제출하면 스팸 판정 위험만 커진다. 같은 URL 재제출은 안전하다.
126
+ */
127
+ export declare function submitIndexNow({ urls, key, keyLocation }: IndexNowInput): Promise<IndexNowResult>;