@doist/doistbot-cli 1.0.12 β 1.0.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/config.js +1 -1
- package/package.json +3 -3
- package/sandbox/dist/core/pi.js +1 -1
- package/sandbox/dist/core/repository-map.js +1 -1
- package/sandbox/dist/tasks/issue-summarize/context.js +1 -0
- package/sandbox/dist/tasks/issue-summarize/model.js +24 -7
- package/sandbox/dist/tasks/issue-summarize/summarize.js +6 -4
- package/sandbox/dist/tasks/issue-triage/hero-group-map.js +47 -3
- package/sandbox/dist/tasks/issue-triage/resolution.js +3 -0
- package/sandbox/src/review/prompts/ai-internal-tools.md +3 -3
- package/sandbox/src/review/prompts/backend-repository-standards.md +102 -0
- package/sandbox/src/review/prompts/outline-map.json +14 -10
- package/sandbox/src/review/prompts/service-production-readiness.md +2 -2
- package/sandbox/src/review/prompts/which-cloud-and-where-to-deploy.md +0 -344
package/dist/config.js
CHANGED
|
@@ -29,7 +29,7 @@ const ALL_REVIEW_FOCUSES = [
|
|
|
29
29
|
'standards',
|
|
30
30
|
];
|
|
31
31
|
const OPENROUTER_REVIEW_MODELS = ['deepseek', 'zai'];
|
|
32
|
-
const CLI_GEMINI_REVIEW_MODEL = 'gemini-3.
|
|
32
|
+
const CLI_GEMINI_REVIEW_MODEL = 'gemini-3.8-flash';
|
|
33
33
|
function createCliGeminiCredential(apiKey) {
|
|
34
34
|
return {
|
|
35
35
|
apiKey,
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@doist/doistbot-cli",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.13",
|
|
4
4
|
"description": "Local self-review CLI for Doistbot",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.js",
|
|
@@ -42,8 +42,8 @@
|
|
|
42
42
|
"dependencies": {
|
|
43
43
|
"@earendil-works/pi-ai": "0.85.1",
|
|
44
44
|
"@earendil-works/pi-coding-agent": "0.85.1",
|
|
45
|
-
"@inquirer/checkbox": "5.2.
|
|
46
|
-
"@inquirer/password": "5.2.
|
|
45
|
+
"@inquirer/checkbox": "5.2.5",
|
|
46
|
+
"@inquirer/password": "5.2.2",
|
|
47
47
|
"@napi-rs/keyring": "1.3.0",
|
|
48
48
|
"@opentelemetry/api": "1.9.1",
|
|
49
49
|
"ajv": "8.20.0",
|
package/sandbox/dist/core/pi.js
CHANGED
|
@@ -63,7 +63,7 @@ export const DEFAULT_OPENROUTER_MODELS = {
|
|
|
63
63
|
zai: 'z-ai/glm-5.3-flash',
|
|
64
64
|
};
|
|
65
65
|
// Gemini flash fallback model.
|
|
66
|
-
export const DEFAULT_GEMINI_FLASH_MODEL = 'gemini-3.
|
|
66
|
+
export const DEFAULT_GEMINI_FLASH_MODEL = 'gemini-3.8-flash';
|
|
67
67
|
export const MAX_PI_THINKING_BY_PROVIDER = {
|
|
68
68
|
deepseek: 'high',
|
|
69
69
|
gemini: 'high',
|
|
@@ -1 +1 @@
|
|
|
1
|
-
export { APP_LABEL_ALIASES, APP_TO_ROUTE_SELECTORS, findIssueRoute, findRepositoryRoutes, findRepositoryTarget, PRODUCT_LABEL_ALIASES, PRODUCT_ROUTES, RESOLUTION_PRECEDENCE, TEAM_LABEL_ALIASES, TEAM_TO_ROUTE_SELECTORS, } from 'doistbot-issue-routing';
|
|
1
|
+
export { APP_LABEL_ALIASES, APP_TO_ROUTE_SELECTORS, findIntegrationRoute, findIntegrationRoutes, findIssueRoute, findRepositoryRoutes, findRepositoryTarget, INTEGRATION_LABEL_ALIASES, INTEGRATION_ROUTES, PRODUCT_LABEL_ALIASES, PRODUCT_ROUTES, RESOLUTION_PRECEDENCE, TEAM_LABEL_ALIASES, TEAM_TO_ROUTE_SELECTORS, } from 'doistbot-issue-routing';
|
|
@@ -11,6 +11,7 @@ export function parseSummarizeContext(env = process.env) {
|
|
|
11
11
|
appId: requiredEnv('APP_ID', env),
|
|
12
12
|
privateKey: requiredEnv('PRIVATE_KEY', env),
|
|
13
13
|
githubToken: requiredEnv('GITHUB_TOKEN', env),
|
|
14
|
+
geminiApiKey: requiredEnv('GEMINI_API_KEY', env),
|
|
14
15
|
openRouterApiKey: requiredEnv('OPENROUTER_API_KEY', env),
|
|
15
16
|
commentId,
|
|
16
17
|
reactionId,
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import { SpanStatusCode } from '@opentelemetry/api';
|
|
2
2
|
import { logger } from '../../core/logging.js';
|
|
3
|
-
import { buildPiImageContent, createPiInvoker, DEFAULT_OPENROUTER_MODELS } from '../../core/pi.js';
|
|
3
|
+
import { buildPiImageContent, createPiInvoker, DEFAULT_GEMINI_FLASH_MODEL, DEFAULT_OPENROUTER_MODELS, } from '../../core/pi.js';
|
|
4
4
|
import { withPhase } from '../../core/tracing.js';
|
|
5
5
|
const ISSUE_SUMMARIZE_TIMEOUT_MS = 5 * 60 * 1000;
|
|
6
|
-
const ISSUE_SUMMARIZE_MULTIMODAL_MODEL =
|
|
6
|
+
const ISSUE_SUMMARIZE_MULTIMODAL_MODEL = DEFAULT_GEMINI_FLASH_MODEL;
|
|
7
7
|
const SUMMARIZE_TEXT_PROVIDER = 'zai';
|
|
8
8
|
// Single source of truth: inherit the repo-wide Z.ai default so a default-model
|
|
9
9
|
// bump in core/pi.ts moves summarize too, and the executed model matches the
|
|
@@ -14,7 +14,7 @@ const SUMMARIZE_MULTIMODAL_PROVIDER = 'gcp.gemini';
|
|
|
14
14
|
// `issue-summarize.model`; each chunk summary uses `issue-summarize.chunk`.
|
|
15
15
|
const SUMMARIZE_MODEL_PHASE = 'issue-summarize.model';
|
|
16
16
|
const SUMMARIZE_CHUNK_PHASE = 'issue-summarize.chunk';
|
|
17
|
-
export function
|
|
17
|
+
export function createIssueSummarizeTextModel(apiKey) {
|
|
18
18
|
return createPiInvoker({
|
|
19
19
|
provider: 'zai',
|
|
20
20
|
apiKey,
|
|
@@ -23,6 +23,15 @@ export function createIssueSummarizeModel(apiKey) {
|
|
|
23
23
|
capabilityPreset: 'read-only',
|
|
24
24
|
});
|
|
25
25
|
}
|
|
26
|
+
export function createIssueSummarizeMultimodalModel(apiKey) {
|
|
27
|
+
return createPiInvoker({
|
|
28
|
+
provider: 'gemini',
|
|
29
|
+
apiKey,
|
|
30
|
+
piProviderId: 'google',
|
|
31
|
+
cwd: process.cwd(),
|
|
32
|
+
capabilityPreset: 'read-only',
|
|
33
|
+
});
|
|
34
|
+
}
|
|
26
35
|
async function invokeIssueSummarizeTextModel({ model, prompt, phaseName, }) {
|
|
27
36
|
return withPhase(phaseName, {
|
|
28
37
|
'gen_ai.operation.name': 'chat',
|
|
@@ -115,11 +124,11 @@ async function invokeMultimodalSummarizeModel({ model, prompt, mediaImages, }) {
|
|
|
115
124
|
}
|
|
116
125
|
});
|
|
117
126
|
}
|
|
118
|
-
export async function invokeIssueSummarizeModel({
|
|
127
|
+
export async function invokeIssueSummarizeModel({ textModel, multimodalModel, prompt, mediaImages, }) {
|
|
119
128
|
if (mediaImages.length === 0) {
|
|
120
129
|
logger.info('Using text-only summarize model request');
|
|
121
130
|
return await invokeIssueSummarizeTextModel({
|
|
122
|
-
model,
|
|
131
|
+
model: textModel,
|
|
123
132
|
prompt,
|
|
124
133
|
phaseName: SUMMARIZE_MODEL_PHASE,
|
|
125
134
|
});
|
|
@@ -127,12 +136,20 @@ export async function invokeIssueSummarizeModel({ model, prompt, mediaImages, })
|
|
|
127
136
|
logger.info('Using multimodal summarize model request', {
|
|
128
137
|
mediaImages: mediaImages.length,
|
|
129
138
|
});
|
|
130
|
-
const multimodal = await invokeMultimodalSummarizeModel({
|
|
139
|
+
const multimodal = await invokeMultimodalSummarizeModel({
|
|
140
|
+
model: multimodalModel,
|
|
141
|
+
prompt,
|
|
142
|
+
mediaImages,
|
|
143
|
+
});
|
|
131
144
|
if (multimodal) {
|
|
132
145
|
return multimodal;
|
|
133
146
|
}
|
|
134
147
|
// Multimodal failed β fall back to a separate text request (own model span).
|
|
135
|
-
return await invokeIssueSummarizeTextModel({
|
|
148
|
+
return await invokeIssueSummarizeTextModel({
|
|
149
|
+
model: textModel,
|
|
150
|
+
prompt,
|
|
151
|
+
phaseName: SUMMARIZE_MODEL_PHASE,
|
|
152
|
+
});
|
|
136
153
|
}
|
|
137
154
|
function buildChunkSummaryPrompt({ chunk, chunkIndex, selectedChunkCount, totalChunks, }) {
|
|
138
155
|
return `You are preparing an intermediate summary for a GitHub issue discussion.
|
|
@@ -7,7 +7,7 @@ import { resolveWriteOctokit } from '../../core/shared.js';
|
|
|
7
7
|
import { withPhase } from '../../core/tracing.js';
|
|
8
8
|
import { parseSummarizeContext } from './context.js';
|
|
9
9
|
import { buildMediaPromptImages, extractMediaReferences, formatMediaContext, inspectMediaReferences, } from './media.js';
|
|
10
|
-
import {
|
|
10
|
+
import { createIssueSummarizeMultimodalModel, createIssueSummarizeTextModel, invokeIssueSummarizeModel, summarizeChunks, } from './model.js';
|
|
11
11
|
import { buildFallbackSummary, ensureStructuredSummary, stripVisualContextSection, } from './output.js';
|
|
12
12
|
import { fillTemplate, loadPromptTemplate } from './prompt.js';
|
|
13
13
|
import { buildThreadEntries, prepareThreadPrompt } from './thread.js';
|
|
@@ -41,7 +41,8 @@ function summarizeMediaInspectionStats(mediaInspections) {
|
|
|
41
41
|
async function summarizeIssue({ ctx, issue, entries, mediaReferences, mediaInspections, }) {
|
|
42
42
|
const template = loadPromptTemplate();
|
|
43
43
|
const threadPrompt = prepareThreadPrompt(entries);
|
|
44
|
-
const
|
|
44
|
+
const textModel = createIssueSummarizeTextModel(ctx.openRouterApiKey);
|
|
45
|
+
const multimodalModel = createIssueSummarizeMultimodalModel(ctx.geminiApiKey);
|
|
45
46
|
logger.info('Prepared summarize thread prompt', {
|
|
46
47
|
entryCount: entries.length,
|
|
47
48
|
chunked: threadPrompt.chunked,
|
|
@@ -50,7 +51,7 @@ async function summarizeIssue({ ctx, issue, entries, mediaReferences, mediaInspe
|
|
|
50
51
|
});
|
|
51
52
|
const threadContent = threadPrompt.chunked
|
|
52
53
|
? await summarizeChunks({
|
|
53
|
-
model:
|
|
54
|
+
model: textModel,
|
|
54
55
|
chunks: threadPrompt.chunks,
|
|
55
56
|
totalChunks: threadPrompt.totalChunks,
|
|
56
57
|
})
|
|
@@ -75,7 +76,8 @@ async function summarizeIssue({ ctx, issue, entries, mediaReferences, mediaInspe
|
|
|
75
76
|
mediaImagesIncluded: mediaPromptImages.length,
|
|
76
77
|
});
|
|
77
78
|
const summary = await invokeIssueSummarizeModel({
|
|
78
|
-
|
|
79
|
+
textModel,
|
|
80
|
+
multimodalModel,
|
|
79
81
|
prompt,
|
|
80
82
|
mediaImages: mediaPromptImages,
|
|
81
83
|
});
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { normalizeScopedLabelValue, scopedLabelValue } from '../../core/label-parsing.js';
|
|
2
|
-
import { APP_LABEL_ALIASES, findIssueRoute, findRepositoryRoutes, findRepositoryTarget as findRoutingRepositoryTarget, PRODUCT_LABEL_ALIASES, PRODUCT_ROUTES, } from '../../core/repository-map.js';
|
|
2
|
+
import { APP_LABEL_ALIASES, findIntegrationRoute, findIntegrationRoutes, findIssueRoute, findRepositoryRoutes, findRepositoryTarget as findRoutingRepositoryTarget, INTEGRATION_LABEL_ALIASES, PRODUCT_LABEL_ALIASES, PRODUCT_ROUTES, } from '../../core/repository-map.js';
|
|
3
3
|
const HERO_GROUPS = {
|
|
4
4
|
'android-hero': { key: 'android-hero', mention: '@Doist/android-hero' },
|
|
5
5
|
'app-copy-hero': { key: 'app-copy-hero', mention: '@Doist/app-copy-hero' },
|
|
@@ -63,11 +63,18 @@ export function resolveHeroTeamSlug(target) {
|
|
|
63
63
|
function resolveAppLabelAlias(value) {
|
|
64
64
|
return APP_LABEL_ALIASES[normalizeScopedLabelValue(value)];
|
|
65
65
|
}
|
|
66
|
+
function resolveIntegrationLabelAlias(value) {
|
|
67
|
+
return INTEGRATION_LABEL_ALIASES[normalizeScopedLabelValue(value)];
|
|
68
|
+
}
|
|
66
69
|
export function isKnownHeroRoutingLabel(label) {
|
|
67
70
|
const teamValue = scopedLabelValue(label, 'Team');
|
|
68
71
|
if (teamValue && TEAM_LABEL_TO_HERO[normalizeScopedLabelValue(teamValue)]) {
|
|
69
72
|
return true;
|
|
70
73
|
}
|
|
74
|
+
const integrationValue = scopedLabelValue(label, 'Integration');
|
|
75
|
+
if (integrationValue && resolveIntegrationLabelAlias(integrationValue)) {
|
|
76
|
+
return true;
|
|
77
|
+
}
|
|
71
78
|
const appValue = scopedLabelValue(label, 'App');
|
|
72
79
|
if (!appValue) {
|
|
73
80
|
return false;
|
|
@@ -91,6 +98,28 @@ function resolveProduct(labels) {
|
|
|
91
98
|
}
|
|
92
99
|
return products.size === 1 ? products.values().next().value : undefined;
|
|
93
100
|
}
|
|
101
|
+
function resolveIntegrationHeroKeys(labels) {
|
|
102
|
+
const product = resolveProduct(labels);
|
|
103
|
+
const heroKeys = new Set();
|
|
104
|
+
for (const label of labels) {
|
|
105
|
+
const integrationValue = scopedLabelValue(label, 'Integration');
|
|
106
|
+
if (!integrationValue) {
|
|
107
|
+
continue;
|
|
108
|
+
}
|
|
109
|
+
const integration = resolveIntegrationLabelAlias(integrationValue);
|
|
110
|
+
if (!integration) {
|
|
111
|
+
continue;
|
|
112
|
+
}
|
|
113
|
+
const route = findIntegrationRoute(integration);
|
|
114
|
+
// An integration from another product is not in scope for this issue, so
|
|
115
|
+
// it must not contribute a hero the repository resolver would never pick.
|
|
116
|
+
if (product && route.product !== product) {
|
|
117
|
+
continue;
|
|
118
|
+
}
|
|
119
|
+
heroKeys.add(route.heroTeam);
|
|
120
|
+
}
|
|
121
|
+
return heroKeys;
|
|
122
|
+
}
|
|
94
123
|
function resolveAppRouteHeroKeys(labels) {
|
|
95
124
|
const routeIds = new Set();
|
|
96
125
|
for (const label of labels) {
|
|
@@ -99,7 +128,9 @@ function resolveAppRouteHeroKeys(labels) {
|
|
|
99
128
|
continue;
|
|
100
129
|
}
|
|
101
130
|
const appLabel = resolveAppLabelAlias(appValue);
|
|
102
|
-
|
|
131
|
+
// `all` spans every route and `integration` names no route of its own β
|
|
132
|
+
// the accompanying `Integration:*` label supplies that hero instead.
|
|
133
|
+
if (appLabel && appLabel !== 'all' && appLabel !== 'integration') {
|
|
103
134
|
routeIds.add(appLabel);
|
|
104
135
|
}
|
|
105
136
|
}
|
|
@@ -135,6 +166,13 @@ export function resolveHeroGroup(labels, primaryRepo) {
|
|
|
135
166
|
if (teamHeroKeys.size > 1) {
|
|
136
167
|
return null;
|
|
137
168
|
}
|
|
169
|
+
const integrationHeroKeys = resolveIntegrationHeroKeys(labels);
|
|
170
|
+
if (integrationHeroKeys.size === 1) {
|
|
171
|
+
return HERO_GROUPS[integrationHeroKeys.values().next().value];
|
|
172
|
+
}
|
|
173
|
+
if (integrationHeroKeys.size > 1) {
|
|
174
|
+
return null;
|
|
175
|
+
}
|
|
138
176
|
const appHeroKeys = resolveAppRouteHeroKeys(labels);
|
|
139
177
|
if (appHeroKeys.size === 1) {
|
|
140
178
|
return HERO_GROUPS[appHeroKeys.values().next().value];
|
|
@@ -143,7 +181,13 @@ export function resolveHeroGroup(labels, primaryRepo) {
|
|
|
143
181
|
return null;
|
|
144
182
|
}
|
|
145
183
|
if (primaryRepo) {
|
|
146
|
-
|
|
184
|
+
// A repository carried by both a product route and an integration route β
|
|
185
|
+
// todoist-macos owns the Safari extension, for example β keeps the product
|
|
186
|
+
// route's hero, so integration routes only cover integration-only repos.
|
|
187
|
+
const productRoutes = findRepositoryRoutes(primaryRepo);
|
|
188
|
+
const repositoryHeroKeys = new Set(productRoutes.length > 0
|
|
189
|
+
? productRoutes.map((route) => route.heroTeam)
|
|
190
|
+
: findIntegrationRoutes(primaryRepo).map((route) => route.heroTeam));
|
|
147
191
|
if (repositoryHeroKeys.size === 1) {
|
|
148
192
|
return HERO_GROUPS[repositoryHeroKeys.values().next().value];
|
|
149
193
|
}
|
|
@@ -12,10 +12,13 @@ export async function resolveTriageScope(params) {
|
|
|
12
12
|
parsed_products: result.parsedLabels.productLabels,
|
|
13
13
|
parsed_apps: result.parsedLabels.appLabels,
|
|
14
14
|
parsed_teams: result.parsedLabels.teamLabels,
|
|
15
|
+
parsed_integrations: result.parsedLabels.integrationLabels,
|
|
15
16
|
matched_routes: result.matchedRoutes,
|
|
17
|
+
matched_integrations: result.matchedIntegrations,
|
|
16
18
|
unknown_product_labels: result.parsedLabels.unknownProductLabels,
|
|
17
19
|
unknown_app_labels: result.parsedLabels.unknownAppLabels,
|
|
18
20
|
unknown_team_labels: result.parsedLabels.unknownTeamLabels,
|
|
21
|
+
unknown_integration_labels: result.parsedLabels.unknownIntegrationLabels,
|
|
19
22
|
});
|
|
20
23
|
span.setAttributes({
|
|
21
24
|
'resolution.status': result.status,
|
|
@@ -202,7 +202,7 @@ flowchart TD
|
|
|
202
202
|
Agent --> NonSensitive
|
|
203
203
|
```
|
|
204
204
|
|
|
205
|
-
> Note: see [`examples/internal-ai-service/`](https://github.com/Doist/
|
|
205
|
+
> Note: see [`examples/internal-ai-service/`](https://github.com/Doist/standards/tree/main/examples/internal-ai-service/) for the server-side service and [`examples/local-cli/`](https://github.com/Doist/standards/tree/main/examples/local-cli/) for the thin local CLI. The example implements Zendesk only; the same pattern applies to Sentry/Datadog.
|
|
206
206
|
|
|
207
207
|
### How to read this diagram
|
|
208
208
|
|
|
@@ -236,9 +236,9 @@ Think of the diagram as three separate rooms, each with different access rules:
|
|
|
236
236
|
|
|
237
237
|
1. **Remove** the direct connections to Zendesk, Sentry, and Datadog from your local AI setup.
|
|
238
238
|
|
|
239
|
-
2. **Set up an internal service** (the Platform team can help) that fetches and cleans data before passing it to the AI β see [`examples/internal-ai-service/main.go`](https://github.com/Doist/
|
|
239
|
+
2. **Set up an internal service** (the Platform team can help) that fetches and cleans data before passing it to the AI β see [`examples/internal-ai-service/main.go`](https://github.com/Doist/standards/tree/main/examples/internal-ai-service/main.go) for a working reference. It should only return what's needed to answer the question β never raw personal information.
|
|
240
240
|
|
|
241
|
-
3. **Point your local tool at that internal service** instead of at the third-party tools directly β see [`examples/local-cli/main.go`](https://github.com/Doist/
|
|
241
|
+
3. **Point your local tool at that internal service** instead of at the third-party tools directly β see [`examples/local-cli/main.go`](https://github.com/Doist/standards/tree/main/examples/local-cli/main.go) for a working reference.
|
|
242
242
|
|
|
243
243
|
4. **Keep your Github, Twist, and Outline connections as-is** β those don't hold user personal data, so they're fine to use directly.
|
|
244
244
|
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
# Standard: Backend Repositories
|
|
2
|
+
|
|
3
|
+
| π Document Type | Team Standard (Backend) |
|
|
4
|
+
| :--------------- | :------------------------------------------------------------------------------------------------------------------------ |
|
|
5
|
+
| π― Status | Draft |
|
|
6
|
+
| π€ Owner | Felipe Rodrigues |
|
|
7
|
+
| π
Last Reviewed | 2026-09-08 |
|
|
8
|
+
| ποΈ References | [Standard: Service Production Readiness](https://handbook.doist.com/doc/standard-service-production-readiness-qgNRng8TJx) |
|
|
9
|
+
| π Version | 0.2 |
|
|
10
|
+
|
|
11
|
+
This standard applies to **all backend repositories at Doist, regardless of language or current ownership.** Today that means `Todoist` and other repositories already owned by the Backend team as well as newer backend services such as `todoist-id` and `automations` - currently developed by squads, and expected to come under Backend team ownership over time.
|
|
12
|
+
|
|
13
|
+
Meeting this standard is part of what makes that handover possible.
|
|
14
|
+
|
|
15
|
+
## Tests
|
|
16
|
+
|
|
17
|
+
Every repository must have automated tests. A repository with no tests is not production-ready.
|
|
18
|
+
|
|
19
|
+
- Tests must run in CI on every pull request.
|
|
20
|
+
- Tests must run in CI on every merge to `main`.
|
|
21
|
+
- Deployments must be gated on tests passing. In GitHub Actions, the deploy job must depend on the test job β via `needs: test` in the same workflow, or by invoking the test workflow as a reusable workflow and depending on that job. A deploy that races its own test run is not gated.
|
|
22
|
+
|
|
23
|
+
## Branch Protection
|
|
24
|
+
|
|
25
|
+
Tests that exist but are never enforced provide false confidence. The `main` branch must be protected:
|
|
26
|
+
|
|
27
|
+
- Changes land via pull request β no direct pushes to `main`.
|
|
28
|
+
- Test, lint, and type-check jobs are configured as **required status checks**, so a PR cannot merge while they fail or haven't run.
|
|
29
|
+
|
|
30
|
+
## Code Quality Hooks
|
|
31
|
+
|
|
32
|
+
- Pre-commit hooks must be used to enforce formatting, linting, and repo hygiene. The runner is language-specific: [prek](https://prek.j178.dev) in Python repositories, [husky](https://typicode.github.io/husky/) in JavaScript/TypeScript ones.
|
|
33
|
+
- The same checks must also run in CI on every pull request and merge to `main` (e.g. `prek run --all-files`, or the `npm` scripts the hooks call). Hooks that only run locally are advisory; developers can skip or misconfigure them.
|
|
34
|
+
|
|
35
|
+
## README
|
|
36
|
+
|
|
37
|
+
Every repository must have a `README.md` with at least:
|
|
38
|
+
|
|
39
|
+
- Project description
|
|
40
|
+
- How to run the project locally
|
|
41
|
+
- How to run tests
|
|
42
|
+
- How to deploy the project
|
|
43
|
+
|
|
44
|
+
## Dependency Updates
|
|
45
|
+
|
|
46
|
+
Automated dependency updates (Renovate or Dependabot) must be enabled. Update PRs go through the same CI gates as any other change.
|
|
47
|
+
|
|
48
|
+
## Python Requirements
|
|
49
|
+
|
|
50
|
+
Python repositories must use the following stack (mirroring the main Todoist codebase):
|
|
51
|
+
|
|
52
|
+
| Concern | Tool |
|
|
53
|
+
| ------------------------------- | ---------------------------- |
|
|
54
|
+
| Project & dependency management | `uv` |
|
|
55
|
+
| Minimum Python version | 3.13 |
|
|
56
|
+
| Tests | `pytest` |
|
|
57
|
+
| Linting & formatting | `ruff check` + `ruff format` |
|
|
58
|
+
| Git hooks | `prek` |
|
|
59
|
+
| Type checking | `mypy` and `pyrefly` |
|
|
60
|
+
| Dead code detection | `vulture` |
|
|
61
|
+
|
|
62
|
+
- Dependencies in `pyproject.toml` must be pinned with upper bounds (`>=X.Y,<X+1`).
|
|
63
|
+
- Type hints are mandatory on production code.
|
|
64
|
+
|
|
65
|
+
## JavaScript / TypeScript Requirements
|
|
66
|
+
|
|
67
|
+
JavaScript and TypeScript repositories must use the following stack (mirroring `automations`, our newest backend TS project):
|
|
68
|
+
|
|
69
|
+
| Concern | Tool |
|
|
70
|
+
| ------------------------------- | ----------------------------------------------- |
|
|
71
|
+
| Language | TypeScript |
|
|
72
|
+
| Runtime | Node.js β current LTS major |
|
|
73
|
+
| Project & dependency management | `npm` (workspaces for monorepos) |
|
|
74
|
+
| Tests | `vitest` |
|
|
75
|
+
| Linting & formatting | `oxlint` + `oxfmt` |
|
|
76
|
+
| Git hooks | `husky` |
|
|
77
|
+
| Type checking | `tsc --noEmit`, exposed as `npm run type-check` |
|
|
78
|
+
|
|
79
|
+
- Production code must be TypeScript. Plain JavaScript only where a tool requires it.
|
|
80
|
+
- The Node version must be pinned in `engines` and in a version file (`.node-version`, which `actions/setup-node` reads via `node-version-file`; `.nvmrc` alongside it if people use `nvm`). Version files pin Node only, so pin the npm major in `engines` as well.
|
|
81
|
+
- `engine-strict=true` must be set in `.npmrc`. Without it npm only warns on an `engines` mismatch, and the pins above do not actually stop local, CI and container builds from drifting apart.
|
|
82
|
+
- Dependencies in `package.json` must be pinned to exact versions, with the lockfile committed.
|
|
83
|
+
- The TypeScript config must extend the shared `@doist/tsconfig` base and keep `strict` on.
|
|
84
|
+
|
|
85
|
+
## Related Documents
|
|
86
|
+
|
|
87
|
+
This standard complements the Doist-wide [Service Production Readiness](https://handbook.doist.com/doc/standard-service-production-readiness-qgNRng8TJx) standard: that one covers how a service must _behave_ in production while this one covers the baseline health a _repository_ must have before its code reaches production.
|
|
88
|
+
|
|
89
|
+
Related engineering-wide guidelines:
|
|
90
|
+
|
|
91
|
+
- [GitHub Repository Guidelines](https://handbook.doist.com/doc/github-repository-guidelines-iJNZ0D2vLN)
|
|
92
|
+
- [Automate linting and formatting](https://handbook.doist.com/doc/automate-linting-and-formatting-omSECLKiJI)
|
|
93
|
+
- [Keeping dependencies up-to-date](https://handbook.doist.com/doc/keeping-dependencies-up-to-date-Qf28E929N5)
|
|
94
|
+
|
|
95
|
+
---
|
|
96
|
+
|
|
97
|
+
## Open questions/Possible future expansions to the Standard
|
|
98
|
+
|
|
99
|
+
- Type checkers: Todoist currently runs `mypy`, `ty`, and `pyrefly` in parallel, but that is an evaluation state, not a recommendation (see [Choosing type checkers for Todoist](https://comms.todoist.com/69/ch/BH4VumxwScG1NAiomtcBq/t/CbVia3jW38koSRmHWXSeV/)). A June 2026 attempt to remove mypy was reverted after it left real gaps (pyrefly's legacy config disabled core checks; ty lacks Pydantic support). Current direction: pyrefly in strict mode is the candidate primary, mypy stays as the mature baseline, ty is a secondary bet pending Pydantic support. `mypy` + `pyrefly` therefore remains the right baseline for new repos; revisit when ty matures.
|
|
100
|
+
- Maybe create a [Copier](https://copier.readthedocs.io/en/stable/) template for Python repos β and an equivalent for TS ones β so new services start compliant? Or even a small template repo with βsane defaultsβ for our tooling.
|
|
101
|
+
- Dead code detection has no JS/TS entry yet. Python repos run `vulture`; `knip` is the closest equivalent, but nothing runs it today.
|
|
102
|
+
- Define the baseline ruff rule set / prek hook list based on the Todoist codebase (its `prek.toml` is a good starting point: ruff, standard pre-commit-hooks, shellcheck, hadolint, mypy/type checkers, vulture on pre-push).
|
|
@@ -1,22 +1,26 @@
|
|
|
1
1
|
{
|
|
2
2
|
"workload-placement.md": {
|
|
3
|
-
"id": "
|
|
4
|
-
"url": "https://handbook.doist.com/doc/standard-workload-placement-
|
|
3
|
+
"id": "",
|
|
4
|
+
"url": "https://handbook.doist.com/doc/standard-workload-placement-"
|
|
5
5
|
},
|
|
6
6
|
"secrets-management.md": {
|
|
7
|
-
"id": "
|
|
8
|
-
"url": "https://handbook.doist.com/doc/standard-secrets-management-
|
|
7
|
+
"id": "",
|
|
8
|
+
"url": "https://handbook.doist.com/doc/standard-secrets-management-"
|
|
9
9
|
},
|
|
10
10
|
"ai-internal-tools.md": {
|
|
11
|
-
"id": "
|
|
12
|
-
"url": "https://handbook.doist.com/doc/standard-internal-ai-tools-
|
|
11
|
+
"id": "",
|
|
12
|
+
"url": "https://handbook.doist.com/doc/standard-internal-ai-tools-"
|
|
13
13
|
},
|
|
14
14
|
"service-production-readiness.md": {
|
|
15
|
-
"id": "
|
|
16
|
-
"url": "https://handbook.doist.com/doc/standard-service-production-readiness-
|
|
15
|
+
"id": "",
|
|
16
|
+
"url": "https://handbook.doist.com/doc/standard-service-production-readiness-"
|
|
17
17
|
},
|
|
18
18
|
"observability-standard.md": {
|
|
19
|
-
"id": "
|
|
20
|
-
"url": "https://handbook.doist.com/doc/standard-observability-
|
|
19
|
+
"id": "",
|
|
20
|
+
"url": "https://handbook.doist.com/doc/standard-observability-"
|
|
21
|
+
},
|
|
22
|
+
"backend-repository-standards.md": {
|
|
23
|
+
"id": "",
|
|
24
|
+
"url": "https://handbook.doist.com/doc/standard-backend-repositories-"
|
|
21
25
|
}
|
|
22
26
|
}
|
|
@@ -49,7 +49,7 @@ All services must produce structured, meaningful logs.
|
|
|
49
49
|
|
|
50
50
|
Logs should support operational debugging and monitoring.
|
|
51
51
|
|
|
52
|
-
See the Observability standard for more details.
|
|
52
|
+
See [the Observability standard for more details](https://handbook.doist.com/doc/standard-observability-vWRaOfitho).
|
|
53
53
|
|
|
54
54
|
## Configuration Management
|
|
55
55
|
|
|
@@ -59,7 +59,7 @@ Configuration must be provided via environment variables.
|
|
|
59
59
|
- Env files are allowed only if variables can be overridden by the environment.
|
|
60
60
|
- Secrets must come from environment variables or read directly from a secrets management service (AWS Secrets Manager).
|
|
61
61
|
|
|
62
|
-
See the Secrets Management standard for more details.
|
|
62
|
+
See [the Secrets Management standard for more details](https://handbook.doist.com/doc/standard-secrets-management-wbKmIfrtgr).
|
|
63
63
|
|
|
64
64
|
## State Persistence
|
|
65
65
|
|
|
@@ -1,344 +0,0 @@
|
|
|
1
|
-
# Platform Standard: Which Cloud & Where to Deploy
|
|
2
|
-
|
|
3
|
-
| π Document Type | Platform Standard |
|
|
4
|
-
| :--------------- | :----------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
5
|
-
| π― Status | Approved |
|
|
6
|
-
| π€ Owner | Luciano Facchinelli |
|
|
7
|
-
| π
Last Reviewed | 2026-03-06 |
|
|
8
|
-
| ποΈ References | [AWS Well-Architected](https://aws.amazon.com/architecture/well-architected/), [GCP Architecture Framework](https://cloud.google.com/architecture/framework) |
|
|
9
|
-
| π Version | 1.0 |
|
|
10
|
-
|
|
11
|
-
# Workload types
|
|
12
|
-
|
|
13
|
-
## Remote State Services
|
|
14
|
-
|
|
15
|
-
- **Definition**: Services that either store state in external systems (databases, caches, object storage) and don't require persistent local disk or doesn't have a state at all.
|
|
16
|
-
Containers are disposable and can be recreated without data loss.
|
|
17
|
-
- **Typical use cases**:
|
|
18
|
-
- REST APIs
|
|
19
|
-
- Web applications
|
|
20
|
-
- Backend microservices
|
|
21
|
-
|
|
22
|
-
## No State / Local State Services
|
|
23
|
-
|
|
24
|
-
- **Definition**: Services that require persistent local disk storage and expect data to survive container restarts. These need node affinity and persistent volumes.
|
|
25
|
-
- **Typical use cases**:
|
|
26
|
-
- Analytics database (i.e ClickHouse)
|
|
27
|
-
- Observability tool (i.e Langfuse)
|
|
28
|
-
- Self-hosted databases
|
|
29
|
-
- Message brokers with persistence
|
|
30
|
-
- Runs a database or database-like system
|
|
31
|
-
- Benefits from node affinity control for performance
|
|
32
|
-
|
|
33
|
-
## Event-Driven / Short-Lived Tasks
|
|
34
|
-
|
|
35
|
-
- **Definition**: Sporadic, short-duration tasks triggered by events that must complete within 15 minutes
|
|
36
|
-
- **Typical use cases**:
|
|
37
|
-
- Webhooks
|
|
38
|
-
- Scheduled jobs
|
|
39
|
-
- Queue processors
|
|
40
|
-
- Glue code between services
|
|
41
|
-
- File processing triggers
|
|
42
|
-
- Code must be small and optimized to reduce consumption from a resource perspective.
|
|
43
|
-
- Cold starts can impact latency for infrequent calls
|
|
44
|
-
- Not suitable for long-running or high-memory processes
|
|
45
|
-
|
|
46
|
-
## Preview / Ephemeral Environments
|
|
47
|
-
|
|
48
|
-
- **Definition**: Temporary, short-lived environments designed for testing, demos, and validation with a clear expiration
|
|
49
|
-
- **Typical use cases**:
|
|
50
|
-
- PR preview deployments
|
|
51
|
-
- Demo environments
|
|
52
|
-
- Temporary test instances
|
|
53
|
-
- Feature branch testing
|
|
54
|
-
- Excellent developer experience for fast iterations
|
|
55
|
-
|
|
56
|
-
## Internal Tools (Low Traffic)
|
|
57
|
-
|
|
58
|
-
- **Definition**: Internal-facing utilities and dashboards with low, predictable usage patterns accessed only by team members
|
|
59
|
-
- **Typical use cases**:
|
|
60
|
-
- Admin dashboards
|
|
61
|
-
- Internal utilities
|
|
62
|
-
- Analytics dashboards
|
|
63
|
-
- Configuration management UIs
|
|
64
|
-
- Reporting tools
|
|
65
|
-
- Low and predictable traffic patterns
|
|
66
|
-
- Users are internal team members only
|
|
67
|
-
- May or may not require access to production resources
|
|
68
|
-
- Simpler reliability requirements than user-facing services
|
|
69
|
-
|
|
70
|
-
## Batch Processing / ML Training
|
|
71
|
-
|
|
72
|
-
- **Definition**: Large-scale data processing or machine learning workloads that run for extended periods, often on a schedule or ad-hoc basis
|
|
73
|
-
- **Typical use cases**:
|
|
74
|
-
- Large batch data jobs
|
|
75
|
-
- ML model training
|
|
76
|
-
- Data pipeline processing
|
|
77
|
-
- One-off data migrations
|
|
78
|
-
- ETL jobs
|
|
79
|
-
- Variable duration β can run for hours or days
|
|
80
|
-
- Often requires burst compute capacity
|
|
81
|
-
- May need GPU resources for ML workloads
|
|
82
|
-
- Clear distinction between experimentation and production
|
|
83
|
-
- **Note: Requires case-by-case evaluation based on requirements**
|
|
84
|
-
|
|
85
|
-
# 1\. Purpose
|
|
86
|
-
|
|
87
|
-
"Where should I deploy this?" and "Can I use GCP for this?" shouldn't require a Platform team consultation. This standard provides clear defaults so teams can move fast while maintaining operational consistency.
|
|
88
|
-
|
|
89
|
-
These two decisions are tightly coupledβyour cloud choice often constrains your compute options, and vice versa. We're treating them as one standard.
|
|
90
|
-
|
|
91
|
-
## 2\. Scope
|
|
92
|
-
|
|
93
|
-
This standard applies to:
|
|
94
|
-
|
|
95
|
-
- All new services, tools, and workloads
|
|
96
|
-
- Migrations of existing services to new compute platforms
|
|
97
|
-
- Any infrastructure that runs production or production-adjacent workloads
|
|
98
|
-
|
|
99
|
-
This does **not** prescribe:
|
|
100
|
-
|
|
101
|
-
- How to configure specific services (separate runbooks)
|
|
102
|
-
- Database or storage choices (separate standard)
|
|
103
|
-
|
|
104
|
-
---
|
|
105
|
-
|
|
106
|
-
# Part A: Which Cloud
|
|
107
|
-
|
|
108
|
-
## 3A. Standard
|
|
109
|
-
|
|
110
|
-
### Default: AWS
|
|
111
|
-
|
|
112
|
-
**All new production workloads MUST be deployed to AWS unless there's a documented exception.**
|
|
113
|
-
|
|
114
|
-
AWS is our primary cloud. This means:
|
|
115
|
-
|
|
116
|
-
- All production databases, queues, and storage
|
|
117
|
-
- All user-facing services
|
|
118
|
-
- All internal services that modifies production data
|
|
119
|
-
|
|
120
|
-
### When GCP is Acceptable
|
|
121
|
-
|
|
122
|
-
GCP may be used for:
|
|
123
|
-
|
|
124
|
-
| Use Case | Rationale | Examples |
|
|
125
|
-
| :-------------------------- | :----------------------------------------------- | :------------------------------------------ |
|
|
126
|
-
| **BigQuery analytics** | Best-in-class data warehouse, existing pipelines | Data team experiments, analytics dashboards |
|
|
127
|
-
| **ML/AI experimentation** | Vertex AI, Colab integration | Prototype ML models before productionizing |
|
|
128
|
-
| **One-off data processing** | Cost-effective for burst compute | Large batch jobs with clear end dates |
|
|
129
|
-
| **Cloud Run previews** | Fast, cheap preview environments | PR preview deployments for web apps |
|
|
130
|
-
|
|
131
|
-
GCP is **NOT acceptable** for:
|
|
132
|
-
|
|
133
|
-
- Services that need to access production databases directly
|
|
134
|
-
- Anything requiring low-latency connections to AWS resources
|
|
135
|
-
- User-facing production services
|
|
136
|
-
- Long-running local state services
|
|
137
|
-
|
|
138
|
-
### Multi-Cloud \= Complexity Tax
|
|
139
|
-
|
|
140
|
-
Every service in GCP means:
|
|
141
|
-
|
|
142
|
-
- Separate IAM, networking, and monitoring
|
|
143
|
-
- Cross-cloud latency and data transfer costs
|
|
144
|
-
- Split operational knowledge on the team
|
|
145
|
-
- Harder incident response
|
|
146
|
-
|
|
147
|
-
The bar for GCP should be: "AWS genuinely can't do this well" β not "GCP has a nicer UI for this."
|
|
148
|
-
|
|
149
|
-
## 4A. Rationale
|
|
150
|
-
|
|
151
|
-
**Why default to AWS?**
|
|
152
|
-
|
|
153
|
-
- Operational consistency: one set of tools, dashboards, runbooks
|
|
154
|
-
- Network topology: VPCs, peering, and security groups are already configured
|
|
155
|
-
- Cost visibility: unified billing, tagging, and attribution
|
|
156
|
-
- Team knowledge: everyone knows AWS; GCP expertise is spotty
|
|
157
|
-
|
|
158
|
-
**Why allow GCP at all?**
|
|
159
|
-
|
|
160
|
-
- BigQuery is genuinely better for our analytics use cases
|
|
161
|
-
- Cloud Run is excellent for preview environments (fast, cheap, ephemeral)
|
|
162
|
-
- Forcing everything into AWS would be dogmatic, not practical
|
|
163
|
-
|
|
164
|
-
---
|
|
165
|
-
|
|
166
|
-
# Part B: Where to Deploy (Compute)
|
|
167
|
-
|
|
168
|
-
## 3B. Standard
|
|
169
|
-
|
|
170
|
-
### Decision Tree
|
|
171
|
-
|
|
172
|
-
```mermaid
|
|
173
|
-
%%{init: {'flowchart': {'curve': 'linear'}}}%%
|
|
174
|
-
flowchart TB
|
|
175
|
-
subgraph PartA["Part A: Cloud Selection"]
|
|
176
|
-
AWS[/"β
AWS<br>(Primary Cloud)"/]
|
|
177
|
-
Q1{"Which Cloud?"}
|
|
178
|
-
GCP_Check{"GCP Use Case?"}
|
|
179
|
-
GCP[/"βοΈ GCP Acceptable"/]
|
|
180
|
-
end
|
|
181
|
-
subgraph GCP_No["π« GCP NOT Acceptable For"]
|
|
182
|
-
direction TB
|
|
183
|
-
N1["Production DB Access"]
|
|
184
|
-
N2["Low-latency to AWS"]
|
|
185
|
-
N3["User-facing Production"]
|
|
186
|
-
N4["Long-running local state"]
|
|
187
|
-
end
|
|
188
|
-
subgraph PartB["Part B: Compute Platform Selection"]
|
|
189
|
-
ECS[/"π¦ ECS Fargate<br>(Default)"/]
|
|
190
|
-
Q2{"What Type<br>of Workload?"}
|
|
191
|
-
EKS[/"βΈοΈ EKS Kubernetes"/]
|
|
192
|
-
Lambda[/"β‘ Lambda"/]
|
|
193
|
-
Platform[/"π€ Consult Platform Team"/]
|
|
194
|
-
CloudRun[/"π Cloud Run"/]
|
|
195
|
-
Q3{"GCP Compute?"}
|
|
196
|
-
end
|
|
197
|
-
subgraph Examples["π Examples"]
|
|
198
|
-
direction TB
|
|
199
|
-
Ex1["REST API β ECS Fargate"]
|
|
200
|
-
Ex2["ClickHouse β EKS"]
|
|
201
|
-
Ex3["Langfuse β EKS"]
|
|
202
|
-
Ex4["Webhooks β Lambda"]
|
|
203
|
-
Ex5["PR Previews β Cloud Run"]
|
|
204
|
-
end
|
|
205
|
-
subgraph Exceptions["β οΈ Exception Process"]
|
|
206
|
-
direction TB
|
|
207
|
-
E1["1. Document rationale"]
|
|
208
|
-
E2["2. Post in Platform channel"]
|
|
209
|
-
E3["3. Get Platform approval"]
|
|
210
|
-
end
|
|
211
|
-
Start(["π New Service/Workload"]) --> Q1
|
|
212
|
-
Q1 -- Default --> AWS
|
|
213
|
-
Q1 -- Exception Cases --> GCP_Check
|
|
214
|
-
GCP_Check -- BigQuery Analytics --> GCP
|
|
215
|
-
GCP_Check -- ML/AI Experimentation --> GCP
|
|
216
|
-
GCP_Check -- "One-off Data Processing" --> GCP
|
|
217
|
-
GCP_Check -- Preview Environments --> GCP
|
|
218
|
-
GCP_Check -- None of above --> AWS
|
|
219
|
-
GCP -. Check Restrictions .-> GCP_No
|
|
220
|
-
AWS --> Q2
|
|
221
|
-
GCP --> Q3
|
|
222
|
-
Q2 -- remote state<br>(API, Web App) --> ECS
|
|
223
|
-
Q2 -- local state<br>(Persistent Storage) --> EKS
|
|
224
|
-
Q2 -- "Event-driven<br>(under 15 min)" --> Lambda
|
|
225
|
-
Q2 -- Batch/ML Training --> Platform
|
|
226
|
-
Q3 -- Previews/Internal Tools --> CloudRun
|
|
227
|
-
ECS -.-> Ex1
|
|
228
|
-
EKS -.-> Ex2 & Ex3
|
|
229
|
-
Lambda -.-> Ex4
|
|
230
|
-
CloudRun -.-> Ex5
|
|
231
|
-
E1 --> E2
|
|
232
|
-
E2 --> E3
|
|
233
|
-
Q2 -- Doesn't fit? --> Exceptions
|
|
234
|
-
|
|
235
|
-
AWS:::aws
|
|
236
|
-
Q1:::decision
|
|
237
|
-
GCP_Check:::decision
|
|
238
|
-
GCP:::gcp
|
|
239
|
-
ECS:::aws
|
|
240
|
-
Q2:::decision
|
|
241
|
-
EKS:::aws
|
|
242
|
-
Lambda:::aws
|
|
243
|
-
CloudRun:::gcp
|
|
244
|
-
Q3:::decision
|
|
245
|
-
classDef aws fill:#FF9900,stroke:#232F3E,color:#232F3E
|
|
246
|
-
classDef gcp fill:#4285F4,stroke:#174EA6,color:white
|
|
247
|
-
classDef decision fill:#f9f,stroke:#333,stroke-width:2px
|
|
248
|
-
classDef compute fill:#90EE90,stroke:#228B22
|
|
249
|
-
```
|
|
250
|
-
|
|
251
|
-
### Compute Platform Characteristics
|
|
252
|
-
|
|
253
|
-
| Platform | Best For | Avoid When | Notes |
|
|
254
|
-
| :------------------- | :----------------------------------------- | :------------------------------------------- | :----------------------------------------------------- |
|
|
255
|
-
| **ECS Fargate** | remote state services, APIs, web apps | Need persistent local storage, need GPU | Our default. Good tooling, well-understood. |
|
|
256
|
-
| **EKS (Kubernetes)** | local state workloads, complex deployments | Simple local state services (overkill) | Higher operational overhead. Worth it for local state. |
|
|
257
|
-
| **Lambda** | Event-driven, short tasks, glue code | Long-running processes, high-cpu needs | Cold starts matter. Keep functions small. |
|
|
258
|
-
| **Cloud Run** | Previews, internal tools, experiments | Production services, AWS-dependent workloads | Great DX, but it's GCP (see Part A). |
|
|
259
|
-
|
|
260
|
-
## 4B. Rationale
|
|
261
|
-
|
|
262
|
-
**Why ECS as default?**
|
|
263
|
-
|
|
264
|
-
- Simpler than Kubernetes for most use cases
|
|
265
|
-
- Fargate \= no EC2 instance management
|
|
266
|
-
- Good integration with AWS services (ALB, CloudWatch, Secrets Manager)
|
|
267
|
-
- Lower cognitive overhead for teams
|
|
268
|
-
|
|
269
|
-
**Why Kubernetes for local state?**
|
|
270
|
-
|
|
271
|
-
- local stateSets, persistent volumes, and operators
|
|
272
|
-
- Better control over scheduling and node affinity
|
|
273
|
-
- Community ecosystem for databases and local state apps
|
|
274
|
-
- We're investing in EKS anyway; leverage it where it shines
|
|
275
|
-
|
|
276
|
-
**Why Lambda for event-driven?**
|
|
277
|
-
|
|
278
|
-
- Pay-per-invocation is cost-effective for sporadic workloads
|
|
279
|
-
- Built-in scaling, no capacity planning
|
|
280
|
-
- Native integration with SQS, SNS, EventBridge
|
|
281
|
-
- Forces good practices (small, focused functions)
|
|
282
|
-
|
|
283
|
-
**Why Cloud Run for previews?**
|
|
284
|
-
|
|
285
|
-
- Deploys in seconds, scales to zero
|
|
286
|
-
- Cheap for low-traffic ephemeral environments
|
|
287
|
-
- Great developer experience
|
|
288
|
-
- Isolated from production (GCP \= natural boundary)
|
|
289
|
-
|
|
290
|
-
---
|
|
291
|
-
|
|
292
|
-
## 5\. How to Apply
|
|
293
|
-
|
|
294
|
-
### Example: New API Service
|
|
295
|
-
|
|
296
|
-
You're building a new internal API that serves data to the mobile apps.
|
|
297
|
-
|
|
298
|
-
1. **Which cloud?** β AWS (it's production, needs DB access)
|
|
299
|
-
2. **What compute?** β ECS Fargate (local state API)
|
|
300
|
-
3. **How to deploy?** β CloudFormation stack, GitHub Actions workflow
|
|
301
|
-
|
|
302
|
-
### Example: Analytics Dashboard
|
|
303
|
-
|
|
304
|
-
You're building an internal dashboard that queries BigQuery and displays charts.
|
|
305
|
-
|
|
306
|
-
1. **Which cloud?** β GCP (BigQuery-native, no prod DB access)
|
|
307
|
-
2. **What compute?** β Cloud Run (internal tool, low traffic)
|
|
308
|
-
3. **How to deploy?** β Cloud Run from container registry
|
|
309
|
-
|
|
310
|
-
### Example: New Observability Tool (like Langfuse)
|
|
311
|
-
|
|
312
|
-
You're deploying a self-hosted observability tool that needs persistent storage.
|
|
313
|
-
|
|
314
|
-
1. **Which cloud?** β AWS (needs to ingest data from AWS services)
|
|
315
|
-
2. **What compute?** β EKS (local state, needs persistent volumes)
|
|
316
|
-
3. **How to deploy?** β Helm chart, ArgoCD
|
|
317
|
-
|
|
318
|
-
## 6\. Exceptions
|
|
319
|
-
|
|
320
|
-
### Requesting an Exception
|
|
321
|
-
|
|
322
|
-
If your use case doesn't fit the decision tree:
|
|
323
|
-
|
|
324
|
-
1. Document: What you're building, why the default doesn't work
|
|
325
|
-
2. Post in Platform channel
|
|
326
|
-
3. Get sign-off before proceeding
|
|
327
|
-
|
|
328
|
-
Common valid exceptions:
|
|
329
|
-
|
|
330
|
-
- Vendor-specific requirements (e.g., "only runs on GCP")
|
|
331
|
-
- Cost optimization for specific workload patterns
|
|
332
|
-
- Experimentation with explicit time bounds
|
|
333
|
-
|
|
334
|
-
### Grandfathered Services
|
|
335
|
-
|
|
336
|
-
We should apply the [Gartnerβs TIME framework](https://unstoppablesoftware.com/understanding-gartners-time-model-maximizing-business-value-with-software-portfolio-management/#:~:text=1%2E%20Tolerate%20%28Low%20Value%2C%20Low%20Cost%2FRisk) when deciding. After that, we may realize some existing services don't follow this standard. We're not migrating them unless there's a compelling reason. Document them, but don't use them as precedent for new services.
|
|
337
|
-
|
|
338
|
-
## 7\. References
|
|
339
|
-
|
|
340
|
-
- [ECS Service Setup Runbook](https://file+.vscode-resource.vscode-cdn.net/Users/arod/work/doist/twist-threads/work/20260114-platform-standards-analysis/TBD)
|
|
341
|
-
- [EKS Deployment Guide](https://file+.vscode-resource.vscode-cdn.net/Users/arod/work/doist/twist-threads/work/20260114-platform-standards-analysis/TBD)
|
|
342
|
-
- [Lambda Best Practices](https://file+.vscode-resource.vscode-cdn.net/Users/arod/work/doist/twist-threads/work/20260114-platform-standards-analysis/TBD)
|
|
343
|
-
- [Cloud Run Setup for Previews](https://file+.vscode-resource.vscode-cdn.net/Users/arod/work/doist/twist-threads/work/20260114-platform-standards-analysis/TBD)
|
|
344
|
-
- \[Cost Tagging Standard\](TBD \- Group 3\)
|