msgctl 0.1.0a1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- msg/__init__.py +2 -0
- msg/admin/__init__.py +0 -0
- msg/admin/backup_retirement.py +56 -0
- msg/admin/backups.py +341 -0
- msg/admin/custodial_check.py +77 -0
- msg/admin/diagnostics.py +750 -0
- msg/admin/market.py +91 -0
- msg/admin/market_check.py +341 -0
- msg/admin/money.py +322 -0
- msg/admin/preflight.py +142 -0
- msg/admin/recovery_replay.py +281 -0
- msg/admin/restore_database.py +37 -0
- msg/admin/root.py +347 -0
- msg/admin/rotation.py +197 -0
- msg/admin/token_delivery_check.py +54 -0
- msg/admin/upgrade_check.py +58 -0
- msg/application.py +163 -0
- msg/bootstrap.py +305 -0
- msg/cli.py +537 -0
- msg/client.py +725 -0
- msg/client_certificates.py +98 -0
- msg/client_content.py +43 -0
- msg/client_custodial.py +206 -0
- msg/client_market.py +104 -0
- msg/client_recovery.py +186 -0
- msg/client_secrets.py +58 -0
- msg/client_tokens.py +99 -0
- msg/client_upgrade.py +188 -0
- msg/config.py +318 -0
- msg/constants.py +11 -0
- msg/core/__init__.py +0 -0
- msg/core/batching.py +23 -0
- msg/core/codec.py +253 -0
- msg/core/contracts.py +128 -0
- msg/core/cursors.py +68 -0
- msg/core/email_address.py +33 -0
- msg/core/errors.py +22 -0
- msg/core/events.py +6 -0
- msg/core/execution_ports.py +20 -0
- msg/core/executor.py +193 -0
- msg/core/models.py +517 -0
- msg/core/packet.py +71 -0
- msg/core/permissions.py +14 -0
- msg/core/query.py +44 -0
- msg/core/read_query.py +60 -0
- msg/core/registry.py +173 -0
- msg/core/requests.py +58 -0
- msg/core/schema_policy.py +28 -0
- msg/core/schemas.py +30 -0
- msg/core/tags.py +21 -0
- msg/core/template_dsl.py +94 -0
- msg/core/text_patch.py +283 -0
- msg/core/tool_execution.py +28 -0
- msg/core/transfer.py +183 -0
- msg/daemon.py +255 -0
- msg/data/__init__.py +0 -0
- msg/data/bootstrap.json +72 -0
- msg/data/favicon.png +0 -0
- msg/data/logo-dark.svg +5 -0
- msg/data/logo.svg +5 -0
- msg/data/recovery-checkpoint.example.json +1 -0
- msg/data/recovery-checkpoint.schema.json +118 -0
- msg/data/shortcodes.json +1 -0
- msg/data/system/AGENTS.md +6 -0
- msg/data/system/rules/_index.md +17 -0
- msg/data/system/rules/auth.md +6 -0
- msg/data/system/rules/files.md +6 -0
- msg/data/system/rules/identity.md +6 -0
- msg/data/system/rules/protocol.md +6 -0
- msg/data/system/rules/read-write.md +6 -0
- msg/data/system/rules/recovery.md +6 -0
- msg/data/system/rules/security.md +10 -0
- msg/data/system/rules/topics.md +6 -0
- msg/extensions/__init__.py +1 -0
- msg/extensions/hosting.py +259 -0
- msg/extensions/keystore.py +96 -0
- msg/extensions/repositories.py +769 -0
- msg/extensions/rss.py +67 -0
- msg/extensions/ssh.py +264 -0
- msg/extensions/ssh_git.py +300 -0
- msg/extensions/tools.py +150 -0
- msg/hosting_runtime.py +138 -0
- msg/market/__init__.py +1 -0
- msg/market/arbitration.py +481 -0
- msg/market/delivery.py +338 -0
- msg/market/delivery_notifications.py +110 -0
- msg/market/delivery_targets.py +76 -0
- msg/market/email.py +100 -0
- msg/market/escrow.py +345 -0
- msg/market/orders.py +218 -0
- msg/market/policy.py +147 -0
- msg/market/rationale.py +91 -0
- msg/market/references.py +19 -0
- msg/market/targets.py +157 -0
- msg/plugins/__init__.py +28 -0
- msg/plugins/achievements.py +316 -0
- msg/plugins/batch.py +32 -0
- msg/plugins/bounty.py +366 -0
- msg/plugins/collaboration.py +261 -0
- msg/plugins/common.py +217 -0
- msg/plugins/communication.py +853 -0
- msg/plugins/content.py +855 -0
- msg/plugins/custodial_lifecycle.py +224 -0
- msg/plugins/delivery.py +230 -0
- msg/plugins/discovery.py +1266 -0
- msg/plugins/discussion.py +162 -0
- msg/plugins/extensions.py +10 -0
- msg/plugins/following.py +78 -0
- msg/plugins/hosting_capacity.py +52 -0
- msg/plugins/identity.py +1654 -0
- msg/plugins/money.py +221 -0
- msg/plugins/offers.py +183 -0
- msg/plugins/orders.py +273 -0
- msg/plugins/recovery.py +376 -0
- msg/plugins/schemas.py +5 -0
- msg/plugins/sharing.py +300 -0
- msg/plugins/store.py +249 -0
- msg/plugins/system.py +75 -0
- msg/plugins/transfer.py +249 -0
- msg/py.typed +0 -0
- msg/security/__init__.py +0 -0
- msg/security/age_keys.py +112 -0
- msg/security/authentication.py +141 -0
- msg/security/authorization.py +300 -0
- msg/security/backup_retirement.py +135 -0
- msg/security/capabilities.py +157 -0
- msg/security/certificates.py +183 -0
- msg/security/crypto.py +102 -0
- msg/security/custodial_migration.py +292 -0
- msg/security/custody_history.py +146 -0
- msg/security/network.py +81 -0
- msg/security/policy.py +67 -0
- msg/security/quarantine.py +19 -0
- msg/security/root_files.py +67 -0
- msg/security/rotation_journal.py +60 -0
- msg/security/sealed_box.py +70 -0
- msg/security/sharing_policy.py +29 -0
- msg/security/token_delivery.py +93 -0
- msg/security/vault.py +186 -0
- msg/storage/__init__.py +0 -0
- msg/storage/capacity.py +100 -0
- msg/storage/custodial_migration.py +22 -0
- msg/storage/git.py +435 -0
- msg/storage/ledger_migration.py +307 -0
- msg/storage/market_migration.py +49 -0
- msg/storage/postgres.py +663 -0
- msg/storage/query.py +39 -0
- msg/storage/read_only.py +126 -0
- msg/storage/session.py +403 -0
- msg/storage/sqlite.py +360 -0
- msg/storage/topic_event_migration.py +26 -0
- msg/storage/valkey_bus.py +45 -0
- msg/transports/__init__.py +0 -0
- msg/transports/client.py +183 -0
- msg/transports/dictionary.py +387 -0
- msg/transports/graphql.py +75 -0
- msg/transports/http.py +72 -0
- msg/transports/http_routes.py +1577 -0
- msg/transports/mcp.py +127 -0
- msg/transports/packet.py +42 -0
- msg/transports/read_tree_path.py +65 -0
- msg/transports/stdio.py +36 -0
- msg/transports/url_safety.py +123 -0
- msg/tui.py +368 -0
- msg/workers/__init__.py +1 -0
- msg/workers/effects.py +406 -0
- msg/workers/leases.py +19 -0
- msg/workers/mail.py +63 -0
- msg/workers/maintenance.py +310 -0
- msg/workers/sandbox.py +74 -0
- msg/workers/sandbox_child.py +189 -0
- msg/workers/webhook.py +160 -0
- msgctl-0.1.0a1.dist-info/METADATA +96 -0
- msgctl-0.1.0a1.dist-info/RECORD +176 -0
- msgctl-0.1.0a1.dist-info/WHEEL +4 -0
- msgctl-0.1.0a1.dist-info/entry_points.txt +4 -0
msg/plugins/discovery.py
ADDED
|
@@ -0,0 +1,1266 @@
|
|
|
1
|
+
"""ACL-filtered reads and rebuildable discovery projections."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
import difflib
|
|
4
|
+
import fnmatch
|
|
5
|
+
import re
|
|
6
|
+
import time
|
|
7
|
+
from dataclasses import replace
|
|
8
|
+
from datetime import timedelta
|
|
9
|
+
from msg.constants import *
|
|
10
|
+
from msg.core.codec import canonical,wire,decode,loads,digest,b64
|
|
11
|
+
from msg.core.errors import Failure,require
|
|
12
|
+
from msg.core.models import Resource,ResourceRef,Revision,HandlerOutput,Credential
|
|
13
|
+
from msg.core.tags import normalize_tag
|
|
14
|
+
from msg.core.read_query import (ReadBudget,MAX_READ_DEPTH,NESTED_FIELDS,ROOT_FIELDS,
|
|
15
|
+
expansion_schema,read_query_version)
|
|
16
|
+
from msg.core.requests import request_for
|
|
17
|
+
from msg.core.template_dsl import render_values
|
|
18
|
+
from msg.plugins.common import *
|
|
19
|
+
from msg.plugins.schemas import *
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
async def visible(app,ctx,request,tx,rid):
|
|
23
|
+
try:
|
|
24
|
+
await check_access(app,ctx,request,tx,rid,'read')
|
|
25
|
+
return True
|
|
26
|
+
except Failure as exc:
|
|
27
|
+
if exc.code in {'permission_denied','credential_ceiling','certificate_gate','tool_certificate_required',
|
|
28
|
+
'delegation_scope','ancestor_inactive'}:
|
|
29
|
+
return False
|
|
30
|
+
raise
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
async def metadata(tx,r):
|
|
34
|
+
data=wire(r)
|
|
35
|
+
data['path']=short_subject_path(await tx.path(r.id))
|
|
36
|
+
if r.revision:
|
|
37
|
+
rev=await tx.revision(ResourceRef(id=r.id))
|
|
38
|
+
data.update(size=rev.content.size,media_type=rev.content.media_type,digest=rev.content.digest)
|
|
39
|
+
for field in ('change_note','source_kind','source_version','source_digest'):
|
|
40
|
+
value=getattr(rev,field)
|
|
41
|
+
if value is not None:
|
|
42
|
+
data[field]=value
|
|
43
|
+
return data
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def short_subject_path(path):
|
|
47
|
+
parts=path.split('/')
|
|
48
|
+
if len(parts)>=3 and parts[1].startswith('@'):
|
|
49
|
+
parts[2]={'keys':'k','certificates':'cert','keystore':'ks'}.get(parts[2],parts[2])
|
|
50
|
+
return '/'.join(parts)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
LINK_RELATIONS=frozenset({'self','t','a','r','p','c','f','q','b','h','v','d'})
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
async def visible_link(app,ctx,request,tx,ref):
|
|
57
|
+
if not await visible(app,ctx,request,tx,ref.id):
|
|
58
|
+
return None
|
|
59
|
+
target=await tx.resource(ref.id)
|
|
60
|
+
if target.state=='purged':
|
|
61
|
+
return None
|
|
62
|
+
if ref.revision is not None:
|
|
63
|
+
try:
|
|
64
|
+
await tx.revision(ref)
|
|
65
|
+
except Failure as exc:
|
|
66
|
+
if exc.code=='revision_not_found':
|
|
67
|
+
return None
|
|
68
|
+
raise
|
|
69
|
+
return {'ref':wire(ref),'path':short_subject_path(await tx.path(ref.id))}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
async def basic_links(app,ctx,request,tx,resource,revision):
|
|
73
|
+
rid=resource.id
|
|
74
|
+
current=ResourceRef(id=rid,revision=revision.id if revision is not None else resource.revision)
|
|
75
|
+
singles={'self':{'ref':wire(current),'path':short_subject_path(await tx.path(rid))}}
|
|
76
|
+
if resource.type in {'post','attachment','file'} and resource.parent:
|
|
77
|
+
parent=await tx.resource(resource.parent)
|
|
78
|
+
if parent.type=='topic':
|
|
79
|
+
link=await visible_link(app,ctx,request,tx,ResourceRef(id=parent.id))
|
|
80
|
+
if link:singles['t']=link
|
|
81
|
+
if revision is not None:
|
|
82
|
+
author=await visible_link(app,ctx,request,tx,ResourceRef(id=revision.author))
|
|
83
|
+
if author:singles['a']=author
|
|
84
|
+
singles['v']={'ref':wire(current),'path':f'/_r/{rid}/rev/{revision.id}'}
|
|
85
|
+
if len(revision.parents)==1:
|
|
86
|
+
singles['d']={'from':wire(ResourceRef(id=rid,revision=revision.parents[0])),
|
|
87
|
+
'to':wire(current),'path':f'/_r/{rid}/l/d'}
|
|
88
|
+
outgoing={relation.type:relation.target for relation in revision.relations
|
|
89
|
+
if relation.type in {'reply_to','thread_root'}}
|
|
90
|
+
if resource.type=='post':
|
|
91
|
+
root=outgoing.get('thread_root',current)
|
|
92
|
+
link=await visible_link(app,ctx,request,tx,root)
|
|
93
|
+
if link:singles['r']=link
|
|
94
|
+
if 'reply_to' in outgoing:
|
|
95
|
+
link=await visible_link(app,ctx,request,tx,outgoing['reply_to'])
|
|
96
|
+
if link:singles['p']=link
|
|
97
|
+
return singles
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
async def relation_page(app,ctx,request,tx,resource,revision,rel,limit,position,snapshot, *, budget=None):
|
|
101
|
+
if rel in {'c','b','f','q'} and position=='':
|
|
102
|
+
position=['','']
|
|
103
|
+
candidates=()
|
|
104
|
+
if rel=='h':
|
|
105
|
+
rows=tx.execute('''SELECT body FROM revisions WHERE resource_id=? AND id>?
|
|
106
|
+
AND created_at<=? ORDER BY id LIMIT 2049''',(resource.id,position,wire(snapshot)))
|
|
107
|
+
candidates=((item.id,ResourceRef(id=resource.id,revision=item.id))
|
|
108
|
+
for (raw,) in rows for item in (decode(Revision,loads(raw)),))
|
|
109
|
+
elif rel in {'c','b'}:
|
|
110
|
+
kinds=('reply_to',) if rel=='c' else ('quote','repost')
|
|
111
|
+
placeholders=','.join('?' for _ in kinds)
|
|
112
|
+
rows=tx.execute(f'''SELECT rel.source_id,rel.type,rel.revision_id
|
|
113
|
+
FROM relations rel JOIN resources r ON r.id=rel.source_id AND r.revision=rel.revision_id
|
|
114
|
+
WHERE rel.target_id=? AND rel.type IN ({placeholders}) AND r.state='active'
|
|
115
|
+
AND r.created_at<=? AND (rel.source_id,rel.type)>(?,?)
|
|
116
|
+
ORDER BY rel.source_id,rel.type LIMIT 2049''',
|
|
117
|
+
(resource.id,*kinds,wire(snapshot),*position))
|
|
118
|
+
candidates=(([source,kind],ResourceRef(id=source,revision=current_revision))
|
|
119
|
+
for source,kind,current_revision in rows)
|
|
120
|
+
elif rel in {'f','q'} and revision is not None:
|
|
121
|
+
kinds={'attachment'} if rel=='f' else {'quote','repost'}
|
|
122
|
+
require(len(revision.relations)<=2048,'query_cost_exceeded')
|
|
123
|
+
candidates=sorted(([relation.target.id,relation.type],relation.target)
|
|
124
|
+
for relation in revision.relations if relation.type in kinds)
|
|
125
|
+
items=[]
|
|
126
|
+
last=position
|
|
127
|
+
more=False
|
|
128
|
+
scanned=0
|
|
129
|
+
for key,ref in candidates:
|
|
130
|
+
if key<=position:
|
|
131
|
+
continue
|
|
132
|
+
scanned+=1
|
|
133
|
+
require(scanned<=2048 and time.monotonic()<ctx.deadline_monotonic,'query_cost_exceeded')
|
|
134
|
+
if budget is not None:budget.scan()
|
|
135
|
+
link=await visible_link(app,ctx,request,tx,ref)
|
|
136
|
+
if link is None:
|
|
137
|
+
continue
|
|
138
|
+
if rel=='h':
|
|
139
|
+
historical=await tx.revision(ref)
|
|
140
|
+
link.update(author=historical.author,created_at=wire(historical.created_at))
|
|
141
|
+
for field in ('change_note','source_kind','source_version'):
|
|
142
|
+
value=getattr(historical,field,None)
|
|
143
|
+
if value is not None:link[field]=wire(value)
|
|
144
|
+
if len(items)==limit:
|
|
145
|
+
more=True
|
|
146
|
+
break
|
|
147
|
+
if budget is not None:budget.node(len(NESTED_FIELDS))
|
|
148
|
+
items.append(link)
|
|
149
|
+
last=key
|
|
150
|
+
return items,last,more
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
async def filtered_tools(app,ctx,request,tx):
|
|
154
|
+
result=[]
|
|
155
|
+
page=await tx.children(TOOLS_SPACE,limit=500)
|
|
156
|
+
for resource in page.items:
|
|
157
|
+
if await app.authorizer.has(ctx.principal,'tool.use',operation_id(request),resource.id,tx):
|
|
158
|
+
result.append({'id':resource.id,'name':resource.name,'path':await tx.path(resource.id),'revision':resource.revision})
|
|
159
|
+
require(bool(result),'tool_certificate_required')
|
|
160
|
+
return result
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
async def read_projection(app,ctx,request,tx,rid, *, revision=None,fields=()):
|
|
164
|
+
resource=await tx.resource(rid)
|
|
165
|
+
if rid==TOOLS_SPACE:
|
|
166
|
+
return {'id':rid,'type':'topic','items':await filtered_tools(app,ctx,request,tx)}
|
|
167
|
+
await check_access(app,ctx,request,tx,rid,'read')
|
|
168
|
+
require(resource.state!='purged','resource_purged')
|
|
169
|
+
meta=await metadata(tx,resource)
|
|
170
|
+
if rid=='t_capabilities':
|
|
171
|
+
specs=[wire(s,compact=True) for s in app.registry.capabilities()]
|
|
172
|
+
return {'version':1,'digest':digest(specs),'capabilities':specs}
|
|
173
|
+
if rid=='t_operations':
|
|
174
|
+
return app.registry.catalog()
|
|
175
|
+
if resource.type=='certificate':
|
|
176
|
+
return {'metadata':meta,'certificate':wire(await tx.certificate(rid)),'revoked':await tx.certificate_revoked(rid)}
|
|
177
|
+
if resource.type=='csr':
|
|
178
|
+
return {'metadata':meta,'request':wire(await tx.csr(rid)),'state':wire(await tx.csr_state(rid))}
|
|
179
|
+
if resource.type=='user':
|
|
180
|
+
subject=await tx.subject(rid)
|
|
181
|
+
meta.update(kind=subject.kind,local_only=subject.local_only)
|
|
182
|
+
if resource.name=='keys' and resource.parent is not None and (await tx.resource(resource.parent)).type=='user':
|
|
183
|
+
keys=[]
|
|
184
|
+
for row in tx.rows('SELECT body FROM credentials WHERE subject=?',(resource.parent,)):
|
|
185
|
+
credential=decode(Credential,loads(row[0]))
|
|
186
|
+
if credential.kind!='token':
|
|
187
|
+
keys.append({'key_id':credential.id,'kind':credential.kind,'public_key':b64(credential.verifier),
|
|
188
|
+
'revoked':credential.revoked_at is not None})
|
|
189
|
+
return {'id':rid,'path':meta['path'],'keys':keys}
|
|
190
|
+
if resource.name=='certificates' and resource.parent is not None and (await tx.resource(resource.parent)).type=='user':
|
|
191
|
+
return {'id':rid,'path':meta['path'],'certificates':[{'id':row[0],'revoked':bool(row[1])} for row in
|
|
192
|
+
tx.rows('SELECT id,revoked FROM certificates WHERE subject=? ORDER BY id',(resource.parent,))]}
|
|
193
|
+
if app.registry.resource_type(resource.type,1).container and resource.type not in {'repo','website'}:
|
|
194
|
+
values=[]
|
|
195
|
+
cursor=None
|
|
196
|
+
while len(values)<50:
|
|
197
|
+
page=await tx.children(rid,cursor,50)
|
|
198
|
+
for child in page.items:
|
|
199
|
+
if child.state=='active' and await visible(app,ctx,request,tx,child.id):
|
|
200
|
+
values.append({'id':child.id,'name':child.name,'type':child.type,'revision':child.revision,
|
|
201
|
+
'path':short_subject_path(await tx.path(child.id))})
|
|
202
|
+
if len(values)==50:
|
|
203
|
+
break
|
|
204
|
+
cursor=page.next_cursor
|
|
205
|
+
if cursor is None:
|
|
206
|
+
break
|
|
207
|
+
meta['items']=values
|
|
208
|
+
meta['list_operation']=next_link(app,'discovery.list',{'parent':rid})
|
|
209
|
+
elif resource.revision:
|
|
210
|
+
rev=await tx.revision(ResourceRef(id=rid,revision=revision))
|
|
211
|
+
meta.update(revision=rev.id,digest=rev.content.digest,size=rev.content.size,media_type=rev.content.media_type)
|
|
212
|
+
textual=rev.content.media_type.startswith('text/') or rev.content.media_type in {'application/json','application/msg-template'}
|
|
213
|
+
if textual and rev.content.size<app.settings.server.limits.max_response_bytes//2:
|
|
214
|
+
raw=await app.contents.read_bytes(rev.content,limit=app.settings.server.limits.max_response_bytes)
|
|
215
|
+
meta['content']=loads(raw) if rev.content.media_type=='application/json' else raw.decode('utf-8')
|
|
216
|
+
else:
|
|
217
|
+
meta['raw_url']=f'/_id/{rid}/revisions/{rev.id}/raw'
|
|
218
|
+
meta['transfer_operation']='transfer.open'
|
|
219
|
+
meta['relations']=wire(rev.relations,compact=True)
|
|
220
|
+
if resource.type=='post' and (not fields or 'links' in fields):
|
|
221
|
+
active=await tx.revision(ResourceRef(id=rid,revision=revision))
|
|
222
|
+
meta['links']=await basic_links(app,ctx,request,tx,resource,active)
|
|
223
|
+
known=set(meta)
|
|
224
|
+
if fields:
|
|
225
|
+
require(set(fields)<=known,'unknown_projection_field')
|
|
226
|
+
return {k:meta[k] for k in fields}
|
|
227
|
+
defaults=('id','type','name','revision','generation','path','content','items','keys','certificates','links',
|
|
228
|
+
'relations','raw_url','transfer_operation','kind','local_only','list_operation',
|
|
229
|
+
'change_note','source_kind','source_version','source_digest')
|
|
230
|
+
if resource.tags:
|
|
231
|
+
defaults=(*defaults,'tags')
|
|
232
|
+
return {k:meta[k] for k in defaults if k in meta}
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
async def markdown_segment(app,blob,offset,max_bytes,fence=False,line_start=True):
|
|
236
|
+
"""Read a bounded UTF-8 window, preferring complete Markdown blocks."""
|
|
237
|
+
require(0<=offset<=blob.size,'invalid_byte_range')
|
|
238
|
+
end=min(blob.size,offset+max_bytes+4)
|
|
239
|
+
raw=b''.join([piece async for piece in app.contents.read(blob,(offset,end))])
|
|
240
|
+
decoded=None
|
|
241
|
+
for trim in range(4):
|
|
242
|
+
try:
|
|
243
|
+
decoded=raw[:len(raw)-trim if trim else len(raw)].decode('utf-8')
|
|
244
|
+
break
|
|
245
|
+
except UnicodeDecodeError as exc:
|
|
246
|
+
require(exc.start>=len(raw)-4,'invalid_utf8_content')
|
|
247
|
+
require(decoded is not None,'invalid_utf8_content')
|
|
248
|
+
used=0
|
|
249
|
+
count=0
|
|
250
|
+
for char in decoded:
|
|
251
|
+
length=len(char.encode('utf-8'))
|
|
252
|
+
if used+length>max_bytes:
|
|
253
|
+
break
|
|
254
|
+
used+=length
|
|
255
|
+
count+=1
|
|
256
|
+
budget=decoded[:count]
|
|
257
|
+
require(bool(budget) or offset==blob.size,'read_window_too_small')
|
|
258
|
+
candidate=None
|
|
259
|
+
scanned=0
|
|
260
|
+
in_fence=fence
|
|
261
|
+
at_line_start=line_start
|
|
262
|
+
state=(in_fence,at_line_start)
|
|
263
|
+
for line in budget.splitlines(keepends=True):
|
|
264
|
+
complete=line.endswith('\n')
|
|
265
|
+
stripped=line.strip()
|
|
266
|
+
if at_line_start and stripped.startswith(('```','~~~')):
|
|
267
|
+
in_fence=not in_fence
|
|
268
|
+
if not in_fence and complete:
|
|
269
|
+
candidate=(scanned+len(line),in_fence,True)
|
|
270
|
+
elif not in_fence and at_line_start and line.startswith('#') and scanned:
|
|
271
|
+
candidate=(scanned,in_fence,True)
|
|
272
|
+
elif not in_fence and not stripped and complete:
|
|
273
|
+
candidate=(scanned+len(line),in_fence,True)
|
|
274
|
+
scanned+=len(line)
|
|
275
|
+
at_line_start=complete
|
|
276
|
+
state=(in_fence,at_line_start)
|
|
277
|
+
if candidate is not None and candidate[0]>0:
|
|
278
|
+
char_end,fence_end,line_end=candidate
|
|
279
|
+
continued=False
|
|
280
|
+
else:
|
|
281
|
+
char_end=count
|
|
282
|
+
fence_end,line_end=state
|
|
283
|
+
continued=offset+len(budget.encode('utf-8'))<blob.size
|
|
284
|
+
text=budget[:char_end]
|
|
285
|
+
byte_end=offset+len(text.encode('utf-8'))
|
|
286
|
+
require(byte_end>offset or offset==blob.size,'read_window_too_small')
|
|
287
|
+
return byte_end,text,fence_end,line_end,continued
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
async def previous_segment(app,blob,target,max_bytes):
|
|
291
|
+
offset=0
|
|
292
|
+
fence=False
|
|
293
|
+
line_start=True
|
|
294
|
+
continued=False
|
|
295
|
+
previous=None
|
|
296
|
+
while offset<target:
|
|
297
|
+
previous=(offset,fence,line_start,continued)
|
|
298
|
+
end,_,fence,line_start,continued=await markdown_segment(
|
|
299
|
+
app,blob,offset,max_bytes,fence,line_start)
|
|
300
|
+
require(end<=target,'invalid_cursor')
|
|
301
|
+
offset=end
|
|
302
|
+
require(offset==target and previous is not None,'invalid_cursor')
|
|
303
|
+
return previous
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def next_link(app,operation,args):
|
|
307
|
+
packet=request_for(operation,args,app.settings.service_url,source='manual',
|
|
308
|
+
request_id='read_'+digest((operation,args))[7:39])
|
|
309
|
+
packet=replace(packet,expires_at=None)
|
|
310
|
+
return f'/-/g/{operation}/j/'+b64(canonical(packet))
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def install(app):
|
|
314
|
+
op,finish=registration(app,'discovery',('identity','content'))
|
|
315
|
+
fields={'type':'array','items':STRING,'maxItems':30,'uniqueItems':True}
|
|
316
|
+
|
|
317
|
+
@op('discovery.get',obj({'id':IDENTIFIER,'revision':IDENTIFIER,'fields':fields,
|
|
318
|
+
'view':{'enum':['json','meta','history']},'known_digest':STRING,'cursor':STRING,'limit':{'type':'integer','minimum':1,'maximum':200}},('id',)),effect='read')
|
|
319
|
+
async def get(ctx,request,tx):
|
|
320
|
+
a=request.arguments
|
|
321
|
+
rid=await resolve(tx,a['id'])
|
|
322
|
+
resource=await tx.resource(rid)
|
|
323
|
+
if a.get('view') in {'meta','history'}:
|
|
324
|
+
await check_access(app,ctx,request,tx,rid,'read')
|
|
325
|
+
if a['view']=='meta':
|
|
326
|
+
data=await metadata(tx,resource)
|
|
327
|
+
else:
|
|
328
|
+
limit=a.get('limit',50)
|
|
329
|
+
principal={'actor':ctx.principal.actor,'subject':ctx.principal.subject,
|
|
330
|
+
'credential_id':ctx.principal.credential_id}
|
|
331
|
+
query={'id':rid,'view':'history','limit':limit}
|
|
332
|
+
cursor,snapshot=app.cursors.decode_page(a['cursor'],'history',query,principal,ctx.now) if a.get('cursor') else (None,ctx.now)
|
|
333
|
+
page=await tx.history(rid,cursor=cursor,limit=limit)
|
|
334
|
+
revisions=[]
|
|
335
|
+
for value in page.items:
|
|
336
|
+
row={'id':value.id,'parents':list(value.parents),'digest':value.manifest_digest,
|
|
337
|
+
'created_at':wire(value.created_at),'actor':value.actor,'author':value.author}
|
|
338
|
+
if value.change_note is not None:
|
|
339
|
+
row['change_note']=value.change_note
|
|
340
|
+
if value.source_version is not None:
|
|
341
|
+
row['source_version']=value.source_version
|
|
342
|
+
revisions.append(row)
|
|
343
|
+
data={'id':rid,'revisions':revisions}
|
|
344
|
+
if page.next_cursor:
|
|
345
|
+
cursor=app.cursors.encode_page('history',query,page.next_cursor,snapshot,
|
|
346
|
+
principal,ctx.now+timedelta(minutes=15))
|
|
347
|
+
data.update(cursor=cursor,next=next_link(app,'discovery.get',{**a,'cursor':cursor}),
|
|
348
|
+
next_requires_auth=ctx.principal.subject is not None)
|
|
349
|
+
else:
|
|
350
|
+
data=await read_projection(app,ctx,request,tx,rid,revision=a.get('revision'),fields=a.get('fields',()))
|
|
351
|
+
# Enveloped clients receive compact JSON, while direct resource reads
|
|
352
|
+
# preserve optional nulls. Match either exact public representation only
|
|
353
|
+
# after the current projection has passed authorization. Never normalize
|
|
354
|
+
# stored Revision/signature bytes or use a cache hint as authority.
|
|
355
|
+
known=a.get('known_digest')
|
|
356
|
+
if known is not None and (known==digest(data) or
|
|
357
|
+
known==digest(wire(data,compact=True))):
|
|
358
|
+
return HandlerOutput(data={'not_modified':True,'digest':known})
|
|
359
|
+
return HandlerOutput(data=data)
|
|
360
|
+
|
|
361
|
+
@op('discovery.read_segment',obj({'id':IDENTIFIER,'revision':IDENTIFIER,
|
|
362
|
+
'max_bytes':{'type':'integer','minimum':32,'maximum':8192},'cursor':STRING}),effect='read')
|
|
363
|
+
async def read_segment(ctx,request,tx):
|
|
364
|
+
a=request.arguments
|
|
365
|
+
principal={'actor':ctx.principal.actor,'subject':ctx.principal.subject,
|
|
366
|
+
'credential_id':ctx.principal.credential_id}
|
|
367
|
+
if a.get('cursor'):
|
|
368
|
+
require(set(a)=={'cursor'},'cursor_query_mismatch')
|
|
369
|
+
query,position=app.cursors.decode_read(a['cursor'],principal,ctx.now)
|
|
370
|
+
ref=decode(ResourceRef,query['ref'])
|
|
371
|
+
max_bytes=query['max_bytes']
|
|
372
|
+
offset=position['offset']
|
|
373
|
+
fence=position['fence']
|
|
374
|
+
line_start=position['line_start']
|
|
375
|
+
continues_previous=position['continued']
|
|
376
|
+
previous=position.get('previous')
|
|
377
|
+
else:
|
|
378
|
+
require(a.get('id') is not None,'read_resource_required')
|
|
379
|
+
rid=await resolve(tx,a['id'])
|
|
380
|
+
ref=ResourceRef(id=rid,revision=a.get('revision'))
|
|
381
|
+
max_bytes=a.get('max_bytes',4096)
|
|
382
|
+
offset=0
|
|
383
|
+
fence=False
|
|
384
|
+
line_start=True
|
|
385
|
+
continues_previous=False
|
|
386
|
+
previous=None
|
|
387
|
+
require(32<=max_bytes<=min(8192,app.settings.server.limits.max_response_bytes//4),
|
|
388
|
+
'query_cost_exceeded')
|
|
389
|
+
await check_access(app,ctx,request,tx,ref.id,'read')
|
|
390
|
+
revision=await tx.revision(ref)
|
|
391
|
+
require(revision.content.media_type.startswith('text/'),'text_required')
|
|
392
|
+
ref=ResourceRef(id=ref.id,revision=revision.id)
|
|
393
|
+
end,text,fence_end,line_end,continued=await markdown_segment(
|
|
394
|
+
app,revision.content,offset,max_bytes,fence,line_start)
|
|
395
|
+
data={'id':ref.id,'revision':revision.id,'range':[offset,end],
|
|
396
|
+
'text':text,'continued_block':continued,'continues_previous':continues_previous}
|
|
397
|
+
if end<revision.content.size:
|
|
398
|
+
cursor=app.cursors.encode_read(ref,max_bytes,end,fence_end,line_end,continued,
|
|
399
|
+
principal,ctx.now+timedelta(minutes=15),
|
|
400
|
+
previous=[offset,fence,line_start,continues_previous])
|
|
401
|
+
data['next']='/_r/c/'+cursor
|
|
402
|
+
if offset>0:
|
|
403
|
+
if previous is None:
|
|
404
|
+
previous=await previous_segment(app,revision.content,offset,max_bytes)
|
|
405
|
+
previous_offset,previous_fence,previous_line,previous_continued=previous
|
|
406
|
+
cursor=app.cursors.encode_read(ref,max_bytes,previous_offset,previous_fence,
|
|
407
|
+
previous_line,previous_continued,principal,
|
|
408
|
+
ctx.now+timedelta(minutes=15))
|
|
409
|
+
data['prev']='/_r/c/'+cursor
|
|
410
|
+
return HandlerOutput(resources=(ref,),data=data)
|
|
411
|
+
|
|
412
|
+
listing={'parent':IDENTIFIER,'type':STRING,'author':IDENTIFIER,'query':STRING,'tag':STRING,
|
|
413
|
+
'state':{'enum':['active','archived','purged']},
|
|
414
|
+
'sort':{'enum':['id','time','name']},'direction':{'enum':['asc','desc']},'limit':{'type':'integer','minimum':1,'maximum':200},'cursor':STRING,'fields':fields}
|
|
415
|
+
async def list_items(ctx,request,tx, *, arguments=None,budget=None):
|
|
416
|
+
a=dict(request.arguments if arguments is None else arguments)
|
|
417
|
+
stable=request.operation=='discovery.read_query'
|
|
418
|
+
internal_page=getattr(request,'internal_page_state',None)
|
|
419
|
+
if internal_page is not None:
|
|
420
|
+
# Only the installed QueryRef adapter supplies this after verifying
|
|
421
|
+
# its MAC, sealed source, principal, expiry and query digest.
|
|
422
|
+
require(stable and set(internal_page)=={'arguments','last','snapshot'},
|
|
423
|
+
'invalid_cursor')
|
|
424
|
+
a=dict(internal_page['arguments'])
|
|
425
|
+
principal={'actor':ctx.principal.actor,'subject':ctx.principal.subject,
|
|
426
|
+
'credential_id':ctx.principal.credential_id}
|
|
427
|
+
if stable and a.get('cursor'):
|
|
428
|
+
saved_query,_=app.cursors.inspect_page(a['cursor'],ctx.now)
|
|
429
|
+
require(saved_query.get('operation')==request.operation and
|
|
430
|
+
(read_query_version(saved_query.get('arguments',{}))==request.contract_version or
|
|
431
|
+
request.contract_version==2 and read_query_version(saved_query.get('arguments',{}))==1),
|
|
432
|
+
'cursor_query_mismatch')
|
|
433
|
+
supplied={k:v for k,v in a.items() if k!='cursor'}
|
|
434
|
+
require(not supplied,'cursor_query_mismatch')
|
|
435
|
+
a={**saved_query['arguments'],'cursor':a['cursor']}
|
|
436
|
+
if a.get('tag') is not None:
|
|
437
|
+
a['tag']=normalize_tag(a['tag'])
|
|
438
|
+
parent=await resolve(tx,a['parent']) if a.get('parent') else None
|
|
439
|
+
if stable and parent:
|
|
440
|
+
a['parent']=parent
|
|
441
|
+
if parent==TOOLS_SPACE:
|
|
442
|
+
return HandlerOutput(data={'items':await filtered_tools(app,ctx,request,tx)})
|
|
443
|
+
if parent:
|
|
444
|
+
await check_access(app,ctx,request,tx,parent,'list')
|
|
445
|
+
limit=a.get('limit',50)
|
|
446
|
+
if stable:
|
|
447
|
+
fields_count=len(a.get('fields',('id','type','name','revision','generation','path')))
|
|
448
|
+
require(limit<=100 and limit*(fields_count+1)<=1000,'query_cost_exceeded')
|
|
449
|
+
query_hash=digest({k:v for k,v in a.items() if k!='cursor'})
|
|
450
|
+
sort=a.get('sort','id')
|
|
451
|
+
column={'id':'r.id','time':'r.created_at','name':'r.name'}[sort]
|
|
452
|
+
descending=a.get('direction','asc')=='desc'
|
|
453
|
+
comparison,ordering=('<','DESC') if descending else ('>','ASC')
|
|
454
|
+
query_args={k:v for k,v in a.items() if k!='cursor'}
|
|
455
|
+
if stable and internal_page is not None:
|
|
456
|
+
position,snapshot=internal_page['last'],parse_time(internal_page['snapshot'])
|
|
457
|
+
elif stable and a.get('cursor'):
|
|
458
|
+
position,snapshot=app.cursors.decode_page(a['cursor'],request.operation,
|
|
459
|
+
query_args,principal,ctx.now)
|
|
460
|
+
else:
|
|
461
|
+
position=app.cursors.decode(a['cursor'],'page',query_hash) if a.get('cursor') else (['\uffff','\uffff'] if descending else ['', ''])
|
|
462
|
+
snapshot=ctx.now
|
|
463
|
+
filters=['r.state=?']
|
|
464
|
+
parameters=[a.get('state','active')]
|
|
465
|
+
if stable:
|
|
466
|
+
filters.append('r.created_at<=?');parameters.append(wire(snapshot))
|
|
467
|
+
if parent:
|
|
468
|
+
filters.append('r.parent=?'); parameters.append(parent)
|
|
469
|
+
if a.get('type'):
|
|
470
|
+
app.registry.resource_type(a['type'],1)
|
|
471
|
+
filters.append('r.type=?'); parameters.append(a['type'])
|
|
472
|
+
if a.get('author'):
|
|
473
|
+
filters.append('r.owner=?'); parameters.append(await resolve(tx,a['author']))
|
|
474
|
+
if a.get('query'):
|
|
475
|
+
text=a['query'].replace('\\','\\\\').replace('%','\\%').replace('_','\\_')
|
|
476
|
+
filters.append("(r.name LIKE ? ESCAPE '\\' OR EXISTS (SELECT 1 FROM projections p WHERE p.resource_id=r.id AND p.text LIKE ? ESCAPE '\\'))")
|
|
477
|
+
parameters.extend(['%'+text+'%','%'+text+'%'])
|
|
478
|
+
if a.get('tag'):
|
|
479
|
+
filters.append('EXISTS (SELECT 1 FROM resource_tags rt WHERE rt.resource_id=r.id AND rt.tag=?)')
|
|
480
|
+
parameters.append(a['tag'])
|
|
481
|
+
values=[]
|
|
482
|
+
last_position=position
|
|
483
|
+
more=False
|
|
484
|
+
scanned=0
|
|
485
|
+
while len(values)<=limit:
|
|
486
|
+
require(time.monotonic()<ctx.deadline_monotonic,'query_cost_exceeded')
|
|
487
|
+
sql=f"SELECT r.body,{column},r.id FROM resources r WHERE {' AND '.join(filters)} AND ({column},r.id){comparison}(?,?) ORDER BY {column} {ordering},r.id {ordering} LIMIT 128"
|
|
488
|
+
rows=tx.rows(sql,(*parameters,*last_position))
|
|
489
|
+
if not rows:
|
|
490
|
+
break
|
|
491
|
+
for raw,order,rid in rows:
|
|
492
|
+
scanned+=1
|
|
493
|
+
require(scanned<=4096 and time.monotonic()<ctx.deadline_monotonic,'query_cost_exceeded')
|
|
494
|
+
if budget is not None:budget.scan()
|
|
495
|
+
last_position=[order,rid]
|
|
496
|
+
resource=decode(Resource,loads(raw))
|
|
497
|
+
if await visible(app,ctx,request,tx,rid):
|
|
498
|
+
if len(values)==limit:
|
|
499
|
+
more=True
|
|
500
|
+
break
|
|
501
|
+
chosen=a.get('fields',('id','type','name','revision','generation','path'))
|
|
502
|
+
if budget is not None:budget.node(len(chosen))
|
|
503
|
+
data=await metadata(tx,resource)
|
|
504
|
+
require(set(chosen)<=set(data),'unknown_projection_field')
|
|
505
|
+
values.append({k:data[k] for k in chosen})
|
|
506
|
+
position=last_position
|
|
507
|
+
if more or len(rows)<128:
|
|
508
|
+
break
|
|
509
|
+
data={'items':values}
|
|
510
|
+
if stable and request.contract_version==3 and values and not more:
|
|
511
|
+
data['cursor']=app.cursors.encode_page(request.operation,query_args,position,snapshot,
|
|
512
|
+
principal,ctx.now+timedelta(minutes=15))
|
|
513
|
+
if more:
|
|
514
|
+
if stable:
|
|
515
|
+
cursor=app.cursors.encode_page(request.operation,query_args,position,snapshot,
|
|
516
|
+
principal,ctx.now+timedelta(minutes=15))
|
|
517
|
+
data.update(cursor=cursor,next='/_r/c/'+cursor,
|
|
518
|
+
next_requires_auth=ctx.principal.subject is not None)
|
|
519
|
+
else:
|
|
520
|
+
cursor=app.cursors.encode('page',query_hash,position)
|
|
521
|
+
data.update(cursor=cursor,next=next_link(app,request.operation,{**a,'cursor':cursor}),
|
|
522
|
+
next_requires_auth=ctx.principal.subject is not None)
|
|
523
|
+
return HandlerOutput(data=data)
|
|
524
|
+
op('discovery.list',obj(listing),effect='read')(list_items)
|
|
525
|
+
op('discovery.search',obj(listing,('query',)),effect='read')(list_items)
|
|
526
|
+
op('discovery.read_query',obj(listing),effect='read')(list_items)
|
|
527
|
+
|
|
528
|
+
nested_schema=obj({**listing,
|
|
529
|
+
'expand':{'type':'array','items':{'enum':['children','replies']},
|
|
530
|
+
'maxItems':2,'uniqueItems':True},
|
|
531
|
+
'nested_first':{'type':'integer','minimum':1,'maximum':10},
|
|
532
|
+
'collection':{'enum':['children','replies']}},())
|
|
533
|
+
|
|
534
|
+
async def nested_page(ctx,request,tx,args, *, budget=None):
|
|
535
|
+
"""An independent, reauthorized page of one resource's collection."""
|
|
536
|
+
principal={'actor':ctx.principal.actor,'subject':ctx.principal.subject,
|
|
537
|
+
'credential_id':ctx.principal.credential_id}
|
|
538
|
+
a=dict(args)
|
|
539
|
+
if a.get('cursor'):
|
|
540
|
+
saved,_=app.cursors.inspect_page(a['cursor'],ctx.now)
|
|
541
|
+
require(saved.get('operation')==request.operation and
|
|
542
|
+
saved.get('arguments',{}).get('collection') in {'children','replies'} and
|
|
543
|
+
set(a)=={'cursor'},'cursor_query_mismatch')
|
|
544
|
+
a={**saved['arguments'],'cursor':a['cursor']}
|
|
545
|
+
require(a.get('collection') in {'children','replies'} and a.get('parent') and
|
|
546
|
+
not ({'expand','nested_first','type','author','query','tag','state',
|
|
547
|
+
'sort','direction'} & a.keys()),'invalid_nested_query')
|
|
548
|
+
parent=await resolve(tx,a['parent'])
|
|
549
|
+
collection=a['collection']
|
|
550
|
+
limit=a.get('limit',5)
|
|
551
|
+
require(1<=limit<=10 and set(a.get('fields',('id','name','type','path'))) <=
|
|
552
|
+
{'id','name','type','path','revision'},'query_cost_exceeded')
|
|
553
|
+
query_args={k:v for k,v in a.items() if k!='cursor'}
|
|
554
|
+
if a.get('cursor'):
|
|
555
|
+
position,snapshot=app.cursors.decode_page(a['cursor'],request.operation,
|
|
556
|
+
query_args,principal,ctx.now)
|
|
557
|
+
else:
|
|
558
|
+
position,snapshot=('',ctx.now) if collection=='children' else (['',''],ctx.now)
|
|
559
|
+
# A parent whose read/list grant was revoked cannot be used to enumerate
|
|
560
|
+
# descendants even if a previously issued cursor still has a valid MAC.
|
|
561
|
+
await check_access(app,ctx,request,tx,parent,
|
|
562
|
+
'list' if collection=='children' else 'read')
|
|
563
|
+
resource=await tx.resource(parent)
|
|
564
|
+
require(resource.state=='active','ancestor_inactive')
|
|
565
|
+
items=[]
|
|
566
|
+
more=False
|
|
567
|
+
last=position
|
|
568
|
+
if collection=='children':
|
|
569
|
+
scanned=0
|
|
570
|
+
scan_position=position
|
|
571
|
+
while len(items)<=limit:
|
|
572
|
+
require(time.monotonic()<ctx.deadline_monotonic,'query_cost_exceeded')
|
|
573
|
+
rows=tx.rows('''SELECT body,id FROM resources WHERE parent=? AND state='active'
|
|
574
|
+
AND created_at<=? AND id>? ORDER BY id LIMIT 128''',
|
|
575
|
+
(parent,wire(snapshot),scan_position))
|
|
576
|
+
if not rows:
|
|
577
|
+
break
|
|
578
|
+
for raw,rid in rows:
|
|
579
|
+
scanned+=1
|
|
580
|
+
require(scanned<=2048 and time.monotonic()<ctx.deadline_monotonic,'query_cost_exceeded')
|
|
581
|
+
if budget is not None:budget.scan()
|
|
582
|
+
scan_position=rid
|
|
583
|
+
if not await visible(app,ctx,request,tx,rid):
|
|
584
|
+
continue
|
|
585
|
+
if len(items)==limit:
|
|
586
|
+
more=True
|
|
587
|
+
break
|
|
588
|
+
child=decode(Resource,loads(raw))
|
|
589
|
+
value=await metadata(tx,child)
|
|
590
|
+
fields=a.get('fields',('id','name','type','path'))
|
|
591
|
+
if budget is not None:budget.node(len(fields))
|
|
592
|
+
items.append({key:value[key] for key in fields})
|
|
593
|
+
last=rid
|
|
594
|
+
if more or len(rows)<128:
|
|
595
|
+
break
|
|
596
|
+
else:
|
|
597
|
+
revision=await tx.revision(ResourceRef(id=parent)) if resource.revision else None
|
|
598
|
+
items,last,more=await relation_page(app,ctx,request,tx,resource,revision,
|
|
599
|
+
'c',limit,position,snapshot,budget=budget)
|
|
600
|
+
end_cursor=None
|
|
601
|
+
if items:
|
|
602
|
+
end_cursor=app.cursors.encode_page(request.operation,query_args,last,snapshot,
|
|
603
|
+
principal,ctx.now+timedelta(minutes=15))
|
|
604
|
+
page_info={'hasNextPage':more,'endCursor':end_cursor}
|
|
605
|
+
data={'items':items,'pageInfo':page_info}
|
|
606
|
+
if more:
|
|
607
|
+
data['next']='/_r/c/'+end_cursor
|
|
608
|
+
data['next_requires_auth']=ctx.principal.subject is not None
|
|
609
|
+
return data
|
|
610
|
+
|
|
611
|
+
@op('discovery.read_query',nested_schema,effect='read',version=2)
|
|
612
|
+
async def read_query_v2(ctx,request,tx):
|
|
613
|
+
a=dict(request.arguments)
|
|
614
|
+
if a.get('cursor'):
|
|
615
|
+
saved,_=app.cursors.inspect_page(a['cursor'],ctx.now)
|
|
616
|
+
require(saved.get('operation')==request.operation and set(a)=={'cursor'} and
|
|
617
|
+
read_query_version(saved.get('arguments',{})) in {1,2},
|
|
618
|
+
'cursor_query_mismatch')
|
|
619
|
+
if saved.get('arguments',{}).get('collection'):
|
|
620
|
+
return HandlerOutput(data=await nested_page(ctx,request,tx,a))
|
|
621
|
+
a=saved['arguments']
|
|
622
|
+
if a.get('collection'):
|
|
623
|
+
return HandlerOutput(data=await nested_page(ctx,request,tx,request.arguments))
|
|
624
|
+
expand=a.get('expand',())
|
|
625
|
+
require(not expand or 'id' in a.get('fields',('id',)),'unknown_projection_field')
|
|
626
|
+
limit=a.get('limit',50)
|
|
627
|
+
nested_first=a.get('nested_first',5)
|
|
628
|
+
require(not expand or (limit<=10 and limit*len(expand)*(nested_first+1)<=100),
|
|
629
|
+
'query_cost_exceeded')
|
|
630
|
+
root_args=request.arguments if request.arguments.get('cursor') else {**a,'expand':list(expand)}
|
|
631
|
+
result=await list_items(ctx,request,tx,arguments=root_args)
|
|
632
|
+
data=dict(result.data)
|
|
633
|
+
items=[]
|
|
634
|
+
for source in data['items']:
|
|
635
|
+
require(time.monotonic()<ctx.deadline_monotonic,'query_cost_exceeded')
|
|
636
|
+
item=dict(source)
|
|
637
|
+
if expand:
|
|
638
|
+
item['collections']={}
|
|
639
|
+
for collection in expand:
|
|
640
|
+
child_args={'parent':item['id'],'collection':collection,'limit':nested_first}
|
|
641
|
+
item['collections'][collection]=await nested_page(ctx,request,tx,child_args)
|
|
642
|
+
items.append(item)
|
|
643
|
+
data['items']=items
|
|
644
|
+
data['pageInfo']={'hasNextPage':bool(data.get('next')),
|
|
645
|
+
'endCursor':data.get('cursor')}
|
|
646
|
+
return HandlerOutput(data=data)
|
|
647
|
+
|
|
648
|
+
tree_schema=obj({**listing,
|
|
649
|
+
'query_version':{'const':3},
|
|
650
|
+
'limit':{'type':'integer','minimum':1,'maximum':100},
|
|
651
|
+
'fields':{'type':'array','items':{'enum':list(ROOT_FIELDS)},
|
|
652
|
+
'minItems':1,'maxItems':len(ROOT_FIELDS),'uniqueItems':True},
|
|
653
|
+
'collection':{'enum':['children','replies']},
|
|
654
|
+
'expand':expansion_schema()})
|
|
655
|
+
|
|
656
|
+
@op('discovery.read_query',tree_schema,effect='read',version=3)
|
|
657
|
+
async def read_query_v3(ctx,request,tx):
|
|
658
|
+
budget=ReadBudget(ctx.deadline_monotonic,app.settings.server.limits.max_response_bytes)
|
|
659
|
+
budget.check()
|
|
660
|
+
principal={'actor':ctx.principal.actor,'subject':ctx.principal.subject,
|
|
661
|
+
'credential_id':ctx.principal.credential_id}
|
|
662
|
+
supplied=dict(request.arguments)
|
|
663
|
+
a=dict(supplied)
|
|
664
|
+
position=snapshot=None
|
|
665
|
+
if a.get('cursor'):
|
|
666
|
+
require(set(a)=={'cursor'},'cursor_query_mismatch')
|
|
667
|
+
saved,_=app.cursors.inspect_page(a['cursor'],ctx.now)
|
|
668
|
+
require(saved.get('operation')==request.operation and
|
|
669
|
+
read_query_version(saved.get('arguments',{}))==3,'cursor_query_mismatch')
|
|
670
|
+
a=dict(saved['arguments'])
|
|
671
|
+
position,snapshot=app.cursors.decode_page(supplied['cursor'],request.operation,
|
|
672
|
+
a,principal,ctx.now)
|
|
673
|
+
# Cursor arguments are revalidated, not trusted as an open-ended
|
|
674
|
+
# query language merely because their MAC is valid.
|
|
675
|
+
app.registry.validate(tree_schema_ref,a)
|
|
676
|
+
a['query_version']=3
|
|
677
|
+
|
|
678
|
+
async def expand_page(page,plan,depth):
|
|
679
|
+
require(depth<=MAX_READ_DEPTH,'query_cost_exceeded')
|
|
680
|
+
values=[]
|
|
681
|
+
for original in page['items']:
|
|
682
|
+
budget.check()
|
|
683
|
+
item=dict(original)
|
|
684
|
+
if plan:
|
|
685
|
+
require('id' in item,'unknown_projection_field')
|
|
686
|
+
item['collections']={}
|
|
687
|
+
for collection,spec in plan.items():
|
|
688
|
+
child={'parent':item['id'],'collection':collection,
|
|
689
|
+
'limit':spec.get('limit',5),
|
|
690
|
+
'fields':spec.get('fields',list(NESTED_FIELDS)),
|
|
691
|
+
'expand':spec.get('expand',{}),'query_version':3}
|
|
692
|
+
item['collections'][collection]=await collection_page(child,depth+1)
|
|
693
|
+
values.append(item)
|
|
694
|
+
page['items']=values
|
|
695
|
+
budget.output(page)
|
|
696
|
+
return page
|
|
697
|
+
|
|
698
|
+
async def collection_page(arguments,depth, *, last=None,boundary=None):
|
|
699
|
+
require(depth<=MAX_READ_DEPTH,'query_cost_exceeded')
|
|
700
|
+
args=dict(arguments)
|
|
701
|
+
args['parent']=await resolve(tx,args['parent'])
|
|
702
|
+
# The leaf reader is shared with v2; the public cursor additionally
|
|
703
|
+
# binds this v3 expansion tree, its projection and stable parent ID.
|
|
704
|
+
leaf={key:value for key,value in args.items() if key not in {'expand','query_version'}}
|
|
705
|
+
if last is not None:
|
|
706
|
+
inner=app.cursors.encode_page(request.operation,leaf,last,boundary,
|
|
707
|
+
principal,ctx.now+timedelta(minutes=15))
|
|
708
|
+
leaf_request={'cursor':inner}
|
|
709
|
+
else:
|
|
710
|
+
leaf_request=leaf
|
|
711
|
+
page=await nested_page(ctx,request,tx,leaf_request,budget=budget)
|
|
712
|
+
if args['collection']=='replies':
|
|
713
|
+
selected=args.get('fields',NESTED_FIELDS)
|
|
714
|
+
values=[]
|
|
715
|
+
for link in page['items']:
|
|
716
|
+
resource=await tx.resource(link['ref']['id'])
|
|
717
|
+
value=await metadata(tx,resource)
|
|
718
|
+
values.append({key:value[key] for key in selected})
|
|
719
|
+
page['items']=values
|
|
720
|
+
end=page['pageInfo']['endCursor']
|
|
721
|
+
if end:
|
|
722
|
+
leaf_query,_=app.cursors.inspect_page(end,ctx.now)
|
|
723
|
+
last,boundary=app.cursors.decode_page(end,request.operation,
|
|
724
|
+
leaf_query['arguments'],principal,ctx.now)
|
|
725
|
+
outer=app.cursors.encode_page(request.operation,args,last,boundary,
|
|
726
|
+
principal,ctx.now+timedelta(minutes=15))
|
|
727
|
+
require(len(outer)<=8192,'query_cost_exceeded')
|
|
728
|
+
page['pageInfo']['endCursor']=outer
|
|
729
|
+
if page['pageInfo']['hasNextPage']:
|
|
730
|
+
page['next']='/_r/c/'+outer
|
|
731
|
+
return await expand_page(page,args.get('expand',{}),depth)
|
|
732
|
+
|
|
733
|
+
if a.get('collection'):
|
|
734
|
+
require(a.get('parent'),'invalid_nested_query')
|
|
735
|
+
data=await collection_page(a,0,last=position,boundary=snapshot)
|
|
736
|
+
else:
|
|
737
|
+
root_args=supplied if supplied.get('cursor') else a
|
|
738
|
+
result=await list_items(ctx,request,tx,arguments=root_args,budget=budget)
|
|
739
|
+
data=dict(result.data)
|
|
740
|
+
data['pageInfo']={'hasNextPage':bool(data.get('next')),
|
|
741
|
+
'endCursor':data.get('cursor')}
|
|
742
|
+
data=await expand_page(data,a.get('expand',{}),0)
|
|
743
|
+
return HandlerOutput(data=data)
|
|
744
|
+
|
|
745
|
+
tree_schema_ref=ResourceRef(id='schema:discovery.read_query:3')
|
|
746
|
+
|
|
747
|
+
lexical_fields={'type':'array','items':STRING,'maxItems':12,'uniqueItems':True}
|
|
748
|
+
lexical_facets={'type':'array','items':{'enum':['type','tag']},
|
|
749
|
+
'maxItems':2,'uniqueItems':True}
|
|
750
|
+
lexical_schema=obj({'scope':IDENTIFIER,'terms':STRING,'exact':STRING,'not_terms':STRING,
|
|
751
|
+
'mode':{'enum':['all','any']},'field':{'enum':['all','name','body','metadata']},
|
|
752
|
+
'type':STRING,'owner':IDENTIFIER,'author':IDENTIFIER,'tag':STRING,
|
|
753
|
+
'state':{'enum':['active','archived']},'created_after':STRING,'created_before':STRING,
|
|
754
|
+
'updated_after':STRING,'updated_before':STRING,'has_attachment':BOOLEAN,
|
|
755
|
+
'depth':{'type':'integer','minimum':0,'maximum':5},'recursive':BOOLEAN,
|
|
756
|
+
'order':{'enum':['relevance','updated','created','name']},
|
|
757
|
+
'limit':{'type':'integer','minimum':1,'maximum':100},'cursor':STRING,
|
|
758
|
+
'snippet':BOOLEAN,'explain':{'enum':['compact']},'fields':lexical_fields,
|
|
759
|
+
'facets':lexical_facets})
|
|
760
|
+
lexical_schema_v1={**lexical_schema,
|
|
761
|
+
'properties':{name:value for name,value in lexical_schema['properties'].items()
|
|
762
|
+
if name!='facets'}}
|
|
763
|
+
lexical_schema_v3={**lexical_schema,
|
|
764
|
+
'properties':{**lexical_schema['properties'],
|
|
765
|
+
'source_kind':{'enum':['release','user','operation']},
|
|
766
|
+
'relation_type':{'enum':['reply_to','thread_root','quote','repost',
|
|
767
|
+
'attachment','template']}}}
|
|
768
|
+
lexical_schema_v4={**lexical_schema_v3,
|
|
769
|
+
'properties':{**lexical_schema_v3['properties'],
|
|
770
|
+
'suggest':BOOLEAN}}
|
|
771
|
+
|
|
772
|
+
@op('discovery.lexical_search',lexical_schema_v1,effect='read')
|
|
773
|
+
@op('discovery.lexical_search',lexical_schema,effect='read',version=2)
|
|
774
|
+
@op('discovery.lexical_search',lexical_schema_v3,effect='read',version=3)
|
|
775
|
+
@op('discovery.lexical_search',lexical_schema_v4,effect='read',version=4)
|
|
776
|
+
async def lexical_search(ctx,request,tx):
|
|
777
|
+
a=dict(request.arguments)
|
|
778
|
+
internal_page=getattr(request,'internal_page_state',None)
|
|
779
|
+
if internal_page is not None:
|
|
780
|
+
require(set(internal_page)=={'arguments','last','snapshot'},'invalid_cursor')
|
|
781
|
+
a=dict(internal_page['arguments'])
|
|
782
|
+
principal={'actor':ctx.principal.actor,'subject':ctx.principal.subject,
|
|
783
|
+
'credential_id':ctx.principal.credential_id}
|
|
784
|
+
if a.get('cursor'):
|
|
785
|
+
saved,_=app.cursors.inspect_page(a['cursor'],ctx.now)
|
|
786
|
+
require(saved.get('operation')==request.operation and set(a)=={'cursor'},
|
|
787
|
+
'cursor_query_mismatch')
|
|
788
|
+
a={**saved['arguments'],'cursor':a['cursor']}
|
|
789
|
+
# A cursor or sealed QueryRef carries arguments from an earlier call.
|
|
790
|
+
# Keep those arguments inside the version selected for this call too.
|
|
791
|
+
require(not (request.contract_version<4 and 'suggest' in a) and
|
|
792
|
+
not (request.contract_version<3 and
|
|
793
|
+
{'source_kind','relation_type'}&a.keys()) and
|
|
794
|
+
not (request.contract_version<2 and 'facets' in a),
|
|
795
|
+
'cursor_query_mismatch')
|
|
796
|
+
require(a.get('scope') is not None,'search_scope_required')
|
|
797
|
+
scope=await resolve(tx,a['scope'])
|
|
798
|
+
await check_access(app,ctx,request,tx,scope,'list')
|
|
799
|
+
a['scope']=scope
|
|
800
|
+
if a.get('tag'):
|
|
801
|
+
a['tag']=normalize_tag(a['tag'])
|
|
802
|
+
for field in ('terms','exact','not_terms'):
|
|
803
|
+
require(len(a.get(field,''))<=512,'query_cost_exceeded')
|
|
804
|
+
terms=a.get('terms','').casefold().split()
|
|
805
|
+
excluded=a.get('not_terms','').casefold().split()
|
|
806
|
+
exact=a.get('exact','').casefold()
|
|
807
|
+
require((terms or exact) and len(terms)<=8 and len(excluded)<=8,
|
|
808
|
+
'search_query_required')
|
|
809
|
+
limit=a.get('limit',50)
|
|
810
|
+
selected=a.get('fields',('id','path','type','name','revision','author','created_at'))
|
|
811
|
+
require(set(selected)<=set(('id','path','type','name','revision','author',
|
|
812
|
+
'owner','created_at','modified_at','score','rank_reason','snippet','links')),
|
|
813
|
+
'unknown_projection_field')
|
|
814
|
+
require(limit*(len(selected)+2)<=1200,'query_cost_exceeded')
|
|
815
|
+
normalized={key:value for key,value in a.items() if key!='cursor'}
|
|
816
|
+
if internal_page is not None:
|
|
817
|
+
position,snapshot=internal_page['last'],parse_time(internal_page['snapshot'])
|
|
818
|
+
elif a.get('cursor'):
|
|
819
|
+
position,snapshot=app.cursors.decode_page(a['cursor'],request.operation,
|
|
820
|
+
normalized,principal,ctx.now)
|
|
821
|
+
else:
|
|
822
|
+
position,snapshot=[],ctx.now
|
|
823
|
+
cutoff={key:parse_time(a[key]) for key in ('created_after','created_before',
|
|
824
|
+
'updated_after','updated_before') if key in a}
|
|
825
|
+
scope_resource=await tx.resource(scope)
|
|
826
|
+
owner=await resolve(tx,a['owner']) if a.get('owner') else None
|
|
827
|
+
author=await resolve(tx,a['author']) if a.get('author') else None
|
|
828
|
+
results=[]
|
|
829
|
+
facet_counts={name:{} for name in a.get('facets',())}
|
|
830
|
+
# Suggestion counts describe *matched readable resources*, not raw
|
|
831
|
+
# indexed terms. Rebuild on every page so revoked grants disappear.
|
|
832
|
+
suggestions={}
|
|
833
|
+
suggest_prefix=terms[-1] if a.get('suggest') and terms else ''
|
|
834
|
+
if a.get('suggest'):
|
|
835
|
+
require(2<=len(suggest_prefix)<=32,'query_cost_exceeded')
|
|
836
|
+
scanned=0
|
|
837
|
+
# Restrict the SQL candidate set before applying the work budget. A
|
|
838
|
+
# global LIMIT lets unrelated (or unreadable) rows starve a small scope.
|
|
839
|
+
owner_scope=scope_resource.type in {'user','organization'}
|
|
840
|
+
candidates=tx.execute('''WITH RECURSIVE subtree(id,depth) AS (
|
|
841
|
+
SELECT id,0 FROM resources WHERE id=?
|
|
842
|
+
UNION ALL
|
|
843
|
+
SELECT r.id,s.depth+1 FROM resources r JOIN subtree s ON r.parent=s.id
|
|
844
|
+
WHERE s.depth<5
|
|
845
|
+
) SELECT body FROM resources WHERE created_at<=? AND
|
|
846
|
+
(id IN (SELECT id FROM subtree) OR (? AND (owner=? OR grp=?)))
|
|
847
|
+
ORDER BY id''',(scope,wire(snapshot),owner_scope,scope,scope))
|
|
848
|
+
for (raw,) in candidates:
|
|
849
|
+
require(time.monotonic()<ctx.deadline_monotonic,'query_cost_exceeded')
|
|
850
|
+
resource=decode(Resource,loads(raw))
|
|
851
|
+
if resource.state!=a.get('state','active'):
|
|
852
|
+
continue
|
|
853
|
+
chain=await tx.ancestors(resource.id)
|
|
854
|
+
ancestors=[item.id for item in chain]
|
|
855
|
+
scoped_owner=(scope_resource.type in {'user','organization'} and
|
|
856
|
+
(resource.owner==scope or resource.group==scope))
|
|
857
|
+
if resource.id!=scope and scope not in ancestors and not scoped_owner:
|
|
858
|
+
continue
|
|
859
|
+
distance=len(ancestors)-ancestors.index(scope) if scope in ancestors else 0
|
|
860
|
+
if distance>a.get('depth',5) or (not a.get('recursive',True) and distance>1):
|
|
861
|
+
continue
|
|
862
|
+
if a.get('type') and resource.type!=a['type']:
|
|
863
|
+
continue
|
|
864
|
+
if owner and resource.owner!=owner:
|
|
865
|
+
continue
|
|
866
|
+
if a.get('tag') and a['tag'] not in resource.tags:
|
|
867
|
+
continue
|
|
868
|
+
if ('created_after' in cutoff and resource.created_at<cutoff['created_after'] or
|
|
869
|
+
'created_before' in cutoff and resource.created_at>cutoff['created_before'] or
|
|
870
|
+
'updated_after' in cutoff and resource.modified_at<cutoff['updated_after'] or
|
|
871
|
+
'updated_before' in cutoff and resource.modified_at>cutoff['updated_before']):
|
|
872
|
+
continue
|
|
873
|
+
if not await visible(app,ctx,request,tx,resource.id):
|
|
874
|
+
continue
|
|
875
|
+
scanned+=1
|
|
876
|
+
require(scanned<=2000,'query_cost_exceeded')
|
|
877
|
+
revision=await tx.revision(ResourceRef(id=resource.id)) if resource.revision else None
|
|
878
|
+
# These predicates inspect only the current revision of an already
|
|
879
|
+
# readable resource. Historical relations and source metadata must
|
|
880
|
+
# not affect rank, facets or page positions.
|
|
881
|
+
if a.get('source_kind') and (revision is None or
|
|
882
|
+
revision.source_kind!=a['source_kind']):
|
|
883
|
+
continue
|
|
884
|
+
if a.get('relation_type') and (revision is None or not any(
|
|
885
|
+
relation.type==a['relation_type'] for relation in revision.relations)):
|
|
886
|
+
continue
|
|
887
|
+
if author and (revision is None or revision.author!=author):
|
|
888
|
+
continue
|
|
889
|
+
if a.get('has_attachment') is not None and bool(revision and any(
|
|
890
|
+
relation.type=='attachment' for relation in revision.relations))!=a['has_attachment']:
|
|
891
|
+
continue
|
|
892
|
+
name=resource.name
|
|
893
|
+
body=''
|
|
894
|
+
if a.get('field','all') in {'all','body'} and revision and revision.content.media_type.startswith('text/'):
|
|
895
|
+
require(revision.content.size<=65536,'query_cost_exceeded')
|
|
896
|
+
body=(await app.contents.read_bytes(revision.content)).decode('utf-8')
|
|
897
|
+
meta=f'{resource.type} {resource.owner} {resource.group}'
|
|
898
|
+
selected_text={'name':name,'body':body,'metadata':meta}
|
|
899
|
+
field=a.get('field','all')
|
|
900
|
+
active=selected_text if field=='all' else {field:selected_text[field]}
|
|
901
|
+
lowered={key:value.casefold() for key,value in active.items()}
|
|
902
|
+
whole=' '.join(lowered.values())
|
|
903
|
+
if terms and not (all(term in whole for term in terms) if a.get('mode','all')=='all'
|
|
904
|
+
else any(term in whole for term in terms)):
|
|
905
|
+
continue
|
|
906
|
+
if exact and exact not in whole:
|
|
907
|
+
continue
|
|
908
|
+
if any(term in whole for term in excluded):
|
|
909
|
+
continue
|
|
910
|
+
score=sum((5 if key=='name' else 1)*sum(value.count(term) for term in terms)
|
|
911
|
+
for key,value in lowered.items())+(3 if exact else 0)
|
|
912
|
+
order=a.get('order','relevance')
|
|
913
|
+
if order=='relevance':
|
|
914
|
+
sort_key=[-score,-int(resource.modified_at.timestamp()*1000000),resource.id]
|
|
915
|
+
elif order=='updated':
|
|
916
|
+
sort_key=[-int(resource.modified_at.timestamp()*1000000),resource.id]
|
|
917
|
+
elif order=='created':
|
|
918
|
+
sort_key=[-int(resource.created_at.timestamp()*1000000),resource.id]
|
|
919
|
+
else:
|
|
920
|
+
sort_key=[resource.name.casefold(),resource.id]
|
|
921
|
+
path=short_subject_path(await tx.path(resource.id))
|
|
922
|
+
item={'ref':wire(ResourceRef(id=resource.id,revision=resource.revision)),
|
|
923
|
+
'id':resource.id,'path':path,'type':resource.type,'name':name,
|
|
924
|
+
'revision':resource.revision,'author':revision.author if revision else None,
|
|
925
|
+
'owner':resource.owner,'created_at':wire(resource.created_at),
|
|
926
|
+
'modified_at':wire(resource.modified_at),'score':score}
|
|
927
|
+
if a.get('snippet'):
|
|
928
|
+
for key,value in active.items():
|
|
929
|
+
low=lowered[key]
|
|
930
|
+
needle=exact if exact and exact in low else next((term for term in terms if term in low),'')
|
|
931
|
+
if needle:
|
|
932
|
+
start=low.index(needle)
|
|
933
|
+
left=max(0,start-40)
|
|
934
|
+
right=min(len(value),start+len(needle)+40)
|
|
935
|
+
item['snippet']={'field':key,'text':value[left:right],
|
|
936
|
+
'range':[start-left,start-left+len(needle)]}
|
|
937
|
+
break
|
|
938
|
+
if a.get('explain')=='compact':
|
|
939
|
+
item['rank_reason']={'matched_fields':[key for key,value in lowered.items()
|
|
940
|
+
if any(term in value for term in terms) or
|
|
941
|
+
bool(exact and exact in value)],
|
|
942
|
+
'terms':terms,'order':order}
|
|
943
|
+
if 'links' in selected:
|
|
944
|
+
from msg.plugins.discovery import basic_links
|
|
945
|
+
item['links']=await basic_links(app,ctx,request,tx,resource,revision)
|
|
946
|
+
if a.get('fields'):
|
|
947
|
+
item={key:item[key] for key in selected if key in item}
|
|
948
|
+
results.append((sort_key,item))
|
|
949
|
+
if a.get('suggest'):
|
|
950
|
+
# Names alone keep this optional projection small and avoid
|
|
951
|
+
# mining arbitrary body text. A resource contributes at most
|
|
952
|
+
# once to each candidate, regardless of repeated words.
|
|
953
|
+
require(len(name)<=256,'query_cost_exceeded')
|
|
954
|
+
words={word.casefold() for word in re.findall(r'[\w-]+',name)}
|
|
955
|
+
for word in words:
|
|
956
|
+
if (suggest_prefix!=word and word.startswith(suggest_prefix)
|
|
957
|
+
and len(word)<=32):
|
|
958
|
+
suggestions[word]=suggestions.get(word,0)+1
|
|
959
|
+
require(len(suggestions)<=100,'query_cost_exceeded')
|
|
960
|
+
# Aggregate only matched resources after the current read grant was
|
|
961
|
+
# checked. Never derive buckets from the SQL candidates or a
|
|
962
|
+
# previous page cursor: grants can disappear between page reads.
|
|
963
|
+
for name,values in facet_counts.items():
|
|
964
|
+
keys=(resource.type,) if name=='type' else resource.tags
|
|
965
|
+
for key in keys:
|
|
966
|
+
values[key]=values.get(key,0)+1
|
|
967
|
+
require(len(values)<=20,'query_cost_exceeded')
|
|
968
|
+
results.sort(key=lambda row:row[0])
|
|
969
|
+
following=[row for row in results if not position or row[0]>position]
|
|
970
|
+
page=following[:limit]
|
|
971
|
+
data={'items':[item for _,item in page]}
|
|
972
|
+
if facet_counts:
|
|
973
|
+
data['facets']={name:[{'value':key,'count':count}
|
|
974
|
+
for key,count in sorted(values.items(),
|
|
975
|
+
key=lambda pair:(-pair[1],pair[0]))]
|
|
976
|
+
for name,values in facet_counts.items()}
|
|
977
|
+
if a.get('suggest'):
|
|
978
|
+
data['suggestions']=[{'value':word,'count':count}
|
|
979
|
+
for word,count in sorted(suggestions.items(),
|
|
980
|
+
key=lambda pair:(-pair[1],pair[0]))[:10]]
|
|
981
|
+
if len(following)>limit:
|
|
982
|
+
cursor=app.cursors.encode_page(request.operation,normalized,page[-1][0],snapshot,
|
|
983
|
+
principal,ctx.now+timedelta(minutes=15))
|
|
984
|
+
data.update(cursor=cursor,next='/_r/c/'+cursor,
|
|
985
|
+
next_requires_auth=ctx.principal.subject is not None)
|
|
986
|
+
return HandlerOutput(data=data)
|
|
987
|
+
|
|
988
|
+
@op('discovery.grep',obj({'scope':IDENTIFIER,'pattern':STRING,'regex':BOOLEAN,
|
|
989
|
+
'glob':STRING,'exclude_glob':STRING,'case_sensitive':BOOLEAN,
|
|
990
|
+
'before':{'type':'integer','minimum':0,'maximum':3},
|
|
991
|
+
'after':{'type':'integer','minimum':0,'maximum':3},
|
|
992
|
+
'max_matches':{'type':'integer','minimum':1,'maximum':100},
|
|
993
|
+
'max_files':{'type':'integer','minimum':1,'maximum':100},
|
|
994
|
+
'files_with_matches':BOOLEAN,'count_only':BOOLEAN},('scope','pattern')),effect='read')
|
|
995
|
+
async def grep(ctx,request,tx):
|
|
996
|
+
a=request.arguments
|
|
997
|
+
scope=await resolve(tx,a['scope'])
|
|
998
|
+
await check_access(app,ctx,request,tx,scope,'list')
|
|
999
|
+
pattern=a['pattern']
|
|
1000
|
+
require(0<len(pattern)<=128,'invalid_grep_pattern')
|
|
1001
|
+
regular=a.get('regex',False)
|
|
1002
|
+
if regular:
|
|
1003
|
+
# Keep the public regex subset free of backtracking quantifiers,
|
|
1004
|
+
# groups, alternation and backreferences.
|
|
1005
|
+
require(not any(char in pattern for char in '(){}|\\*+?'),
|
|
1006
|
+
'invalid_grep_pattern')
|
|
1007
|
+
try:
|
|
1008
|
+
expression=re.compile(pattern,0 if a.get('case_sensitive',True) else re.IGNORECASE)
|
|
1009
|
+
except re.error as exc:
|
|
1010
|
+
raise Failure('invalid_grep_pattern') from exc
|
|
1011
|
+
else:
|
|
1012
|
+
expression=(re.compile(re.escape(pattern),re.IGNORECASE)
|
|
1013
|
+
if not a.get('case_sensitive',True) else None)
|
|
1014
|
+
max_files=a.get('max_files',50)
|
|
1015
|
+
max_matches=a.get('max_matches',50)
|
|
1016
|
+
seen_files=0
|
|
1017
|
+
bytes_read=0
|
|
1018
|
+
matches=[]
|
|
1019
|
+
files=[]
|
|
1020
|
+
count=0
|
|
1021
|
+
truncated=False
|
|
1022
|
+
scanned=0
|
|
1023
|
+
candidates=tx.execute('''WITH RECURSIVE subtree(id) AS (
|
|
1024
|
+
SELECT id FROM resources WHERE id=?
|
|
1025
|
+
UNION ALL SELECT r.id FROM resources r JOIN subtree s ON r.parent=s.id
|
|
1026
|
+
) SELECT body FROM resources WHERE state='active'
|
|
1027
|
+
AND id IN (SELECT id FROM subtree) ORDER BY id''',(scope,))
|
|
1028
|
+
for (raw,) in candidates:
|
|
1029
|
+
require(time.monotonic()<ctx.deadline_monotonic,'query_cost_exceeded')
|
|
1030
|
+
resource=decode(Resource,loads(raw))
|
|
1031
|
+
if resource.id!=scope and scope not in {item.id for item in await tx.ancestors(resource.id)}:
|
|
1032
|
+
continue
|
|
1033
|
+
if not await visible(app,ctx,request,tx,resource.id):
|
|
1034
|
+
continue
|
|
1035
|
+
if not resource.revision:
|
|
1036
|
+
continue
|
|
1037
|
+
path=short_subject_path(await tx.path(resource.id))
|
|
1038
|
+
if a.get('glob') and not fnmatch.fnmatchcase(path,a['glob']):
|
|
1039
|
+
continue
|
|
1040
|
+
if a.get('exclude_glob') and fnmatch.fnmatchcase(path,a['exclude_glob']):
|
|
1041
|
+
continue
|
|
1042
|
+
scanned+=1
|
|
1043
|
+
require(scanned<=2000,'query_cost_exceeded')
|
|
1044
|
+
revision=await tx.revision(ResourceRef(id=resource.id))
|
|
1045
|
+
if not revision.content.media_type.startswith('text/'):
|
|
1046
|
+
continue
|
|
1047
|
+
require(revision.content.size<=65536,'query_cost_exceeded')
|
|
1048
|
+
if seen_files>=max_files:
|
|
1049
|
+
truncated=True
|
|
1050
|
+
break
|
|
1051
|
+
seen_files+=1
|
|
1052
|
+
bytes_read+=revision.content.size
|
|
1053
|
+
require(bytes_read<=1048576,'query_cost_exceeded')
|
|
1054
|
+
lines=(await app.contents.read_bytes(revision.content)).decode('utf-8').splitlines()
|
|
1055
|
+
found_file=False
|
|
1056
|
+
for line_no,line in enumerate(lines,1):
|
|
1057
|
+
if expression is not None:
|
|
1058
|
+
positions=((found.start(),found.end()) for found in expression.finditer(line))
|
|
1059
|
+
else:
|
|
1060
|
+
def positions_in_line():
|
|
1061
|
+
offset=0
|
|
1062
|
+
while (start:=line.find(pattern,offset))>=0:
|
|
1063
|
+
yield start,start+len(pattern)
|
|
1064
|
+
offset=start+len(pattern)
|
|
1065
|
+
positions=positions_in_line()
|
|
1066
|
+
for start,end in positions:
|
|
1067
|
+
count+=1
|
|
1068
|
+
if not found_file:
|
|
1069
|
+
files.append({'ref':wire(ResourceRef(id=resource.id,revision=revision.id)),
|
|
1070
|
+
'path':path})
|
|
1071
|
+
found_file=True
|
|
1072
|
+
if not (a.get('count_only') or a.get('files_with_matches')):
|
|
1073
|
+
left=max(0,line_no-1-a.get('before',0))
|
|
1074
|
+
right=min(len(lines),line_no+a.get('after',0))
|
|
1075
|
+
matches.append({'ref':wire(ResourceRef(id=resource.id,revision=revision.id)),
|
|
1076
|
+
'path':path,'line_hint':line_no,'range':[start,end],
|
|
1077
|
+
'context':'\n'.join(lines[left:right])[:512]})
|
|
1078
|
+
if count>=max_matches:
|
|
1079
|
+
truncated=True
|
|
1080
|
+
break
|
|
1081
|
+
if truncated:
|
|
1082
|
+
break
|
|
1083
|
+
if truncated:
|
|
1084
|
+
break
|
|
1085
|
+
if a.get('count_only'):
|
|
1086
|
+
return HandlerOutput(data={'count':count,'truncated':truncated})
|
|
1087
|
+
if a.get('files_with_matches'):
|
|
1088
|
+
return HandlerOutput(data={'files':files,'truncated':truncated})
|
|
1089
|
+
return HandlerOutput(data={'matches':matches,'truncated':truncated})
|
|
1090
|
+
|
|
1091
|
+
@op('discovery.operations',obj({'known_digest':STRING}),effect='read')
|
|
1092
|
+
async def operations(ctx,request,tx):
|
|
1093
|
+
data=app.registry.catalog()
|
|
1094
|
+
if request.arguments.get('known_digest')==data['digest']:
|
|
1095
|
+
data={'not_modified':True,'digest':data['digest']}
|
|
1096
|
+
return HandlerOutput(data=data)
|
|
1097
|
+
|
|
1098
|
+
@op('discovery.capabilities',obj({'known_digest':STRING}),effect='read')
|
|
1099
|
+
async def capabilities(ctx,request,tx):
|
|
1100
|
+
values=[wire(s,compact=True) for s in app.registry.capabilities()]
|
|
1101
|
+
hashed=digest(values)
|
|
1102
|
+
return HandlerOutput(data={'not_modified':True,'digest':hashed} if request.arguments.get('known_digest')==hashed
|
|
1103
|
+
else {'version':1,'digest':hashed,'capabilities':values})
|
|
1104
|
+
|
|
1105
|
+
@op('discovery.schema',obj({'operation':STRING},('operation',)),effect='read')
|
|
1106
|
+
async def schema(ctx,request,tx):
|
|
1107
|
+
name=request.arguments['operation']
|
|
1108
|
+
operation,separator,version=name.rpartition('@')
|
|
1109
|
+
if separator:
|
|
1110
|
+
require(version.isdecimal() and int(version)>0,'invalid_operation_version')
|
|
1111
|
+
spec=app.registry.operation(operation,int(version))
|
|
1112
|
+
else:
|
|
1113
|
+
spec=app.registry.operation(name)
|
|
1114
|
+
require('network' in spec.entries,'entry_not_allowed')
|
|
1115
|
+
description=app.registry.describe(spec)
|
|
1116
|
+
return HandlerOutput(data={'operation':description,'input':app.registry.schema(spec.input_schema),
|
|
1117
|
+
'output':app.registry.schema(spec.output_schema),
|
|
1118
|
+
'requires_rules':description['requires_rules']})
|
|
1119
|
+
|
|
1120
|
+
@op('discovery.diff',obj({'left':REF,'right':REF,'offset':INTEGER,'limit':{'type':'integer','minimum':1,'maximum':500}},
|
|
1121
|
+
('left','right')),effect='read')
|
|
1122
|
+
async def diff(ctx,request,tx):
|
|
1123
|
+
refs=[decode(ResourceRef,request.arguments[key]) for key in ('left','right')]
|
|
1124
|
+
content=[]
|
|
1125
|
+
for ref in refs:
|
|
1126
|
+
await check_access(app,ctx,request,tx,ref.id,'read')
|
|
1127
|
+
revision=await tx.revision(ref)
|
|
1128
|
+
require(revision.content.media_type.startswith('text/') or revision.content.media_type=='application/json','text_diff_required')
|
|
1129
|
+
content.append((await app.contents.read_bytes(revision.content)).decode('utf-8').splitlines(keepends=True))
|
|
1130
|
+
lines=list(difflib.unified_diff(*content,fromfile=refs[0].id+'@'+str(refs[0].revision),tofile=refs[1].id+'@'+str(refs[1].revision)))
|
|
1131
|
+
offset,limit=request.arguments.get('offset',0),request.arguments.get('limit',100)
|
|
1132
|
+
return HandlerOutput(data={'diff':''.join(lines[offset:offset+limit]),'next_offset':offset+limit if offset+limit<len(lines) else None})
|
|
1133
|
+
|
|
1134
|
+
@op('discovery.references',obj({'id':IDENTIFIER,'limit':{'type':'integer','minimum':1,'maximum':200}},('id',)),effect='read')
|
|
1135
|
+
async def references(ctx,request,tx):
|
|
1136
|
+
rid=await resolve(tx,request.arguments['id'])
|
|
1137
|
+
await check_access(app,ctx,request,tx,rid,'read')
|
|
1138
|
+
items=[]
|
|
1139
|
+
for source,kind,body in tx.execute('SELECT rel.source_id,rel.type,rel.body FROM relations rel JOIN resources r ON r.id=rel.source_id AND r.revision=rel.revision_id WHERE rel.target_id=? ORDER BY rel.source_id',(rid,)):
|
|
1140
|
+
if await visible(app,ctx,request,tx,source):
|
|
1141
|
+
items.append({'source_id':source,'type':kind,'target':loads(body)['target']})
|
|
1142
|
+
if len(items)>=request.arguments.get('limit',50):
|
|
1143
|
+
break
|
|
1144
|
+
return HandlerOutput(data={'items':items})
|
|
1145
|
+
|
|
1146
|
+
@op('discovery.links',obj({'id':IDENTIFIER,'rel':{'enum':sorted(LINK_RELATIONS)},
|
|
1147
|
+
'cursor':STRING,'limit':{'type':'integer','minimum':1,'maximum':100}}),effect='read')
|
|
1148
|
+
async def links(ctx,request,tx):
|
|
1149
|
+
args=dict(request.arguments)
|
|
1150
|
+
principal={'actor':ctx.principal.actor,'subject':ctx.principal.subject,
|
|
1151
|
+
'credential_id':ctx.principal.credential_id}
|
|
1152
|
+
if args.get('cursor'):
|
|
1153
|
+
saved,_=app.cursors.inspect_page(args['cursor'],ctx.now)
|
|
1154
|
+
require(saved.get('operation')=='discovery.links' and set(args)=={'cursor'},
|
|
1155
|
+
'cursor_query_mismatch')
|
|
1156
|
+
args={**saved['arguments'],'cursor':args['cursor']}
|
|
1157
|
+
require(args.get('id') is not None,'read_resource_required')
|
|
1158
|
+
rid=await resolve(tx,args['id'])
|
|
1159
|
+
await check_access(app,ctx,request,tx,rid,'read')
|
|
1160
|
+
resource=await tx.resource(rid)
|
|
1161
|
+
require(resource.state!='purged','resource_purged')
|
|
1162
|
+
revision=await tx.revision(ResourceRef(id=rid)) if resource.revision else None
|
|
1163
|
+
rel=args.get('rel')
|
|
1164
|
+
if rel is not None:
|
|
1165
|
+
require(rel in LINK_RELATIONS,'unknown_link_relation')
|
|
1166
|
+
limit=args.get('limit',50)
|
|
1167
|
+
query_args={'id':rid,'rel':rel,'limit':limit}
|
|
1168
|
+
if args.get('cursor'):
|
|
1169
|
+
position,snapshot=app.cursors.decode_page(args['cursor'],request.operation,
|
|
1170
|
+
query_args,principal,ctx.now)
|
|
1171
|
+
else:
|
|
1172
|
+
position,snapshot='',ctx.now
|
|
1173
|
+
current=ResourceRef(id=rid,revision=resource.revision)
|
|
1174
|
+
singles=await basic_links(app,ctx,request,tx,resource,revision)
|
|
1175
|
+
collections={'c','f','q','b','h'}
|
|
1176
|
+
if rel in collections:
|
|
1177
|
+
items,last,more=await relation_page(app,ctx,request,tx,resource,revision,rel,
|
|
1178
|
+
limit,position,snapshot)
|
|
1179
|
+
data={'id':rid,'revision':resource.revision,'rel':rel,'items':items}
|
|
1180
|
+
if more:
|
|
1181
|
+
cursor=app.cursors.encode_page(request.operation,query_args,last,snapshot,principal,
|
|
1182
|
+
ctx.now+timedelta(minutes=15))
|
|
1183
|
+
data.update(cursor=cursor,next='/_r/c/'+cursor,
|
|
1184
|
+
next_requires_auth=ctx.principal.subject is not None)
|
|
1185
|
+
return HandlerOutput(data=data)
|
|
1186
|
+
if rel is not None:
|
|
1187
|
+
require(rel in singles,'not_found')
|
|
1188
|
+
return HandlerOutput(data=singles[rel])
|
|
1189
|
+
linkset=dict(singles)
|
|
1190
|
+
for kind in sorted(collections):
|
|
1191
|
+
items,_,_=await relation_page(app,ctx,request,tx,resource,revision,kind,1,'',ctx.now)
|
|
1192
|
+
if items:
|
|
1193
|
+
linkset[kind]={'source':wire(current),'path':f'/_r/{rid}/l/{kind}',
|
|
1194
|
+
'collection':True}
|
|
1195
|
+
return HandlerOutput(data={'id':rid,'revision':resource.revision,'links':linkset})
|
|
1196
|
+
|
|
1197
|
+
@op('discovery.diff_view',obj({'id':IDENTIFIER,'known_revision':IDENTIFIER,
|
|
1198
|
+
'old_revision':IDENTIFIER,'new_revision':IDENTIFIER,'previous':BOOLEAN,
|
|
1199
|
+
'offset':{'type':'integer','minimum':0},'limit':{'type':'integer','minimum':1,'maximum':500}},
|
|
1200
|
+
('id',)),effect='read')
|
|
1201
|
+
async def diff_view(ctx,request,tx):
|
|
1202
|
+
args=request.arguments
|
|
1203
|
+
rid=await resolve(tx,args['id'])
|
|
1204
|
+
await check_access(app,ctx,request,tx,rid,'read')
|
|
1205
|
+
resource=await tx.resource(rid)
|
|
1206
|
+
require(resource.state!='purged' and resource.revision is not None,'revision_not_found')
|
|
1207
|
+
variants=sum((bool(args.get('previous')),bool(args.get('known_revision')),
|
|
1208
|
+
bool(args.get('old_revision') or args.get('new_revision'))))
|
|
1209
|
+
require(variants==1,'invalid_diff_range')
|
|
1210
|
+
if args.get('previous'):
|
|
1211
|
+
current=await tx.revision(ResourceRef(id=rid,revision=resource.revision))
|
|
1212
|
+
require(len(current.parents)==1,'previous_revision_not_found')
|
|
1213
|
+
old,new=current.parents[0],current.id
|
|
1214
|
+
elif args.get('known_revision'):
|
|
1215
|
+
old,new=args['known_revision'],resource.revision
|
|
1216
|
+
else:
|
|
1217
|
+
require(bool(args.get('old_revision')) and bool(args.get('new_revision')),
|
|
1218
|
+
'invalid_diff_range')
|
|
1219
|
+
old,new=args['old_revision'],args['new_revision']
|
|
1220
|
+
before=await tx.revision(ResourceRef(id=rid,revision=old))
|
|
1221
|
+
after=await tx.revision(ResourceRef(id=rid,revision=new))
|
|
1222
|
+
for revision in (before,after):
|
|
1223
|
+
require(revision.content.media_type.startswith('text/') or
|
|
1224
|
+
revision.content.media_type=='application/json','text_diff_required')
|
|
1225
|
+
require(revision.content.size<=4*app.settings.server.limits.max_response_bytes,
|
|
1226
|
+
'diff_requires_transfer')
|
|
1227
|
+
prior_bytes=await app.contents.read_bytes(before.content)
|
|
1228
|
+
current_bytes=await app.contents.read_bytes(after.content)
|
|
1229
|
+
require(prior_bytes.count(b'\n')+current_bytes.count(b'\n')<=20000,
|
|
1230
|
+
'query_cost_exceeded')
|
|
1231
|
+
prior=prior_bytes.decode('utf-8').splitlines(keepends=True)
|
|
1232
|
+
current=current_bytes.decode('utf-8').splitlines(keepends=True)
|
|
1233
|
+
path=short_subject_path(await tx.path(rid))
|
|
1234
|
+
lines=list(difflib.unified_diff(prior,current,fromfile=path+'@'+old,tofile=path+'@'+new))
|
|
1235
|
+
offset,limit=args.get('offset',0),args.get('limit',500)
|
|
1236
|
+
data={'from':wire(ResourceRef(id=rid,revision=old)),
|
|
1237
|
+
'to':wire(ResourceRef(id=rid,revision=new)),
|
|
1238
|
+
'diff':''.join(lines[offset:offset+limit])}
|
|
1239
|
+
for label,revision in (('from_source',before),('to_source',after)):
|
|
1240
|
+
source={name:getattr(revision,name) for name in
|
|
1241
|
+
('change_note','source_kind','source_version','source_digest')
|
|
1242
|
+
if getattr(revision,name) is not None}
|
|
1243
|
+
if source:
|
|
1244
|
+
data[label]=source
|
|
1245
|
+
if offset+limit<len(lines):
|
|
1246
|
+
data['next_offset']=offset+limit
|
|
1247
|
+
data['next']=f'/_r/{rid}/diff/{old}/{new}/o/{offset+limit}'
|
|
1248
|
+
return HandlerOutput(data=data)
|
|
1249
|
+
|
|
1250
|
+
@op('discovery.raw',obj({'id':IDENTIFIER,'revision':IDENTIFIER,'offset':INTEGER,'length':INTEGER},('id',)),effect='read')
|
|
1251
|
+
async def raw(ctx,request,tx):
|
|
1252
|
+
a=request.arguments
|
|
1253
|
+
rid=await resolve(tx,a['id'])
|
|
1254
|
+
await check_access(app,ctx,request,tx,rid,'read')
|
|
1255
|
+
resource=await tx.resource(rid)
|
|
1256
|
+
require(resource.state!='purged','resource_purged')
|
|
1257
|
+
ref=ResourceRef(id=rid,revision=a.get('revision'))
|
|
1258
|
+
revision=await tx.revision(ref)
|
|
1259
|
+
offset=a.get('offset',0)
|
|
1260
|
+
end=offset+a.get('length',revision.content.size-offset)
|
|
1261
|
+
require(0<=offset<=end<=revision.content.size,'invalid_byte_range')
|
|
1262
|
+
ref=ResourceRef(id=rid,revision=revision.id)
|
|
1263
|
+
return HandlerOutput(resources=(ref,),data={'content':wire(revision.content),'range':[offset,end],
|
|
1264
|
+
'filename':resource.name},output=ref)
|
|
1265
|
+
|
|
1266
|
+
finish()
|