rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: generation
|
|
3
|
+
description: >
|
|
4
|
+
The grounded-answer method: answer a question using ONLY the supplied evidence, cite the bracketed chunk id
|
|
5
|
+
that supports each claim, abstain rather than guess when the evidence does not support an answer, and hedge
|
|
6
|
+
according to the certainty of graph-derived facts. A single grounded, cited LLM act (FR-C.9 / FR-Q.6). The applying
|
|
7
|
+
capability enforces the hard guarantees in code (dropping fabricated citations, coercing an uncited answer to
|
|
8
|
+
an abstention) -- this skill teaches only the reading.
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
# Answer generation: grounded, cited, willing to abstain
|
|
12
|
+
|
|
13
|
+
This skill teaches a **method** for turning retrieved evidence into a final answer that never makes a claim
|
|
14
|
+
without a citation. It is a single grounded reading, not a workflow.
|
|
15
|
+
|
|
16
|
+
## The method
|
|
17
|
+
|
|
18
|
+
- **Answer using ONLY the evidence below.** Do not use outside knowledge; if a fact is not in the evidence,
|
|
19
|
+
it is not available to you.
|
|
20
|
+
- **Cite the bracketed chunk id that supports each claim** in `citations`. Every claim in the answer must be
|
|
21
|
+
traceable to an evidence item by its `[chunk_id]`.
|
|
22
|
+
- **Abstain rather than guess.** If the evidence does not support an answer at all, abstain and do not
|
|
23
|
+
fabricate one.
|
|
24
|
+
- **Hedge honestly instead of over-answering.** There are three outcomes, not two. When the evidence only
|
|
25
|
+
*partially* or *tangentially* addresses the question -- it mentions related material but does not actually
|
|
26
|
+
state what was asked -- do NOT present that mention as a confident answer. Give only what the evidence
|
|
27
|
+
supports, cite it, and say plainly what the evidence does *not* establish: this is a **partial** answer, not a
|
|
28
|
+
full one. Reserve a full, confident answer for when the evidence genuinely states it. (An unsupported
|
|
29
|
+
confident answer is the one failure to avoid; an honest "the evidence mentions X but does not state Y" is
|
|
30
|
+
correct behavior, not a miss.)
|
|
31
|
+
- **Judge each item by its actual text; verify the text really instantiates the concept the question asks
|
|
32
|
+
about.** Do not assume an item answers the question just because it was retrieved. If the text does not match
|
|
33
|
+
the concept asked -- e.g. the question asks for a monetary/maximum liability cap but the text is a
|
|
34
|
+
force-majeure / excused-performance clause, or asks for minimum commitments but the text is research notes,
|
|
35
|
+
definitions, or table fragments -- that item does **not** answer the question. Give only what the text
|
|
36
|
+
genuinely supports and mark `<partial/>`, or abstain if nothing supports it. (A confident answer built from
|
|
37
|
+
off-topic evidence is the exact failure to avoid.)
|
|
38
|
+
- **Hedge according to the certainty note, if one is given.** Some questions come with a short certainty note
|
|
39
|
+
flagging that part of the evidence is uncertain or inferred rather than directly stated. When present, present
|
|
40
|
+
the points that depend on such evidence tentatively (an inference, not a settled fact). Never mention certainty,
|
|
41
|
+
confidence, or any internal label to the reader; let it shape only how tentatively you phrase the answer.
|
|
42
|
+
- **The answer is prose for a person; keep the engine's markers out of it.** Evidence items carry machine
|
|
43
|
+
markers -- the citation id, `[dimension=value; ...]` typed-property groups, any `[Exception ... (inferred)]`
|
|
44
|
+
framing, and the `<partial/>` marker itself -- that are **inputs to your judgement, not facts to relay**. Never
|
|
45
|
+
repeat, name, quote, or describe them to the reader, and never narrate how certain or uncertain the engine is
|
|
46
|
+
about a fact. In particular, `<partial/>` is a SIGNAL the engine reads to flag a partial answer, never words for
|
|
47
|
+
the reader: emit it (per the abstention rule above) but keep it out of the answer prose -- write the partial
|
|
48
|
+
answer in plain language, not the tag. Let the markers shape only *how confidently* you answer, then state the
|
|
49
|
+
substance in plain language. Quote the clause's **real text** when it helps; never quote the markers. (Citation
|
|
50
|
+
ids are recorded separately for the reader, so you never spell an id out in the answer prose.)
|
|
51
|
+
- **State a rule together with its inferred exceptions.** When an evidence item is framed as an exception or
|
|
52
|
+
carve-out to another provision (e.g. "[Exception to the liability cap (inferred)] ..."), do not omit it or
|
|
53
|
+
read it as a separate contradictory fact: answer with the rule AND its exceptions in one breath ("capped at
|
|
54
|
+
X, EXCEPT ... for [the carve-outs]"), citing each, and present the exception as an inference -- so the reader
|
|
55
|
+
sees both the limit and the conditions under which it lifts.
|
|
56
|
+
|
|
57
|
+
## What this skill does NOT own (the applying capability's job, enforced in CODE)
|
|
58
|
+
|
|
59
|
+
- **citation validity** -- a citation to an id not present in the evidence is dropped by the capability, not
|
|
60
|
+
trusted from the model;
|
|
61
|
+
- **the no-claim-without-a-citation guarantee** (FR-Q.6) -- an answer that ends up with no valid citation is
|
|
62
|
+
coerced to an abstention by the capability;
|
|
63
|
+
- **the empty-evidence short-circuit** (abstain without a model call) and the **output contract**
|
|
64
|
+
(`GeneratedAnswer`). The model reads; the code guarantees.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""generation agent-skill folder: SKILL.md (the grounded-answer method)."""
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: generic_compliance_judgment
|
|
3
|
+
description: >
|
|
4
|
+
The DOMAIN-AGNOSTIC compliance-judgment method: given ONE subject (a practice, document, statement, or
|
|
5
|
+
scenario) and ONE applicable regulatory requirement, and seeing only the subject text (never the actor's
|
|
6
|
+
internal records or evidence), decide whether the subject clearly VIOLATES the requirement, clearly SATISFIES
|
|
7
|
+
it, or cannot be judged from the text and must be escalated for human review. Applies to ANY regulatory domain
|
|
8
|
+
(safety, privacy, financial, environmental, advertising, ...). The advertising-specific method
|
|
9
|
+
(`compliance_judgment`) is a SPECIALIZATION of this base method with FTC substantiation/disclosure doctrine;
|
|
10
|
+
this base carries none of that domain doctrine. The applying capability owns the deterministic guarantees
|
|
11
|
+
(verdict vocab, conservative default, both-sided citation) -- this skill teaches only the reading.
|
|
12
|
+
---
|
|
13
|
+
|
|
14
|
+
# Compliance judgment (generic): does this subject satisfy or violate this requirement?
|
|
15
|
+
|
|
16
|
+
This skill teaches a **method**, not a behavior, and it is **domain-agnostic** -- it works for any regulation,
|
|
17
|
+
not one vertical. It extends the grounding-judge idea (ADR-0028) from "is X supported by cue Y?" to "does the
|
|
18
|
+
subject X satisfy or violate requirement Y?". A capability applies it with its own contract
|
|
19
|
+
(`ComplianceFinding`) and its own guarantees (see "What this skill does NOT own").
|
|
20
|
+
|
|
21
|
+
## The one hard constraint: you see only the subject text
|
|
22
|
+
|
|
23
|
+
You are given the requirement and the subject. You **cannot** see the actor's internal records, evidence,
|
|
24
|
+
files, or context beyond the text in front of you. So you can only judge what the *subject text itself* shows.
|
|
25
|
+
This constraint is the whole reason the verdict is three-way, not two-way.
|
|
26
|
+
|
|
27
|
+
## The three verdicts
|
|
28
|
+
|
|
29
|
+
Judge the subject against **this one requirement**, reading the requirement's obligation/prohibition/permission
|
|
30
|
+
literally:
|
|
31
|
+
|
|
32
|
+
- **violation** — reserve this for what is **clearly wrong from the subject text itself**: the text shows the
|
|
33
|
+
required act was **not done** (an obligation the subject plainly failed to meet), or a prohibited act **was
|
|
34
|
+
done**, or a stated condition is plainly **breached**. The breach must be evident in the text, not inferred
|
|
35
|
+
from missing evidence.
|
|
36
|
+
|
|
37
|
+
- **needs_review** — the requirement plausibly applies, but the subject text **does not show enough** to confirm
|
|
38
|
+
either compliance or breach (the relevant record, evidence, or detail isn't in the text). You cannot verify it
|
|
39
|
+
from the text alone, so **escalate** it: a human will check the actor's records/evidence. Do **not** call this
|
|
40
|
+
a violation (you don't have the proof) and do **not** clear it as compliant (you can't confirm it either).
|
|
41
|
+
|
|
42
|
+
- **compliant** — the requirement is clearly met **from the text**, or the requirement **does not bite** on this
|
|
43
|
+
subject at all (it governs a different situation than the one the text describes).
|
|
44
|
+
|
|
45
|
+
## The discipline
|
|
46
|
+
|
|
47
|
+
Only say **violation** when the text clearly shows the breach; only say **compliant** when the text clearly
|
|
48
|
+
clears it (or the requirement plainly does not apply); **otherwise `needs_review`**. Uncertainty is escalation,
|
|
49
|
+
never a silent pass. Give a one-sentence rationale grounded in the subject text and the requirement, and a
|
|
50
|
+
confidence in [0,1]. Refer to the material you are judging as "the subject" -- never assume a domain.
|
|
51
|
+
|
|
52
|
+
## What this skill does NOT own (the applying capability's job)
|
|
53
|
+
|
|
54
|
+
- the verdict **vocabulary** and the **conservative default** -- an unreadable or missing verdict maps to
|
|
55
|
+
`needs_review` deterministically, in the capability, not here;
|
|
56
|
+
- the **both-sided citation** -- the exact subject span and the exact requirement clause are attached from the
|
|
57
|
+
INPUTS by the capability, never authored by this skill (the model rules; it never fabricates a citation);
|
|
58
|
+
- any **roll-up** of findings and the **human gate**.
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: okf_navigate
|
|
3
|
+
description: >
|
|
4
|
+
The embedding-free knowledge-navigation method: find the concepts in an Open Knowledge Format (OKF)
|
|
5
|
+
bundle that answer a question by progressive disclosure, not by vector similarity. Keep the bundle in
|
|
6
|
+
interpreter variables, read index signposts and frontmatter with tools, dispatch a selector sub-agent
|
|
7
|
+
to choose which signposts to explore and use the judgeBodies tool to judge candidate bodies in parallel,
|
|
8
|
+
and return the shortlist of concept ids. Applied by the okf_navigate capability (FR-K.6).
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
# OKF navigation: progressive disclosure over a knowledge bundle, without similarity
|
|
12
|
+
|
|
13
|
+
This skill teaches a **method**, not a behavior. An OKF bundle is a directory tree of markdown files
|
|
14
|
+
(one concept per file) with an `index.md` in each directory listing its entries with a one-line
|
|
15
|
+
description, and cross-links between related concepts. The bundle is large; you cannot load it all into
|
|
16
|
+
your context, and you must **never** compute an embedding similarity. You navigate it the way a person
|
|
17
|
+
skims a table of contents: read the signposts, follow the promising ones, read only the documents worth
|
|
18
|
+
reading.
|
|
19
|
+
|
|
20
|
+
## The method: interpreter holds the bundle → a workflow selects and reads via sub-agents
|
|
21
|
+
|
|
22
|
+
You do not see the bundle. You write a **workflow** in the `eval` tool that reads the bundle with `tools.*`
|
|
23
|
+
into interpreter variables (never into your context) and makes the two judgments that need a model:
|
|
24
|
+
|
|
25
|
+
1. **Selection.** For a directory, `tools.readIndex(...)` returns its signposts (each a `path`, a
|
|
26
|
+
`description`, and `is_dir`). You cannot judge them yourself (they are interpreter data, not in your
|
|
27
|
+
context), so you dispatch an **`okf_selector`** sub-agent with the list and it returns which to explore.
|
|
28
|
+
Descend into chosen subdirectories; collect chosen concept files as candidates.
|
|
29
|
+
2. **Reading.** You judge candidate bodies with **`tools.judgeBodies({ rel_paths })`** — hand it a *batch* of
|
|
30
|
+
candidate paths and it reads and judges them **in parallel**, returning one boolean per path (in order). A
|
|
31
|
+
relevant concept's id goes on the shortlist, and you push its cross-links onto the frontier to expand.
|
|
32
|
+
Reading is a tool, not a sub-agent, on purpose: the interpreter runs one JS engine, so dispatching readers
|
|
33
|
+
with `task()` would serialize them; `tools.judgeBodies` fans them out in the tool instead.
|
|
34
|
+
|
|
35
|
+
The bundle stays in interpreter variables; the selector is only ever called (via `task()`) on a focused list
|
|
36
|
+
of signposts, and body judging happens inside `tools.judgeBodies`. The selector and the judge both already
|
|
37
|
+
know the question — you pass the selector only the signposts, and the judge only the batch of paths.
|
|
38
|
+
|
|
39
|
+
### The canonical workflow (write it this way)
|
|
40
|
+
|
|
41
|
+
**Your ONLY action is to emit this workflow to the `eval` tool in one call, then return its JSON result and
|
|
42
|
+
stop.** The bundle is reachable ONLY through the `tools.*` calls inside `eval` — there is NO filesystem, so
|
|
43
|
+
never call `ls`, `glob`, `read_file`, or `write_file`; they find nothing and waste the turn. Do not answer
|
|
44
|
+
from your own knowledge; the answer only comes from running the workflow.
|
|
45
|
+
|
|
46
|
+
Call the tools with an **object argument** exactly as their signatures show (for example
|
|
47
|
+
`tools.readIndex({ rel_dir: dir })`, `tools.judgeBodies({ rel_paths: batch })`). The `okf_selector` sub-agent
|
|
48
|
+
returns a JSON **string** in its text; `JSON.parse` it (extract the object with a `{ ... }` match first). Do
|
|
49
|
+
NOT pass a `responseSchema` to `task()` — forcing structured output makes a reasoning model reject the call;
|
|
50
|
+
the selector is instructed to reply with JSON, so parse its text. `tools.judgeBodies` returns a real array of
|
|
51
|
+
booleans (not text) — use it directly, no parsing.
|
|
52
|
+
|
|
53
|
+
```javascript
|
|
54
|
+
// Navigate an OKF bundle by progressive disclosure. The bundle is read via tools.* into interpreter
|
|
55
|
+
// variables, never into your context; a model is only ever called (task()) on a list of signposts or one
|
|
56
|
+
// document body. No query-to-document similarity is ever computed.
|
|
57
|
+
const BUDGET = await tools.frontierBudget(); // max document bodies to read
|
|
58
|
+
const shortlist = []; // concept ids judged relevant
|
|
59
|
+
const visited = new Set(); // concept paths already read
|
|
60
|
+
|
|
61
|
+
// SELECTION: hand a list of signposts to the selector sub-agent; it returns the indices worth exploring.
|
|
62
|
+
// Send BOTH the name (`path`) and the `description`: a directory's description may be just a count, so the
|
|
63
|
+
// name carries the signal; a concept's description is its summary. The selector needs both to choose well.
|
|
64
|
+
async function pick(signposts) {
|
|
65
|
+
const raw = await task({
|
|
66
|
+
description: "Signposts to choose from:\n" +
|
|
67
|
+
JSON.stringify(signposts.map((s, i) => ({ i, name: s.path, description: s.description, isDir: s.is_dir }))),
|
|
68
|
+
subagentType: "okf_selector",
|
|
69
|
+
});
|
|
70
|
+
let keep = [];
|
|
71
|
+
try { keep = JSON.parse(raw.match(/\{[\s\S]*\}/)[0]).keep || []; } catch (e) { keep = []; }
|
|
72
|
+
return keep.filter((i) => Number.isInteger(i) && i >= 0 && i < signposts.length).map((i) => signposts[i]);
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
// Descend the tree to a depth bound. The selector gates DIRECTORIES (which subtrees are worth exploring),
|
|
76
|
+
// but once a directory is chosen, collect ALL of its concept files -- do not sub-select concepts. This
|
|
77
|
+
// maximizes recall: a chosen category's clauses are all read (the reader is the precision filter). A
|
|
78
|
+
// signpost's `path` is already resolved (root-relative) by the tool -- pass it straight to readIndex/readBody.
|
|
79
|
+
async function collect(dir, depth) {
|
|
80
|
+
if (depth > 3) return [];
|
|
81
|
+
const signposts = await tools.readIndex({ rel_dir: dir });
|
|
82
|
+
if (!signposts.length) return [];
|
|
83
|
+
const dirs = signposts.filter((s) => s.is_dir);
|
|
84
|
+
const out = signposts.filter((s) => !s.is_dir).map((s) => s.path); // ALL concepts in this directory
|
|
85
|
+
if (dirs.length) {
|
|
86
|
+
const chosenDirs = await pick(dirs); // the selector chooses only which subdirectories to descend
|
|
87
|
+
for (const d of chosenDirs) out.push(...(await collect(d.path, depth + 1)));
|
|
88
|
+
}
|
|
89
|
+
return out;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
const frontier = await collect("", 0);
|
|
93
|
+
const log = []; // one entry per body judged: { path, relevant } — the traversal's decision trace
|
|
94
|
+
const BATCH = 16; // paths handed to judgeBodies at once; the tool judges them in parallel internally
|
|
95
|
+
// READING: hand batches of candidate paths to judgeBodies (it reads + judges them in PARALLEL and returns a
|
|
96
|
+
// boolean per path), keep the relevant, expand relevant concepts' links. Do NOT read/judge one body at a
|
|
97
|
+
// time -- that serializes the slow step. The tool owns the fan-out; you just batch and combine.
|
|
98
|
+
while (frontier.length && shortlist.length < BUDGET) {
|
|
99
|
+
const batch = [];
|
|
100
|
+
while (frontier.length && batch.length < BATCH) {
|
|
101
|
+
const p = frontier.shift();
|
|
102
|
+
if (!visited.has(p)) { visited.add(p); batch.push(p); }
|
|
103
|
+
}
|
|
104
|
+
if (!batch.length) break;
|
|
105
|
+
const verdicts = await tools.judgeBodies({ rel_paths: batch }); // parallel [bool], one per path, in order
|
|
106
|
+
for (let k = 0; k < batch.length && shortlist.length < BUDGET; k++) {
|
|
107
|
+
log.push({ path: batch[k], relevant: verdicts[k] });
|
|
108
|
+
if (verdicts[k]) {
|
|
109
|
+
shortlist.push(await tools.conceptId({ rel_path: batch[k] }));
|
|
110
|
+
for (const link of await tools.related({ rel_path: batch[k] })) frontier.push(link); // lateral expansion
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
JSON.stringify({ shortlist, considered: log.length, log });
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
**These rules are not optional:**
|
|
118
|
+
|
|
119
|
+
1. **Never compute similarity.** There is no embedding step and no query-to-document scoring; relevance
|
|
120
|
+
is judged by `tools.judgeBodies` reading each document, and selection by the selector reading signposts.
|
|
121
|
+
2. **The judgments are not yours to make.** Which signposts to explore comes from `okf_selector`; whether a
|
|
122
|
+
document is relevant comes from `tools.judgeBodies`. You do not decide these yourself — you cannot, the
|
|
123
|
+
signposts and bodies are interpreter data, not in your context.
|
|
124
|
+
3. **Read bodies only after the signpost sift.** Select from `index.md` descriptions first; read a body
|
|
125
|
+
only for a concept the selector kept. Bodies read must stay a small fraction of concepts considered.
|
|
126
|
+
4. **Pass tool arguments as an object**, matching each tool's signature (`{ rel_dir }`, `{ rel_path }`).
|
|
127
|
+
5. **Return `JSON.stringify({ shortlist })`** as the final expression, and nothing else.
|
|
128
|
+
|
|
129
|
+
## What this skill does NOT own (deferred to the applying capability)
|
|
130
|
+
|
|
131
|
+
- **Which model each sub-agent uses** — the model-profile seam (the navigation judgments run on the strong
|
|
132
|
+
model), never a hardcoded flag.
|
|
133
|
+
- **The bounds** (depth, frontier budget) — supplied as run parameters via `tools.*`.
|
|
134
|
+
- **The bundle itself** — compiled by `okf_compile` (FR-K.1-K.4); this method only reads it.
|
|
135
|
+
|
|
136
|
+
The method is the shape of the computation; the capability supplies the tools, the sub-agents, the bounds,
|
|
137
|
+
and the tests. Keep this file about the shape.
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: requirement_extraction
|
|
3
|
+
description: >
|
|
4
|
+
The regulatory-rule extraction method: read one § section of a regulation and pull out the distinct DEONTIC
|
|
5
|
+
rules it states (each an obligation / prohibition / permission), with the actor it binds, the claim types it
|
|
6
|
+
applies to, and any evidence standard. Used by the requirement_extraction SUBGRAPH's extraction node. Because
|
|
7
|
+
a section spreads its rules across the whole text, extraction runs multi-call (skeleton-then-fill) -- which is
|
|
8
|
+
why the capability is a subgraph, not a single-shot skill. The schema is the co-located asset template.py; the
|
|
9
|
+
deterministic mapping to the closed Requirement vocab is the requirement_adaptation FUNCTION's job.
|
|
10
|
+
---
|
|
11
|
+
|
|
12
|
+
# Requirement extraction: what deontic rules does this regulatory section state?
|
|
13
|
+
|
|
14
|
+
This skill teaches a **method**, not a behavior. It turns one regulatory § section into the deontic rules the
|
|
15
|
+
compliance check will judge against. The extraction **schema** is the co-located asset **`template.py`**
|
|
16
|
+
(`ExtractedRegulationSection` → `ExtractedRequirement[]`), filled by the model.
|
|
17
|
+
|
|
18
|
+
## What to extract (the schema — `template.py`)
|
|
19
|
+
|
|
20
|
+
Per section, an `ExtractedRegulationSection` whose `requirements` are the distinct **operative** rules. For each:
|
|
21
|
+
|
|
22
|
+
- **requirement_text** — one rule, paraphrased in a sentence: what must, must not, or may be done.
|
|
23
|
+
- **deontic_type** — `obligation` (must/required), `prohibition` (must not/may not), or `permission` (may).
|
|
24
|
+
- **actor** — who the rule binds (advertiser, endorser, expert, …).
|
|
25
|
+
- **claim_types** — which advertising claim types it applies to (from the closed vocab), when the rule is
|
|
26
|
+
claim-type-specific; leave empty when it applies by context (the applicability map handles that downstream).
|
|
27
|
+
- **evidence_standard** — the substantiation the rule requires, if any.
|
|
28
|
+
|
|
29
|
+
Extract only **operative rules**, not the section's definitions or purpose statements.
|
|
30
|
+
|
|
31
|
+
## The reliability method (docling-graph, from GP-1B + EXTRACT-TUNE)
|
|
32
|
+
|
|
33
|
+
1. **source must be a file path**, not a raw string (docling-graph `stat()`s it — write a temp file).
|
|
34
|
+
2. **`structured_output=False`** (json_object) — the strict nested json_schema returns nothing on some models.
|
|
35
|
+
3. **a `max_tokens` cap** + a wide `preamble_chars` (a full section, not a contract preamble).
|
|
36
|
+
4. **`extraction_contract="auto"`** -- CRITICAL. A regulatory section spreads its rules across the whole text, so
|
|
37
|
+
a single `"direct"` call SILENTLY self-rations and loses most of them (measured: §255.5 → 6 direct vs 31
|
|
38
|
+
dense). `"auto"` picks dense (skeleton-then-fill, multiple calls) on long sections. This multi-call is the
|
|
39
|
+
reason the capability is a **subgraph**.
|
|
40
|
+
|
|
41
|
+
## What this skill does NOT own (the subgraph / the function)
|
|
42
|
+
|
|
43
|
+
- the deterministic mapping to the closed vocab (`deontic_type` coercion -- off-vocab → AMBIGUOUS; off-vocab
|
|
44
|
+
claim_type dropped; blank rule skipped), the `citation` (= the section) and the content-hash `requirement_id`
|
|
45
|
+
-- the `requirement_adaptation` FUNCTION;
|
|
46
|
+
- the extract → adapt chaining, retry/dead-letter, and the write -- the `requirement_extraction` SUBGRAPH and the
|
|
47
|
+
`compliance_ingestion` corpus driver.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""requirement_extraction agent-skill folder: SKILL.md + the template.py schema asset."""
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""Schema ASSET for the `requirement_extraction` subgraph's extraction node (referenced by this folder's SKILL.md).
|
|
2
|
+
|
|
3
|
+
The docling-graph extraction template the extraction node fills: `ExtractedRegulationSection` (a § section) with
|
|
4
|
+
its `ExtractedRequirement`s. Loose strings by design (robust to model output); the deterministic
|
|
5
|
+
`requirement_adaptation` FUNCTION maps them to the closed CC-1 `Requirement` vocab. Co-located with the skill
|
|
6
|
+
because the schema IS part of the authored extraction method (an Agent-Skill asset).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
12
|
+
|
|
13
|
+
from rag_wright.capabilities.dg_extraction import edge
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class ExtractedRequirement(BaseModel):
|
|
17
|
+
"""One rule the LLM reads out of a regulatory section (a docling-graph child entity). Loose strings by
|
|
18
|
+
design (robust to model output); the adapter maps them to the closed CC-1 vocab."""
|
|
19
|
+
|
|
20
|
+
model_config = ConfigDict(graph_id_fields=["requirement_text"], extra="ignore", populate_by_name=True)
|
|
21
|
+
|
|
22
|
+
requirement_text: str = Field(
|
|
23
|
+
description="One rule the section states, paraphrased in a single sentence: what must, must not, or may be done")
|
|
24
|
+
deontic_type: str = Field(
|
|
25
|
+
default="obligation",
|
|
26
|
+
description="obligation (must / required), prohibition (must not / may not), or permission (may / allowed)")
|
|
27
|
+
actor: str = Field(default="", description="Who the rule binds, e.g. advertiser, endorser, expert")
|
|
28
|
+
claim_types: list[str] = Field(
|
|
29
|
+
default_factory=list,
|
|
30
|
+
description=("Which advertising claim types this rule applies to, chosen from: efficacy, comparative, "
|
|
31
|
+
"pricing, health, environmental, endorsement, performance, guarantee"))
|
|
32
|
+
applicability: list[str] = Field(
|
|
33
|
+
default_factory=list,
|
|
34
|
+
description=("P3a (Gap 2): the conditions under which THIS rule applies, as 'dimension: value' pairs (one "
|
|
35
|
+
"per entry) -- for ANY policy domain, not only advertising. E.g. 'jurisdiction: California', "
|
|
36
|
+
"'employee_class: hourly', 'data_category: biometric', 'product_category: supplement'. Leave "
|
|
37
|
+
"empty if the rule applies unconditionally. (Advertising claim types go in claim_types.)"))
|
|
38
|
+
evidence_standard: str = Field(
|
|
39
|
+
default="", description="The substantiation the rule requires, if any (e.g. competent and reliable scientific evidence)")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class ExtractedRegulationSection(BaseModel):
|
|
43
|
+
"""A regulatory section and the distinct rules it states (the docling-graph root entity)."""
|
|
44
|
+
|
|
45
|
+
model_config = ConfigDict(graph_id_fields=["section"], extra="ignore", populate_by_name=True)
|
|
46
|
+
|
|
47
|
+
section: str = Field(description="The section number, e.g. 255.5")
|
|
48
|
+
requirements: list[ExtractedRequirement] = edge(
|
|
49
|
+
"STATES_REQUIREMENT", default_factory=list,
|
|
50
|
+
description="The distinct rules stated in this section (one entry per rule)")
|
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: rlm
|
|
3
|
+
description: >
|
|
4
|
+
The recursive-language-model (RLM) divide-and-conquer method: when a working set is too large for
|
|
5
|
+
one context window, load it into the code interpreter as data, then write a recursive workflow that
|
|
6
|
+
dispatches the work to sub-agents in code (a fresh decomposer per level, a worker per leaf) and
|
|
7
|
+
combines their results, so the model never attends over the full volume. Applied by the RLM chunking
|
|
8
|
+
capability (ingestion) and the RLM synthesis capability (query).
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
# RLM: divide and conquer over a working set too large for one prompt
|
|
12
|
+
|
|
13
|
+
This skill teaches a **method**, not a behavior. It is authored software (FR-C.10), not a build-tool
|
|
14
|
+
feature. Two capabilities apply it with their own contracts and tests: **RLM chunking** (FR-I.1,
|
|
15
|
+
ingestion) and **RLM synthesis** (FR-Q.5, query). This document is deliberately general; every
|
|
16
|
+
guarantee an application needs (determinism, boundary validation, gating, idempotence) is the
|
|
17
|
+
**applying capability's** job, not the method's (see "What this skill does NOT own").
|
|
18
|
+
|
|
19
|
+
## The problem
|
|
20
|
+
|
|
21
|
+
A prompt has a bounded context window. Real working sets, a whole parsed document or a large candidate
|
|
22
|
+
set of retrieved chunks, routinely exceed it, or fit but degrade quality when the model must attend
|
|
23
|
+
over everything at once. Stuffing the whole volume into one call is the failure mode this method avoids.
|
|
24
|
+
|
|
25
|
+
## The method: interpreter holds the whole → a recursive workflow dispatches sub-agents → combine
|
|
26
|
+
|
|
27
|
+
1. **Get the working set from the runtime, as data — the WHOLE set.** Call
|
|
28
|
+
`const workingSet = await tools.workingSet();` — the runtime returns the full input as ordinary
|
|
29
|
+
interpreter values (strings, lists, dicts), *not* into a prompt. Load **all** of it: never sample,
|
|
30
|
+
slice, or subset it, and never decide part of it suffices — you cannot see it, so you cannot judge
|
|
31
|
+
that. Immediately assert you loaded the whole delivered set against `tools.workingSetSize()` (see the
|
|
32
|
+
canonical workflow); a short load is a silent evidence drop the coverage tail cannot catch, because the
|
|
33
|
+
tail only guarantees coverage over what you loaded. The interpreter, not the model, holds the state and
|
|
34
|
+
the recursion stack, and is not bounded by a context window. The working set is never in your context;
|
|
35
|
+
you only ever hold it as an interpreter variable.
|
|
36
|
+
|
|
37
|
+
2. **Write a recursive `decompose()` workflow in code that dispatches sub-agents.** This is a
|
|
38
|
+
**workflow**: fan the work out to sub-agents with `task()` from interpreter code, never one grinding
|
|
39
|
+
tool call at a time. `decompose(slice)` does one of two things:
|
|
40
|
+
- If the slice is small and focused enough, dispatch it to an **`rlm_slice_worker`** (a leaf): a
|
|
41
|
+
full agentic loop that handles that one slice with its own tools and skills, and sees only that
|
|
42
|
+
slice.
|
|
43
|
+
- Otherwise dispatch a **fresh `rlm_decomposer`** to decide this level's split, then **call
|
|
44
|
+
`decompose()` again on each returned sub-slice.** The recursion lives here, in the interpreter:
|
|
45
|
+
the same function re-enters itself to arbitrary depth, and each level's decomposer is a fresh
|
|
46
|
+
agent with fresh context. A model is only ever called on a focused slice, never on the whole.
|
|
47
|
+
|
|
48
|
+
The recursion is driven by the **interpreter**, not by an agent dispatching itself: on this runtime
|
|
49
|
+
a sub-agent cannot dispatch to itself (a self-referential agent is not constructible). The
|
|
50
|
+
interpreter re-dispatching a fresh `rlm_decomposer` per level *is* the recursion, and it delivers
|
|
51
|
+
the per-level fresh context the method wants.
|
|
52
|
+
|
|
53
|
+
3. **Combine the results in code, recursively if needed.** Collect the per-leaf outputs and reduce them
|
|
54
|
+
in code (a fan-in). When the combined intermediate is itself too large, apply the same reduce
|
|
55
|
+
recursively, so a model is called on the *reduced* material, never on the raw whole.
|
|
56
|
+
|
|
57
|
+
The invariant across all three steps: **the model is only ever called on a small, focused slice; the
|
|
58
|
+
interpreter holds the whole and drives the recursion.**
|
|
59
|
+
|
|
60
|
+
### The canonical workflow (write it this way)
|
|
61
|
+
|
|
62
|
+
The recursive descent, written into the `eval` tool. Read the working set from the runtime tool
|
|
63
|
+
`tools.workingSet()` — it returns the whole working set as a JavaScript value that lives in the
|
|
64
|
+
interpreter and never enters your context. Do **not** expect the working set in your prompt; call the
|
|
65
|
+
tool.
|
|
66
|
+
|
|
67
|
+
Write this workflow **exactly** — the decomposer dispatch MUST carry the item count (`over N items`) or
|
|
68
|
+
the decomposer cannot return index cuts and you fall back to a flat, non-recursive run:
|
|
69
|
+
|
|
70
|
+
```javascript
|
|
71
|
+
// Recursive divide-and-conquer over a working set of items. Read it from the runtime tool (a JS value
|
|
72
|
+
// that stays in the interpreter and never enters the model's context), then fan the work out to
|
|
73
|
+
// sub-agents in code (a "workflow"), never one grinding tool call at a time. The interpreter holds the
|
|
74
|
+
// working set and the recursion stack; the model is only ever called on a focused slice.
|
|
75
|
+
const workingSet = await tools.workingSet(); // [{id, ...}, ...] delivered by the runtime, never in context
|
|
76
|
+
// LOAD-COMPLETENESS ASSERTION (T37): verify you loaded the WHOLE delivered set before anything else. The
|
|
77
|
+
// size comes from the runtime (a scalar it cannot under-read); if the load is short, fail loud rather than
|
|
78
|
+
// silently working over a truncated set — the coverage tail below only guarantees coverage over what you
|
|
79
|
+
// loaded, so an under-read here is a silent evidence drop nothing downstream catches.
|
|
80
|
+
const _delivered = await tools.workingSetSize();
|
|
81
|
+
if (workingSet.length !== _delivered) {
|
|
82
|
+
throw new Error("LOAD UNDER-READ: loaded " + workingSet.length + " of " + _delivered + " delivered items");
|
|
83
|
+
}
|
|
84
|
+
const _MAX_DEPTH = 3; // hard cap on recursion — beyond this, force leaf (no runaway splits)
|
|
85
|
+
const _splitDepths = []; // the depths at which decompose() re-entered itself (proof of descent)
|
|
86
|
+
const _handled = new Set(); // ids of working-set items a leaf worker covered
|
|
87
|
+
async function decompose(items, depth) {
|
|
88
|
+
if (items.length === 0) return []; // empty slice — nothing to dispatch (no-op leaf)
|
|
89
|
+
if (depth >= _MAX_DEPTH) {
|
|
90
|
+
for (const it of items) _handled.add(it.id);
|
|
91
|
+
return [await task({ description: "handle leaf depth " + depth + " over " + items.length + " items: " + JSON.stringify(items), subagentType: "rlm_slice_worker" })];
|
|
92
|
+
}
|
|
93
|
+
const decision = JSON.parse(await task({
|
|
94
|
+
description: "decompose depth " + depth + " over " + items.length + " items",
|
|
95
|
+
subagentType: "rlm_decomposer",
|
|
96
|
+
}));
|
|
97
|
+
if (decision.leaf) {
|
|
98
|
+
for (const it of items) _handled.add(it.id);
|
|
99
|
+
return [await task({ description: "handle leaf depth " + depth + " over " + items.length + " items: " + JSON.stringify(items), subagentType: "rlm_slice_worker" })];
|
|
100
|
+
}
|
|
101
|
+
_splitDepths.push(depth);
|
|
102
|
+
// decision.cuts partition `items` into contiguous groups (no item lost); recurse on every group.
|
|
103
|
+
const bounds = [0, ...decision.cuts, items.length];
|
|
104
|
+
const groups = [];
|
|
105
|
+
for (let i = 0; i < bounds.length - 1; i++) groups.push(items.slice(bounds[i], bounds[i + 1]));
|
|
106
|
+
const handled = await Promise.all(groups.map((g) => decompose(g, depth + 1)));
|
|
107
|
+
return handled.flat();
|
|
108
|
+
}
|
|
109
|
+
const _leaves = await decompose(workingSet, 0);
|
|
110
|
+
// COVERAGE TAIL (structural, in-interpreter): the code holds EVERY item, so it guarantees coverage even
|
|
111
|
+
// if the recursion missed a deep leaf out of context — a silent drop otherwise (T37). Any uncovered item
|
|
112
|
+
// is dispatched now, not dropped. This is code checking coverage, not the model asked to be thorough.
|
|
113
|
+
const _missed = workingSet.filter((it) => !_handled.has(it.id));
|
|
114
|
+
if (_missed.length) {
|
|
115
|
+
for (const it of _missed) _handled.add(it.id);
|
|
116
|
+
_leaves.push(await task({ description: "cover " + _missed.length + " missed items: " + JSON.stringify(_missed), subagentType: "rlm_slice_worker" }));
|
|
117
|
+
}
|
|
118
|
+
JSON.stringify({
|
|
119
|
+
leaves: _leaves,
|
|
120
|
+
leafCount: _leaves.length,
|
|
121
|
+
covered: _handled.size,
|
|
122
|
+
total: workingSet.length,
|
|
123
|
+
missed: _missed.length,
|
|
124
|
+
maxSplitDepth: _splitDepths.length ? Math.max(..._splitDepths) : -1,
|
|
125
|
+
});
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
**These rules are not optional. Follow them exactly:**
|
|
129
|
+
|
|
130
|
+
1. **The split is the decomposer's job, never yours.** How to partition each level MUST come from a
|
|
131
|
+
`rlm_decomposer` return value (`decision.cuts`), obtained by dispatching the decomposer. You do not
|
|
132
|
+
decide the split yourself and you do not know the grouping in advance; only the decomposer does.
|
|
133
|
+
2. **Recurse on the decomposer's output.** When the decomposer returns `cuts`, slice `items` into those
|
|
134
|
+
groups and call `decompose()` again on **each group**. A group may itself split, up to `_MAX_DEPTH`
|
|
135
|
+
(3) — beyond that the code force-terminates the branch as a leaf without calling the decomposer
|
|
136
|
+
again (a hard cap against runaway recursion). Stop a branch early when the decomposer marks that
|
|
137
|
+
slice a leaf (`decision.leaf === true`), then dispatch a `rlm_slice_worker`.
|
|
138
|
+
3. **One `decompose()` per node, one decomposer dispatch per node.** A single decomposer call for the
|
|
139
|
+
whole working set is wrong: that is a flat split, and it defeats the method.
|
|
140
|
+
4. **The coverage tail guarantees completeness — never skip it, and never rely on it as a licence to be
|
|
141
|
+
incomplete.** After the recursion, the tail (above) diffs the working set against `handled` and
|
|
142
|
+
dispatches any item the recursion missed, in interpreter code, before returning. This is *code checking
|
|
143
|
+
coverage against the whole working set you hold*, not you being asked to be thorough — the structural
|
|
144
|
+
guarantee the method needs, because out of context a missed deep leaf is a silent drop that still
|
|
145
|
+
reports success. Still write a complete recursion; the tail is the safety net, not the method.
|
|
146
|
+
|
|
147
|
+
**Anti-pattern (do NOT do this):**
|
|
148
|
+
|
|
149
|
+
```javascript
|
|
150
|
+
// WRONG: calling the decomposer once, then hardcoding the slice list and skipping the recursion.
|
|
151
|
+
await task({ description: "split it", subagentType: "rlm_decomposer" });
|
|
152
|
+
const slices = [ {id: "part_a", ...}, {id: "part_b", ...}, {id: "part_c", ...} ]; // <-- you invented these
|
|
153
|
+
for (const s of slices) { await task({ subagentType: "rlm_slice_worker", description: "handle " + s.id }); }
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
That is the failure this method exists to prevent: it looks like it ran, but the slices came from you, not
|
|
157
|
+
from a recursive decomposition, so the working set was never actually divided and conquered. If you find
|
|
158
|
+
yourself writing a literal array of slices, stop: you should be recursing on `decision.parts` instead.
|
|
159
|
+
|
|
160
|
+
## How the two capabilities apply it
|
|
161
|
+
|
|
162
|
+
- **RLM chunking (FR-I.1).** Working set = one whole parsed document. `decompose` splits along topic /
|
|
163
|
+
section / chapter boundaries into semantically coherent, capped chunks; each leaf worker summarizes
|
|
164
|
+
its chunk. The applying capability keeps the boundary and `chunk_id` computation deterministic
|
|
165
|
+
(that stays a separate, exact function; it is not the RLM method's job).
|
|
166
|
+
- **RLM synthesis (FR-Q.5).** Working set = the candidate chunks a query retrieved. `decompose` slices
|
|
167
|
+
the candidate set; each leaf worker extracts the query-relevant facts from its slice; the code-side
|
|
168
|
+
fan-in reduce combines the extracts, so synthesis never attends over the full chunk volume.
|
|
169
|
+
|
|
170
|
+
## The dynamic-dispatch trigger
|
|
171
|
+
|
|
172
|
+
This workflow only fans out if the run is triggered for code-driven dispatch. The skill declares that
|
|
173
|
+
requirement (`skillRuntime.requiresDynamicDispatch`); the runtime that hosts the RLM node applies the
|
|
174
|
+
trigger so this `eval` workflow fires, rather than a slower one-slice-at-a-time fallback. The skill does
|
|
175
|
+
not carry the trigger phrasing itself.
|
|
176
|
+
|
|
177
|
+
## What this skill does NOT own (deferred to the applying capability)
|
|
178
|
+
|
|
179
|
+
- **Determinism and reproducibility** — temperature zero or structured output, and stable ids. (RLM
|
|
180
|
+
chunking owns this for `chunk_id`s and boundaries, in a separate deterministic function.)
|
|
181
|
+
- **Boundary validation** — checking that produced slices are well-formed and within caps.
|
|
182
|
+
- **Gating** — content-hash gating so unchanged input does no work; incremental, resumable runs.
|
|
183
|
+
- **Model choice** — which model each role uses, via the model-profile seam (never a hardcoded flag).
|
|
184
|
+
|
|
185
|
+
The method is the shape of the computation; the capability supplies the contract, the guarantees, and
|
|
186
|
+
the tests. Keep this file about the shape.
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""RLM skill package — the divide-and-conquer method (FR-C.10).
|
|
2
|
+
|
|
3
|
+
Exposes the authored RLM method: the recursive workflow that holds the working set in the
|
|
4
|
+
interpreter, dispatches sub-agents per level (a fresh ``rlm_decomposer`` each) and per leaf
|
|
5
|
+
(``rlm_slice_worker``), then combines results. Two applying capabilities build on this skill:
|
|
6
|
+
RLM chunking (FR-I.1, ingestion) and RLM synthesis (FR-Q.5, query).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from rag_wright.skills.rlm.agent import (
|
|
10
|
+
GRANTED_SUBAGENTS,
|
|
11
|
+
RLM_DECOMPOSER,
|
|
12
|
+
RLM_SLICE_WORKER,
|
|
13
|
+
RLM_WORKFLOW_JS,
|
|
14
|
+
build_rlm_agent,
|
|
15
|
+
decomposer_config,
|
|
16
|
+
method_prompt,
|
|
17
|
+
rlm_interpreter_session,
|
|
18
|
+
slice_worker_config,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
__all__ = [
|
|
22
|
+
"GRANTED_SUBAGENTS",
|
|
23
|
+
"RLM_DECOMPOSER",
|
|
24
|
+
"RLM_SLICE_WORKER",
|
|
25
|
+
"RLM_WORKFLOW_JS",
|
|
26
|
+
"build_rlm_agent",
|
|
27
|
+
"decomposer_config",
|
|
28
|
+
"method_prompt",
|
|
29
|
+
"rlm_interpreter_session",
|
|
30
|
+
"slice_worker_config",
|
|
31
|
+
]
|