hedit 0.7.5a2__tar.gz → 0.7.5.dev2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. {hedit-0.7.5a2/hedit.egg-info → hedit-0.7.5.dev2}/PKG-INFO +1 -1
  2. {hedit-0.7.5a2 → hedit-0.7.5.dev2/hedit.egg-info}/PKG-INFO +1 -1
  3. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/hedit.egg-info/SOURCES.txt +2 -0
  4. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/pyproject.toml +1 -1
  5. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/agents/workflow.py +108 -30
  6. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/api/main.py +171 -1
  7. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/api/models.py +8 -0
  8. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/cli/api_executor.py +8 -0
  9. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/cli/client.py +12 -0
  10. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/cli/main.py +2 -0
  11. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/version.py +2 -2
  12. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_api_endpoints.py +431 -0
  13. hedit-0.7.5.dev2/tests/test_keyword_extraction.py +368 -0
  14. hedit-0.7.5.dev2/tests/test_no_extend_propagation.py +85 -0
  15. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/LICENSE +0 -0
  16. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/PKG_README.md +0 -0
  17. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/README.md +0 -0
  18. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/hedit.egg-info/dependency_links.txt +0 -0
  19. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/hedit.egg-info/entry_points.txt +0 -0
  20. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/hedit.egg-info/requires.txt +0 -0
  21. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/hedit.egg-info/top_level.txt +0 -0
  22. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/setup.cfg +0 -0
  23. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/__init__.py +0 -0
  24. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/agents/__init__.py +0 -0
  25. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/agents/annotation_agent.py +0 -0
  26. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/agents/assessment_agent.py +0 -0
  27. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/agents/evaluation_agent.py +0 -0
  28. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/agents/feedback_summarizer.py +0 -0
  29. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/agents/feedback_triage_agent.py +0 -0
  30. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/agents/state.py +0 -0
  31. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/agents/validation_agent.py +0 -0
  32. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/agents/vision_agent.py +0 -0
  33. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/api/__init__.py +0 -0
  34. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/api/security.py +0 -0
  35. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/cli/__init__.py +0 -0
  36. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/cli/commands/__init__.py +0 -0
  37. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/cli/config.py +0 -0
  38. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/cli/executor.py +0 -0
  39. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/cli/local_executor.py +0 -0
  40. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/cli/output.py +0 -0
  41. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/scripts/__init__.py +0 -0
  42. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/scripts/process_feedback.py +0 -0
  43. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/telemetry/__init__.py +0 -0
  44. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/telemetry/collector.py +0 -0
  45. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/telemetry/schema.py +0 -0
  46. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/telemetry/storage.py +0 -0
  47. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/utils/__init__.py +0 -0
  48. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/utils/error_remediation.py +0 -0
  49. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/utils/github_client.py +0 -0
  50. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/utils/hed_comprehensive_guide.py +0 -0
  51. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/utils/hed_rules.py +0 -0
  52. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/utils/image_processing.py +0 -0
  53. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/utils/json_schema_loader.py +0 -0
  54. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/utils/litellm_llm.py +0 -0
  55. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/utils/openrouter_llm.py +0 -0
  56. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/utils/schema_loader.py +0 -0
  57. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/validation/__init__.py +0 -0
  58. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/validation/hed_lsp.py +0 -0
  59. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/src/validation/hed_validator.py +0 -0
  60. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_annotation_agent.py +0 -0
  61. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_cli_client.py +0 -0
  62. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_cli_config.py +0 -0
  63. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_cli_integration.py +0 -0
  64. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_cli_main.py +0 -0
  65. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_comprehensive_guide.py +0 -0
  66. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_error_remediation.py +0 -0
  67. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_feedback_integration.py +0 -0
  68. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_feedback_triage.py +0 -0
  69. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_github_client.py +0 -0
  70. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_hed_lsp.py +0 -0
  71. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_integration_openrouter.py +0 -0
  72. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_json_schema_loader.py +0 -0
  73. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_litellm_llm.py +0 -0
  74. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_openrouter_llm.py +0 -0
  75. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_schema_loader.py +0 -0
  76. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_security.py +0 -0
  77. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_state.py +0 -0
  78. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_telemetry.py +0 -0
  79. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_validation.py +0 -0
  80. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_validation_agent.py +0 -0
  81. {hedit-0.7.5a2 → hedit-0.7.5.dev2}/tests/test_version.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: hedit
3
- Version: 0.7.5a2
3
+ Version: 0.7.5.dev2
4
4
  Summary: Multi-agent system for HED annotation generation and validation
5
5
  Author-email: Annotation Garden Initiative <info@annotation.garden>
6
6
  License-Expression: MIT
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: hedit
3
- Version: 0.7.5a2
3
+ Version: 0.7.5.dev2
4
4
  Summary: Multi-agent system for HED annotation generation and validation
5
5
  Author-email: Annotation Garden Initiative <info@annotation.garden>
6
6
  License-Expression: MIT
@@ -66,7 +66,9 @@ tests/test_github_client.py
66
66
  tests/test_hed_lsp.py
67
67
  tests/test_integration_openrouter.py
68
68
  tests/test_json_schema_loader.py
69
+ tests/test_keyword_extraction.py
69
70
  tests/test_litellm_llm.py
71
+ tests/test_no_extend_propagation.py
70
72
  tests/test_openrouter_llm.py
71
73
  tests/test_schema_loader.py
72
74
  tests/test_security.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "hedit"
7
- version = "0.7.5a2"
7
+ version = "0.7.5.dev2"
8
8
  description = "Multi-agent system for HED annotation generation and validation"
9
9
  readme = "PKG_README.md"
10
10
  requires-python = ">=3.12"
@@ -9,6 +9,7 @@ import time
9
9
  from pathlib import Path
10
10
 
11
11
  from langchain_core.language_models import BaseChatModel
12
+ from langchain_core.messages import HumanMessage, SystemMessage
12
13
  from langgraph.graph import END, StateGraph
13
14
 
14
15
  from src.agents.annotation_agent import AnnotationAgent
@@ -17,6 +18,7 @@ from src.agents.evaluation_agent import EvaluationAgent
17
18
  from src.agents.feedback_summarizer import FeedbackSummarizer
18
19
  from src.agents.state import HedAnnotationState
19
20
  from src.agents.validation_agent import ValidationAgent
21
+ from src.utils import extract_text_content
20
22
  from src.utils.schema_loader import HedSchemaLoader
21
23
  from src.validation.hed_lsp import HedLspClient, is_hed_lsp_available
22
24
 
@@ -63,8 +65,8 @@ class HedAnnotationWorkflow:
63
65
  """
64
66
  # Store schema directory (None means use HED library to fetch from GitHub)
65
67
  self.schema_dir = schema_dir
66
- # Enable semantic search only if hed-lsp CLI is available
67
- self.enable_semantic_search = enable_semantic_search and is_hed_lsp_available()
68
+ # Keyword extraction always runs; LSP enrichment requires hed-lsp CLI
69
+ self.enable_semantic_search = enable_semantic_search
68
70
 
69
71
  # Initialize legacy schema loader for validation
70
72
  self.schema_loader = HedSchemaLoader()
@@ -74,6 +76,9 @@ class HedAnnotationWorkflow:
74
76
  assess_llm = assessment_llm or llm
75
77
  feed_llm = feedback_llm or llm
76
78
 
79
+ # Store feedback LLM for keyword extraction (cheap/fast model)
80
+ self.feedback_llm = feed_llm
81
+
77
82
  # Initialize agents with JSON schema support and per-agent LLMs
78
83
  self.annotation_agent = AnnotationAgent(llm, schema_dir=self.schema_dir)
79
84
  self.validation_agent = ValidationAgent(
@@ -85,15 +90,19 @@ class HedAnnotationWorkflow:
85
90
  self.assessment_agent = AssessmentAgent(assess_llm, schema_dir=self.schema_dir)
86
91
  self.feedback_summarizer = FeedbackSummarizer(feed_llm)
87
92
 
88
- # Initialize hed-lsp client for semantic search
93
+ # Initialize hed-lsp client for semantic search (optional enrichment)
89
94
  self.hed_lsp_client: HedLspClient | None = None
90
- if self.enable_semantic_search:
95
+ if self.enable_semantic_search and is_hed_lsp_available():
91
96
  try:
92
97
  self.hed_lsp_client = HedLspClient()
93
98
  logger.info("[WORKFLOW] hed-lsp CLI available for semantic tag suggestions")
94
99
  except RuntimeError as e:
95
100
  logger.warning(f"[WORKFLOW] hed-lsp CLI not available: {e}")
96
- self.enable_semantic_search = False
101
+ elif self.enable_semantic_search:
102
+ logger.info(
103
+ "[WORKFLOW] hed-lsp CLI not in PATH; "
104
+ "keyword extraction will run without LSP enrichment"
105
+ )
97
106
 
98
107
  # Build graph
99
108
  self.graph = self._build_graph()
@@ -156,11 +165,58 @@ class HedAnnotationWorkflow:
156
165
 
157
166
  return workflow.compile() # type: ignore[return-value]
158
167
 
168
+ async def _extract_keywords(self, description: str) -> list[str]:
169
+ """Extract HED-relevant keywords from a natural language description.
170
+
171
+ Uses the feedback LLM (cheap/fast model) to identify key concepts
172
+ that can be mapped to HED tags via the LSP suggest tool.
173
+
174
+ Args:
175
+ description: Natural language event or image description
176
+
177
+ Returns:
178
+ List of extracted keywords (max 20)
179
+ """
180
+ system_prompt = (
181
+ "You are a keyword extractor for neuroscience event descriptions. "
182
+ "Extract the most important concepts that could map to HED "
183
+ "(Hierarchical Event Descriptors) tags.\n\n"
184
+ "Extract:\n"
185
+ "- Objects/entities (person, car, button, screen, face, etc.)\n"
186
+ "- Actions/events (pressing, flashing, appearing, moving, etc.)\n"
187
+ "- Properties/attributes (red, large, fast, loud, etc.)\n"
188
+ "- Spatial relationships (left, center, above, etc.)\n"
189
+ "- Temporal aspects (onset, offset, duration, etc.)\n"
190
+ "- Sensory modalities (visual, auditory, tactile, etc.)\n\n"
191
+ "Return ONLY a comma-separated list of single words or short phrases "
192
+ "(2-3 words max). Return at most 20 keywords. "
193
+ "Do not include any other text, explanation, or formatting."
194
+ )
195
+
196
+ try:
197
+ response = await self.feedback_llm.ainvoke(
198
+ [
199
+ SystemMessage(content=system_prompt),
200
+ HumanMessage(content=f"Description: {description}"),
201
+ ]
202
+ )
203
+ raw_text = extract_text_content(response.content)
204
+ # Parse comma-separated keywords, strip whitespace, filter empty
205
+ keywords = [kw.strip() for kw in raw_text.split(",") if kw.strip()]
206
+ # Limit to 20 keywords
207
+ keywords = keywords[:20]
208
+ logger.info(f"[WORKFLOW] Extracted {len(keywords)} keywords: {keywords}")
209
+ return keywords
210
+ except Exception as e:
211
+ logger.warning("[WORKFLOW] Keyword extraction failed: %s", e, exc_info=True)
212
+ return []
213
+
159
214
  async def _semantic_preprocess_node(self, state: HedAnnotationState) -> dict:
160
- """Semantic preprocessing node: Use hed-lsp CLI to suggest relevant tags.
215
+ """Semantic preprocessing node: Extract keywords and suggest HED tags.
161
216
 
162
217
  This node runs before annotation to provide semantic hints based on
163
- the input description. Uses hed-lsp CLI for tag suggestions.
218
+ the input description. It uses the feedback LLM to extract keywords,
219
+ then passes those keywords to hed-lsp CLI for tag suggestions.
164
220
  Only runs on the first iteration.
165
221
 
166
222
  Args:
@@ -176,33 +232,55 @@ class HedAnnotationWorkflow:
176
232
 
177
233
  logger.info("[WORKFLOW] Entering semantic_preprocess node")
178
234
 
179
- # Use hed-lsp CLI to suggest tags from the description
180
- semantic_hints = []
181
- keywords: list[str] = []
235
+ # Step 1: Extract keywords from the description using LLM
236
+ keywords = await self._extract_keywords(state["input_description"])
237
+
238
+ # Step 2: Use hed-lsp CLI to get tag suggestions for each keyword
239
+ semantic_hints: list[dict] = []
182
240
 
183
- if self.hed_lsp_client:
241
+ if keywords and self.hed_lsp_client:
184
242
  try:
185
- # Get tag suggestions directly from the description
186
- result = self.hed_lsp_client.suggest(state["input_description"])
187
- if result.success:
188
- semantic_hints = [
189
- {
190
- "tag": s.tag,
191
- "prefix": "", # hed-lsp returns full tags
192
- "score": s.score or 0.0,
193
- "source": "hed-lsp",
194
- }
195
- for s in result.suggestions
196
- ]
197
- # Extract keywords from the tags for logging
198
- keywords = [s.tag.split("/")[-1] for s in result.suggestions]
199
- logger.info(
200
- f"[WORKFLOW] hed-lsp suggested {len(semantic_hints)} tags: {keywords[:5]}..."
201
- )
202
- else:
203
- logger.warning(f"[WORKFLOW] hed-lsp suggestion failed: {result.error}")
243
+ # Query hed-lsp for each keyword individually for better results
244
+ for keyword in keywords:
245
+ result = self.hed_lsp_client.suggest(keyword)
246
+ if result.success:
247
+ for s in result.suggestions:
248
+ semantic_hints.append(
249
+ {
250
+ "tag": s.tag,
251
+ "keyword": keyword,
252
+ "score": s.score or 0.0,
253
+ "source": "hed-lsp",
254
+ }
255
+ )
256
+ else:
257
+ logger.debug(
258
+ "[WORKFLOW] hed-lsp suggestion failed for '%s': %s",
259
+ keyword,
260
+ result.error,
261
+ )
262
+
263
+ # Deduplicate by tag, keeping highest score
264
+ seen_tags: dict[str, dict] = {}
265
+ for hint in semantic_hints:
266
+ tag = hint["tag"]
267
+ if tag not in seen_tags or hint["score"] > seen_tags[tag]["score"]:
268
+ seen_tags[tag] = hint
269
+ semantic_hints = sorted(seen_tags.values(), key=lambda h: h["score"], reverse=True)
270
+
271
+ logger.info(
272
+ "[WORKFLOW] hed-lsp suggested %d unique tags from %d keywords",
273
+ len(semantic_hints),
274
+ len(keywords),
275
+ )
204
276
  except Exception as e:
205
277
  logger.warning("[WORKFLOW] hed-lsp error: %s", e, exc_info=True)
278
+ elif keywords:
279
+ # LSP not available; still store keywords for the annotation agent
280
+ logger.info(
281
+ "[WORKFLOW] hed-lsp not available; storing %d extracted keywords",
282
+ len(keywords),
283
+ )
206
284
 
207
285
  return {
208
286
  "extracted_keywords": keywords,
@@ -649,6 +649,7 @@ async def annotate(
649
649
  schema_version=request.schema_version,
650
650
  max_validation_attempts=request.max_validation_attempts,
651
651
  run_assessment=request.run_assessment,
652
+ no_extend=request.no_extend,
652
653
  config=config,
653
654
  )
654
655
  latency_ms = int((time.time() - start_time) * 1000)
@@ -856,6 +857,7 @@ async def annotate_from_image(
856
857
  schema_version=request.schema_version,
857
858
  max_validation_attempts=request.max_validation_attempts,
858
859
  run_assessment=request.run_assessment,
860
+ no_extend=request.no_extend,
859
861
  config=config,
860
862
  )
861
863
  latency_ms = int((time.time() - start_time) * 1000)
@@ -925,6 +927,64 @@ async def annotate_from_image(
925
927
  ) from e
926
928
 
927
929
 
930
+ async def _collect_stream_telemetry(
931
+ request: AnnotationRequest | ImageAnnotationRequest,
932
+ req: Request,
933
+ current_state: dict,
934
+ start_time: float,
935
+ source: str,
936
+ description: str,
937
+ ) -> None:
938
+ """Collect telemetry for streaming endpoints.
939
+
940
+ Shared helper used by both /annotate/stream and /annotate-from-image/stream.
941
+ Silently returns if telemetry is disabled or collector is not initialized.
942
+
943
+ Args:
944
+ request: The annotation request (text or image)
945
+ req: FastAPI request for header extraction
946
+ current_state: Current workflow state dict
947
+ start_time: Workflow start time (from time.time())
948
+ source: Telemetry source identifier (e.g., "api-stream", "api-image-stream")
949
+ description: Input description text (or image description for image endpoints)
950
+ """
951
+ if not request.telemetry_enabled or not telemetry_collector:
952
+ return
953
+
954
+ latency_ms = int((time.time() - start_time) * 1000)
955
+
956
+ # Get model info from request body, BYOK headers, or server config
957
+ model_name = (
958
+ request.model
959
+ or req.headers.get("x-openrouter-model")
960
+ or os.getenv("ANNOTATION_MODEL", "openai/gpt-oss-120b")
961
+ )
962
+ temperature = request.temperature
963
+ if temperature is None:
964
+ temp_header = req.headers.get("x-openrouter-temperature")
965
+ if temp_header is not None:
966
+ try:
967
+ temperature = float(temp_header)
968
+ except ValueError:
969
+ temperature = None
970
+ if temperature is None:
971
+ temperature = _byok_config.get("temperature", 0.1)
972
+
973
+ event = TelemetryEvent.create(
974
+ description=description,
975
+ schema_version=request.schema_version,
976
+ hed_string=current_state.get("current_annotation", ""),
977
+ iterations=current_state.get("validation_attempts", 0),
978
+ validation_errors=current_state.get("validation_errors", []),
979
+ model=model_name,
980
+ provider=request.provider or req.headers.get("x-openrouter-provider"),
981
+ temperature=temperature,
982
+ latency_ms=latency_ms,
983
+ source=source,
984
+ )
985
+ await telemetry_collector.collect(event)
986
+
987
+
928
988
  @app.post("/annotate/stream")
929
989
  async def annotate_stream(
930
990
  request: AnnotationRequest,
@@ -1019,6 +1079,7 @@ async def annotate_stream(
1019
1079
  request.schema_version,
1020
1080
  request.max_validation_attempts,
1021
1081
  run_assessment=request.run_assessment,
1082
+ no_extend=request.no_extend,
1022
1083
  )
1023
1084
 
1024
1085
  # Node name to user-friendly stage mapping
@@ -1039,6 +1100,9 @@ async def annotate_stream(
1039
1100
  # SSE padding comment to force Safari to open the stream
1040
1101
  yield ": stream opened\n\n"
1041
1102
 
1103
+ start_time = time.time()
1104
+ current_state = initial_state.copy()
1105
+
1042
1106
  try:
1043
1107
  # Send initial start event
1044
1108
  yield send_event(
@@ -1046,7 +1110,6 @@ async def annotate_stream(
1046
1110
  )
1047
1111
 
1048
1112
  # Track state and progress
1049
- current_state = initial_state.copy()
1050
1113
  last_stage = None
1051
1114
  validation_attempt = 0
1052
1115
 
@@ -1129,6 +1192,20 @@ async def annotate_stream(
1129
1192
  }
1130
1193
 
1131
1194
  yield send_event("result", result)
1195
+
1196
+ # Collect telemetry after sending result but before done event
1197
+ try:
1198
+ await _collect_stream_telemetry(
1199
+ request=request,
1200
+ req=req,
1201
+ current_state=current_state,
1202
+ start_time=start_time,
1203
+ source="api-stream",
1204
+ description=request.description,
1205
+ )
1206
+ except Exception:
1207
+ logging.debug("Telemetry collection failed for streaming request", exc_info=True)
1208
+
1132
1209
  yield send_event("done", {"message": "Workflow completed"})
1133
1210
 
1134
1211
  except asyncio.CancelledError:
@@ -1142,6 +1219,18 @@ async def annotate_stream(
1142
1219
  "error_type": "timeout",
1143
1220
  },
1144
1221
  )
1222
+ # Collect telemetry on error
1223
+ try:
1224
+ await _collect_stream_telemetry(
1225
+ request=request,
1226
+ req=req,
1227
+ current_state=current_state,
1228
+ start_time=start_time,
1229
+ source="api-stream",
1230
+ description=request.description,
1231
+ )
1232
+ except Exception:
1233
+ logging.debug("Telemetry collection failed on timeout", exc_info=True)
1145
1234
  yield send_event("done", {"message": "Workflow ended with error"})
1146
1235
  except RateLimitError:
1147
1236
  logging.exception("Streaming workflow rate limit")
@@ -1152,6 +1241,18 @@ async def annotate_stream(
1152
1241
  "error_type": "rate_limit",
1153
1242
  },
1154
1243
  )
1244
+ # Collect telemetry on error
1245
+ try:
1246
+ await _collect_stream_telemetry(
1247
+ request=request,
1248
+ req=req,
1249
+ current_state=current_state,
1250
+ start_time=start_time,
1251
+ source="api-stream",
1252
+ description=request.description,
1253
+ )
1254
+ except Exception:
1255
+ logging.debug("Telemetry collection failed on rate limit", exc_info=True)
1155
1256
  yield send_event("done", {"message": "Workflow ended with error"})
1156
1257
  except Exception:
1157
1258
  logging.exception("Streaming workflow error")
@@ -1162,6 +1263,18 @@ async def annotate_stream(
1162
1263
  "error_type": "internal",
1163
1264
  },
1164
1265
  )
1266
+ # Collect telemetry on error
1267
+ try:
1268
+ await _collect_stream_telemetry(
1269
+ request=request,
1270
+ req=req,
1271
+ current_state=current_state,
1272
+ start_time=start_time,
1273
+ source="api-stream",
1274
+ description=request.description,
1275
+ )
1276
+ except Exception:
1277
+ logging.debug("Telemetry collection failed on error", exc_info=True)
1165
1278
  yield send_event("done", {"message": "Workflow ended with error"})
1166
1279
 
1167
1280
  return StreamingResponse(
@@ -1306,6 +1419,10 @@ async def annotate_from_image_stream(
1306
1419
  # SSE padding comment to force Safari to open the stream
1307
1420
  yield ": stream opened\n\n"
1308
1421
 
1422
+ start_time = time.time()
1423
+ current_state: dict = {}
1424
+ image_description = ""
1425
+
1309
1426
  try:
1310
1427
  # Send initial start event
1311
1428
  yield send_event(
@@ -1337,6 +1454,7 @@ async def annotate_from_image_stream(
1337
1454
  request.schema_version,
1338
1455
  request.max_validation_attempts,
1339
1456
  run_assessment=request.run_assessment,
1457
+ no_extend=request.no_extend,
1340
1458
  )
1341
1459
 
1342
1460
  # Track state and progress
@@ -1425,6 +1543,22 @@ async def annotate_from_image_stream(
1425
1543
  }
1426
1544
 
1427
1545
  yield send_event("result", result)
1546
+
1547
+ # Collect telemetry after sending result but before done event
1548
+ try:
1549
+ await _collect_stream_telemetry(
1550
+ request=request,
1551
+ req=req,
1552
+ current_state=current_state,
1553
+ start_time=start_time,
1554
+ source="api-image-stream",
1555
+ description=image_description,
1556
+ )
1557
+ except Exception:
1558
+ logging.debug(
1559
+ "Telemetry collection failed for image streaming request", exc_info=True
1560
+ )
1561
+
1428
1562
  yield send_event("done", {"message": "Workflow completed"})
1429
1563
 
1430
1564
  except asyncio.CancelledError:
@@ -1438,6 +1572,18 @@ async def annotate_from_image_stream(
1438
1572
  "error_type": "timeout",
1439
1573
  },
1440
1574
  )
1575
+ # Collect telemetry on error
1576
+ try:
1577
+ await _collect_stream_telemetry(
1578
+ request=request,
1579
+ req=req,
1580
+ current_state=current_state,
1581
+ start_time=start_time,
1582
+ source="api-image-stream",
1583
+ description=image_description or "image-annotation-failed",
1584
+ )
1585
+ except Exception:
1586
+ logging.debug("Telemetry collection failed on image timeout", exc_info=True)
1441
1587
  yield send_event("done", {"message": "Workflow ended with error"})
1442
1588
  except RateLimitError:
1443
1589
  logging.exception("Streaming image workflow rate limit")
@@ -1448,6 +1594,18 @@ async def annotate_from_image_stream(
1448
1594
  "error_type": "rate_limit",
1449
1595
  },
1450
1596
  )
1597
+ # Collect telemetry on error
1598
+ try:
1599
+ await _collect_stream_telemetry(
1600
+ request=request,
1601
+ req=req,
1602
+ current_state=current_state,
1603
+ start_time=start_time,
1604
+ source="api-image-stream",
1605
+ description=image_description or "image-annotation-failed",
1606
+ )
1607
+ except Exception:
1608
+ logging.debug("Telemetry collection failed on image rate limit", exc_info=True)
1451
1609
  yield send_event("done", {"message": "Workflow ended with error"})
1452
1610
  except Exception:
1453
1611
  logging.exception("Streaming image annotation workflow error")
@@ -1458,6 +1616,18 @@ async def annotate_from_image_stream(
1458
1616
  "error_type": "internal",
1459
1617
  },
1460
1618
  )
1619
+ # Collect telemetry on error
1620
+ try:
1621
+ await _collect_stream_telemetry(
1622
+ request=request,
1623
+ req=req,
1624
+ current_state=current_state,
1625
+ start_time=start_time,
1626
+ source="api-image-stream",
1627
+ description=image_description or "image-annotation-failed",
1628
+ )
1629
+ except Exception:
1630
+ logging.debug("Telemetry collection failed on image error", exc_info=True)
1461
1631
  yield send_event("done", {"message": "Workflow ended with error"})
1462
1632
 
1463
1633
  return StreamingResponse(
@@ -55,6 +55,10 @@ class AnnotationRequest(BaseModel):
55
55
  le=1.0,
56
56
  examples=[0.1, 0.3, 0.7],
57
57
  )
58
+ no_extend: bool = Field(
59
+ default=False,
60
+ description="If True, prohibit tag extensions (use only existing HED vocabulary)",
61
+ )
58
62
  telemetry_enabled: bool = Field(
59
63
  default=True,
60
64
  description="Allow telemetry collection for this request",
@@ -187,6 +191,10 @@ class ImageAnnotationRequest(BaseModel):
187
191
  le=1.0,
188
192
  examples=[0.1, 0.3, 0.7],
189
193
  )
194
+ no_extend: bool = Field(
195
+ default=False,
196
+ description="If True, prohibit tag extensions (use only existing HED vocabulary)",
197
+ )
190
198
  telemetry_enabled: bool = Field(
191
199
  default=True,
192
200
  description="Allow telemetry collection for this request",
@@ -87,6 +87,7 @@ class APIExecutionBackend(ExecutionBackend):
87
87
  schema_version: str = "8.4.0",
88
88
  max_validation_attempts: int = 5,
89
89
  run_assessment: bool = False,
90
+ no_extend: bool = False,
90
91
  **kwargs: Any,
91
92
  ) -> dict[str, Any]:
92
93
  """Generate HED annotation via API."""
@@ -96,6 +97,7 @@ class APIExecutionBackend(ExecutionBackend):
96
97
  schema_version=schema_version,
97
98
  max_validation_attempts=max_validation_attempts,
98
99
  run_assessment=run_assessment,
100
+ no_extend=no_extend,
99
101
  )
100
102
  except APIError as e:
101
103
  raise ExecutionError(
@@ -110,6 +112,7 @@ class APIExecutionBackend(ExecutionBackend):
110
112
  schema_version: str = "8.4.0",
111
113
  max_validation_attempts: int = 5,
112
114
  run_assessment: bool = False,
115
+ no_extend: bool = False,
113
116
  **kwargs: Any,
114
117
  ) -> Generator[tuple[str, dict[str, Any]], None, None]:
115
118
  """Generate HED annotation with streaming progress via API.
@@ -122,6 +125,7 @@ class APIExecutionBackend(ExecutionBackend):
122
125
  schema_version=schema_version,
123
126
  max_validation_attempts=max_validation_attempts,
124
127
  run_assessment=run_assessment,
128
+ no_extend=no_extend,
125
129
  )
126
130
  except APIError as e:
127
131
  raise ExecutionError(
@@ -137,6 +141,7 @@ class APIExecutionBackend(ExecutionBackend):
137
141
  schema_version: str = "8.4.0",
138
142
  max_validation_attempts: int = 5,
139
143
  run_assessment: bool = False,
144
+ no_extend: bool = False,
140
145
  **kwargs: Any,
141
146
  ) -> dict[str, Any]:
142
147
  """Generate HED annotation from image via API."""
@@ -147,6 +152,7 @@ class APIExecutionBackend(ExecutionBackend):
147
152
  schema_version=schema_version,
148
153
  max_validation_attempts=max_validation_attempts,
149
154
  run_assessment=run_assessment,
155
+ no_extend=no_extend,
150
156
  )
151
157
  except APIError as e:
152
158
  raise ExecutionError(
@@ -162,6 +168,7 @@ class APIExecutionBackend(ExecutionBackend):
162
168
  schema_version: str = "8.4.0",
163
169
  max_validation_attempts: int = 5,
164
170
  run_assessment: bool = False,
171
+ no_extend: bool = False,
165
172
  **kwargs: Any,
166
173
  ) -> Generator[tuple[str, dict[str, Any]], None, None]:
167
174
  """Generate HED annotation from image with streaming progress via API.
@@ -175,6 +182,7 @@ class APIExecutionBackend(ExecutionBackend):
175
182
  schema_version=schema_version,
176
183
  max_validation_attempts=max_validation_attempts,
177
184
  run_assessment=run_assessment,
185
+ no_extend=no_extend,
178
186
  )
179
187
  except APIError as e:
180
188
  raise ExecutionError(