python-alfresco-mcp-server 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- alfresco_mcp_server/__init__.py +27 -0
- alfresco_mcp_server/config.py +104 -0
- alfresco_mcp_server/fastmcp_server.py +259 -0
- alfresco_mcp_server/prompts/__init__.py +7 -0
- alfresco_mcp_server/prompts/search_and_analyze.py +67 -0
- alfresco_mcp_server/resources/__init__.py +7 -0
- alfresco_mcp_server/resources/repository_resources.py +205 -0
- alfresco_mcp_server/tools/__init__.py +13 -0
- alfresco_mcp_server/tools/core/__init__.py +27 -0
- alfresco_mcp_server/tools/core/browse_repository.py +164 -0
- alfresco_mcp_server/tools/core/cancel_checkout.py +160 -0
- alfresco_mcp_server/tools/core/checkin_document.py +264 -0
- alfresco_mcp_server/tools/core/checkout_document.py +258 -0
- alfresco_mcp_server/tools/core/create_folder.py +105 -0
- alfresco_mcp_server/tools/core/delete_node.py +88 -0
- alfresco_mcp_server/tools/core/download_document.py +214 -0
- alfresco_mcp_server/tools/core/get_node_properties.py +195 -0
- alfresco_mcp_server/tools/core/update_node_properties.py +138 -0
- alfresco_mcp_server/tools/core/upload_document.py +244 -0
- alfresco_mcp_server/tools/search/__init__.py +15 -0
- alfresco_mcp_server/tools/search/advanced_search.py +204 -0
- alfresco_mcp_server/tools/search/cmis_search.py +180 -0
- alfresco_mcp_server/tools/search/search_by_metadata.py +183 -0
- alfresco_mcp_server/tools/search/search_content.py +186 -0
- alfresco_mcp_server/utils/__init__.py +34 -0
- alfresco_mcp_server/utils/connection.py +114 -0
- alfresco_mcp_server/utils/file_type_analysis.py +163 -0
- alfresco_mcp_server/utils/json_utils.py +135 -0
- python_alfresco_mcp_server-1.1.0.dist-info/METADATA +631 -0
- python_alfresco_mcp_server-1.1.0.dist-info/RECORD +34 -0
- python_alfresco_mcp_server-1.1.0.dist-info/WHEEL +5 -0
- python_alfresco_mcp_server-1.1.0.dist-info/entry_points.txt +2 -0
- python_alfresco_mcp_server-1.1.0.dist-info/licenses/LICENSE +201 -0
- python_alfresco_mcp_server-1.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Content search tool for Alfresco MCP Server.
|
|
3
|
+
Each tool is self-contained with its own validation, business logic, and env handling.
|
|
4
|
+
"""
|
|
5
|
+
import logging
|
|
6
|
+
from typing import Optional
|
|
7
|
+
from fastmcp import Context
|
|
8
|
+
|
|
9
|
+
from ...utils.connection import ensure_connection
|
|
10
|
+
from ...utils.json_utils import safe_format_output
|
|
11
|
+
|
|
12
|
+
logger = logging.getLogger(__name__)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
async def search_content_impl(
|
|
16
|
+
search_query: str,
|
|
17
|
+
max_results: int = 25,
|
|
18
|
+
node_type: str = "cm:content",
|
|
19
|
+
ctx: Optional[Context] = None
|
|
20
|
+
) -> str:
|
|
21
|
+
"""Search for content in Alfresco repository.
|
|
22
|
+
|
|
23
|
+
Args:
|
|
24
|
+
search_query: Search query string
|
|
25
|
+
max_results: Maximum number of results to return (default: 25)
|
|
26
|
+
node_type: Type of nodes to search for (default: "cm:content" - searches documents)
|
|
27
|
+
ctx: MCP context for progress reporting
|
|
28
|
+
|
|
29
|
+
Returns:
|
|
30
|
+
Formatted search results
|
|
31
|
+
"""
|
|
32
|
+
# Parameter validation and extraction
|
|
33
|
+
try:
|
|
34
|
+
# Extract parameters with fallback handling
|
|
35
|
+
if hasattr(search_query, 'value'):
|
|
36
|
+
actual_query = str(search_query.value)
|
|
37
|
+
else:
|
|
38
|
+
actual_query = str(search_query)
|
|
39
|
+
|
|
40
|
+
if hasattr(max_results, 'value'):
|
|
41
|
+
actual_max_results = int(max_results.value)
|
|
42
|
+
else:
|
|
43
|
+
actual_max_results = int(max_results)
|
|
44
|
+
|
|
45
|
+
if hasattr(node_type, 'value'):
|
|
46
|
+
actual_node_type = str(node_type.value)
|
|
47
|
+
else:
|
|
48
|
+
actual_node_type = str(node_type)
|
|
49
|
+
|
|
50
|
+
# Default to cm:content if empty
|
|
51
|
+
if not actual_node_type.strip():
|
|
52
|
+
actual_node_type = "cm:content"
|
|
53
|
+
|
|
54
|
+
# Clean and normalize for display (prevent Unicode encoding issues)
|
|
55
|
+
safe_query_display = safe_format_output(str(actual_query))
|
|
56
|
+
safe_node_type_display = safe_format_output(str(actual_node_type))
|
|
57
|
+
|
|
58
|
+
except Exception as e:
|
|
59
|
+
logger.error(f"Parameter extraction error: {e}")
|
|
60
|
+
return safe_format_output(f"ERROR: Parameter error: {str(e)}")
|
|
61
|
+
|
|
62
|
+
if not actual_query.strip():
|
|
63
|
+
return """Content Search Tool
|
|
64
|
+
|
|
65
|
+
Usage: Provide a search query to search Alfresco repository content.
|
|
66
|
+
|
|
67
|
+
Example searches:
|
|
68
|
+
- admin (finds items with 'admin' in name or content)
|
|
69
|
+
- name:test* (finds items with names starting with 'test')
|
|
70
|
+
- modified:[2024-01-01 TO 2024-12-31] (finds items modified in 2024)
|
|
71
|
+
- TYPE:"cm:content" (finds all documents)
|
|
72
|
+
- TYPE:"cm:folder" (finds all folders)
|
|
73
|
+
|
|
74
|
+
Search uses AFTS (Alfresco Full Text Search) syntax for flexible content discovery.
|
|
75
|
+
By default, searches for documents (cm:content) unless a different type is specified.
|
|
76
|
+
"""
|
|
77
|
+
|
|
78
|
+
if ctx:
|
|
79
|
+
await ctx.info(safe_format_output(f"Content search for: '{safe_query_display}'"))
|
|
80
|
+
await ctx.report_progress(0.0)
|
|
81
|
+
|
|
82
|
+
try:
|
|
83
|
+
# Get all clients that ensure_connection() already created
|
|
84
|
+
master_client = await ensure_connection()
|
|
85
|
+
|
|
86
|
+
# Import search_utils
|
|
87
|
+
from python_alfresco_api.utils import search_utils
|
|
88
|
+
|
|
89
|
+
# Access the search client that was already created
|
|
90
|
+
search_client = master_client.search
|
|
91
|
+
|
|
92
|
+
logger.info(f"Content search for: '{safe_query_display}', type: '{safe_node_type_display}'")
|
|
93
|
+
|
|
94
|
+
if ctx:
|
|
95
|
+
await ctx.report_progress(0.3)
|
|
96
|
+
|
|
97
|
+
# Build search query to include node_type filter
|
|
98
|
+
final_query = actual_query
|
|
99
|
+
|
|
100
|
+
# Add node_type filter if not already in query
|
|
101
|
+
has_type_in_query = "TYPE:" in final_query.upper()
|
|
102
|
+
if not has_type_in_query:
|
|
103
|
+
if final_query == "*":
|
|
104
|
+
final_query = f'TYPE:"{actual_node_type}"'
|
|
105
|
+
else:
|
|
106
|
+
final_query = f'({final_query}) AND TYPE:"{actual_node_type}"'
|
|
107
|
+
|
|
108
|
+
# Use the correct working pattern: search_utils.simple_search with existing search_client
|
|
109
|
+
try:
|
|
110
|
+
search_results = search_utils.simple_search(search_client, final_query, max_items=actual_max_results)
|
|
111
|
+
|
|
112
|
+
if search_results and hasattr(search_results, 'list_'):
|
|
113
|
+
entries_list = search_results.list_.entries if search_results.list_ else []
|
|
114
|
+
logger.info(f"Found {len(entries_list)} content search results")
|
|
115
|
+
|
|
116
|
+
if ctx:
|
|
117
|
+
await ctx.report_progress(1.0)
|
|
118
|
+
|
|
119
|
+
if not entries_list:
|
|
120
|
+
return "0"
|
|
121
|
+
|
|
122
|
+
result_text = f"Found {len(entries_list)} item(s) matching the search query:\n\n"
|
|
123
|
+
|
|
124
|
+
for i, entry in enumerate(entries_list, 1):
|
|
125
|
+
# Debug: Log the entry structure
|
|
126
|
+
logger.debug(f"Entry {i} type: {type(entry)}, content: {entry}")
|
|
127
|
+
|
|
128
|
+
# Handle different possible entry structures
|
|
129
|
+
node = None
|
|
130
|
+
if isinstance(entry, dict):
|
|
131
|
+
if 'entry' in entry:
|
|
132
|
+
node = entry['entry']
|
|
133
|
+
elif 'name' in entry: # Direct node structure
|
|
134
|
+
node = entry
|
|
135
|
+
else:
|
|
136
|
+
logger.warning(f"Unknown entry structure: {entry}")
|
|
137
|
+
continue
|
|
138
|
+
elif hasattr(entry, 'entry'): # ResultSetRowEntry object
|
|
139
|
+
node = entry.entry
|
|
140
|
+
else:
|
|
141
|
+
logger.warning(f"Entry is not a dict or ResultSetRowEntry: {type(entry)}")
|
|
142
|
+
continue
|
|
143
|
+
|
|
144
|
+
if node:
|
|
145
|
+
# Handle both dict and ResultNode objects
|
|
146
|
+
if isinstance(node, dict):
|
|
147
|
+
name = str(node.get('name', 'Unknown'))
|
|
148
|
+
node_id = str(node.get('id', 'Unknown'))
|
|
149
|
+
node_type_actual = str(node.get('nodeType', 'Unknown'))
|
|
150
|
+
created_at = str(node.get('createdAt', 'Unknown'))
|
|
151
|
+
else:
|
|
152
|
+
# ResultNode object - access attributes directly
|
|
153
|
+
name = str(getattr(node, 'name', 'Unknown'))
|
|
154
|
+
node_id = str(getattr(node, 'id', 'Unknown'))
|
|
155
|
+
node_type_actual = str(getattr(node, 'node_type', 'Unknown'))
|
|
156
|
+
created_at = str(getattr(node, 'created_at', 'Unknown'))
|
|
157
|
+
|
|
158
|
+
# Clean JSON-friendly formatting (no markdown syntax)
|
|
159
|
+
# Apply safe formatting to individual fields to prevent emoji encoding issues
|
|
160
|
+
safe_name = safe_format_output(name)
|
|
161
|
+
safe_node_id = safe_format_output(node_id)
|
|
162
|
+
safe_node_type = safe_format_output(node_type_actual)
|
|
163
|
+
safe_created_at = safe_format_output(created_at)
|
|
164
|
+
|
|
165
|
+
result_text += f"{i}. {safe_name}\n"
|
|
166
|
+
result_text += f" - ID: {safe_node_id}\n"
|
|
167
|
+
result_text += f" - Type: {safe_node_type}\n"
|
|
168
|
+
result_text += f" - Created: {safe_created_at}\n\n"
|
|
169
|
+
|
|
170
|
+
return safe_format_output(result_text)
|
|
171
|
+
else:
|
|
172
|
+
return safe_format_output(f"ERROR: Content search failed - invalid response from Alfresco")
|
|
173
|
+
|
|
174
|
+
except Exception as e:
|
|
175
|
+
logger.error(f"Content search failed: {e}")
|
|
176
|
+
return safe_format_output(f"ERROR: Content search failed: {str(e)}")
|
|
177
|
+
|
|
178
|
+
except Exception as e:
|
|
179
|
+
# Preserve Unicode characters in error messages
|
|
180
|
+
error_msg = f"ERROR: Content search failed: {str(e)}"
|
|
181
|
+
if ctx:
|
|
182
|
+
await ctx.error(safe_format_output(error_msg))
|
|
183
|
+
return safe_format_output(error_msg)
|
|
184
|
+
|
|
185
|
+
if ctx:
|
|
186
|
+
await ctx.info(safe_format_output("Content search completed!"))
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# Utilities module for Alfresco MCP Server
|
|
2
|
+
|
|
3
|
+
from .connection import (
|
|
4
|
+
get_alfresco_config,
|
|
5
|
+
get_connection,
|
|
6
|
+
get_search_utils,
|
|
7
|
+
get_node_utils,
|
|
8
|
+
)
|
|
9
|
+
|
|
10
|
+
from .file_type_analysis import (
|
|
11
|
+
detect_file_extension_from_content,
|
|
12
|
+
analyze_content_type,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
from .json_utils import (
|
|
16
|
+
make_json_safe,
|
|
17
|
+
safe_format_output,
|
|
18
|
+
escape_unicode_for_json,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
__all__ = [
|
|
22
|
+
# Connection utilities
|
|
23
|
+
"get_alfresco_config",
|
|
24
|
+
"get_connection",
|
|
25
|
+
"get_search_utils",
|
|
26
|
+
"get_node_utils",
|
|
27
|
+
# File type analysis
|
|
28
|
+
"detect_file_extension_from_content",
|
|
29
|
+
"analyze_content_type",
|
|
30
|
+
# JSON utilities
|
|
31
|
+
"make_json_safe",
|
|
32
|
+
"safe_format_output",
|
|
33
|
+
"escape_unicode_for_json",
|
|
34
|
+
]
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Connection utilities for Alfresco MCP Server.
|
|
3
|
+
Handles client creation and connection management.
|
|
4
|
+
"""
|
|
5
|
+
import logging
|
|
6
|
+
import os
|
|
7
|
+
from typing import Optional
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
logger = logging.getLogger(__name__)
|
|
11
|
+
|
|
12
|
+
# Global connection cache
|
|
13
|
+
_master_client = None
|
|
14
|
+
_client_factory = None
|
|
15
|
+
|
|
16
|
+
def get_alfresco_config() -> dict:
|
|
17
|
+
"""Get Alfresco configuration from environment variables."""
|
|
18
|
+
return {
|
|
19
|
+
'alfresco_url': os.getenv('ALFRESCO_URL', 'http://localhost:8080'),
|
|
20
|
+
'username': os.getenv('ALFRESCO_USERNAME', 'admin'),
|
|
21
|
+
'password': os.getenv('ALFRESCO_PASSWORD', 'admin'),
|
|
22
|
+
'verify_ssl': os.getenv('ALFRESCO_VERIFY_SSL', 'false').lower() == 'true',
|
|
23
|
+
'timeout': int(os.getenv('ALFRESCO_TIMEOUT', '30'))
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
async def ensure_connection():
|
|
28
|
+
"""Ensure we have a working connection to Alfresco using python-alfresco-api."""
|
|
29
|
+
global _master_client, _client_factory
|
|
30
|
+
|
|
31
|
+
if _master_client is None:
|
|
32
|
+
try:
|
|
33
|
+
# Import here to avoid circular imports
|
|
34
|
+
from python_alfresco_api import ClientFactory
|
|
35
|
+
|
|
36
|
+
config = get_alfresco_config()
|
|
37
|
+
|
|
38
|
+
logger.info(">> Creating Alfresco clients...")
|
|
39
|
+
|
|
40
|
+
# Use ClientFactory to create authenticated client (original Sunday pattern)
|
|
41
|
+
factory = ClientFactory(
|
|
42
|
+
base_url=config['alfresco_url'],
|
|
43
|
+
username=config['username'],
|
|
44
|
+
password=config['password'],
|
|
45
|
+
verify_ssl=config['verify_ssl'],
|
|
46
|
+
timeout=config['timeout']
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
# Store the factory globally for other functions to use
|
|
50
|
+
_client_factory = factory
|
|
51
|
+
|
|
52
|
+
_master_client = factory.create_master_client()
|
|
53
|
+
logger.info("Master client created successfully")
|
|
54
|
+
|
|
55
|
+
# Test connection - use method that initializes and gets
|
|
56
|
+
try:
|
|
57
|
+
# Use ensure_httpx_client to initialize, then test simple call
|
|
58
|
+
_master_client.core.ensure_httpx_client()
|
|
59
|
+
logger.info("Connection test successful!")
|
|
60
|
+
except Exception as conn_error:
|
|
61
|
+
logger.warning(f"Connection test failed: {conn_error}")
|
|
62
|
+
|
|
63
|
+
except Exception as e:
|
|
64
|
+
logger.error(f"ERROR: Failed to create clients: {str(e)}")
|
|
65
|
+
raise e
|
|
66
|
+
|
|
67
|
+
return _master_client
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def get_connection():
|
|
71
|
+
"""Get the cached connection without async (for sync operations)."""
|
|
72
|
+
return _master_client
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
async def get_search_client():
|
|
76
|
+
"""Get the search client for search operations (using master_client for auth compatibility)."""
|
|
77
|
+
master_client = await ensure_connection()
|
|
78
|
+
# Return master_client which has simple_search access and working authentication
|
|
79
|
+
return master_client
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
async def get_core_client():
|
|
83
|
+
"""Get the core client for core operations (using master_client for auth compatibility)."""
|
|
84
|
+
master_client = await ensure_connection()
|
|
85
|
+
# Return the actual core client that has nodes, folders, etc.
|
|
86
|
+
return master_client.core
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
async def get_client_factory():
|
|
90
|
+
"""Get the client factory for advanced operations."""
|
|
91
|
+
await ensure_connection()
|
|
92
|
+
if not _client_factory:
|
|
93
|
+
raise RuntimeError("Connection not initialized. Call ensure_connection() first.")
|
|
94
|
+
return _client_factory
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def get_search_utils():
|
|
98
|
+
"""Get the search_utils module from python-alfresco-api."""
|
|
99
|
+
try:
|
|
100
|
+
from python_alfresco_api.utils import search_utils
|
|
101
|
+
return search_utils
|
|
102
|
+
except ImportError as e:
|
|
103
|
+
logger.error(f"Failed to import search_utils: {e}")
|
|
104
|
+
raise
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def get_node_utils():
|
|
108
|
+
"""Get the node_utils module from python-alfresco-api."""
|
|
109
|
+
try:
|
|
110
|
+
from python_alfresco_api.utils import node_utils
|
|
111
|
+
return node_utils
|
|
112
|
+
except ImportError as e:
|
|
113
|
+
logger.error(f"Failed to import node_utils: {e}")
|
|
114
|
+
raise
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
"""
|
|
2
|
+
File type analysis utility for Alfresco MCP Server.
|
|
3
|
+
Provides content type analysis and suggestions for different file types.
|
|
4
|
+
"""
|
|
5
|
+
import pathlib
|
|
6
|
+
import mimetypes
|
|
7
|
+
from typing import Dict, List, Optional
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def detect_file_extension_from_content(content: bytes) -> Optional[str]:
|
|
11
|
+
"""Detect file type from content and return appropriate extension.
|
|
12
|
+
|
|
13
|
+
Args:
|
|
14
|
+
content: Raw file content bytes
|
|
15
|
+
|
|
16
|
+
Returns:
|
|
17
|
+
File extension (e.g., '.pdf', '.txt', '.jpg') or None if unknown
|
|
18
|
+
"""
|
|
19
|
+
if not content:
|
|
20
|
+
return None
|
|
21
|
+
|
|
22
|
+
# Check for common file signatures (magic numbers)
|
|
23
|
+
if content.startswith(b'%PDF'):
|
|
24
|
+
return '.pdf'
|
|
25
|
+
elif content.startswith(b'\xff\xd8\xff'):
|
|
26
|
+
return '.jpg'
|
|
27
|
+
elif content.startswith(b'\x89PNG\r\n\x1a\n'):
|
|
28
|
+
return '.png'
|
|
29
|
+
elif content.startswith(b'GIF87a') or content.startswith(b'GIF89a'):
|
|
30
|
+
return '.gif'
|
|
31
|
+
elif content.startswith(b'PK\x03\x04') or content.startswith(b'PK\x05\x06') or content.startswith(b'PK\x07\x08'):
|
|
32
|
+
# ZIP-based formats (could be .zip, .docx, .xlsx, .pptx, etc.)
|
|
33
|
+
# For simplicity, assume .zip unless we detect Office formats
|
|
34
|
+
if b'word/' in content[:1024] or b'xl/' in content[:1024] or b'ppt/' in content[:1024]:
|
|
35
|
+
# Likely Office document but hard to determine exact type
|
|
36
|
+
return '.zip' # Conservative choice
|
|
37
|
+
return '.zip'
|
|
38
|
+
elif content.startswith(b'\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1'):
|
|
39
|
+
# Microsoft Office old format
|
|
40
|
+
return '.doc' # Could also be .xls, .ppt but .doc is most common
|
|
41
|
+
elif content.startswith(b'<?xml') or content.startswith(b'\xef\xbb\xbf<?xml'):
|
|
42
|
+
return '.xml'
|
|
43
|
+
elif content.startswith(b'<!DOCTYPE html') or content.startswith(b'<html'):
|
|
44
|
+
return '.html'
|
|
45
|
+
elif content.startswith(b'{\n') or content.startswith(b'{"') or content.startswith(b'[\n') or content.startswith(b'[{'):
|
|
46
|
+
return '.json'
|
|
47
|
+
else:
|
|
48
|
+
# Try to detect if it's text content
|
|
49
|
+
try:
|
|
50
|
+
# Try to decode as UTF-8 text
|
|
51
|
+
text_content = content.decode('utf-8')
|
|
52
|
+
# Check if it contains mostly printable characters
|
|
53
|
+
printable_chars = sum(1 for c in text_content if c.isprintable() or c.isspace())
|
|
54
|
+
if len(text_content) > 0 and printable_chars / len(text_content) > 0.8:
|
|
55
|
+
return '.txt'
|
|
56
|
+
except (UnicodeDecodeError, UnicodeError):
|
|
57
|
+
pass
|
|
58
|
+
|
|
59
|
+
# Unknown binary format
|
|
60
|
+
return None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def analyze_content_type(filename: str, mime_type: str, content: bytes) -> dict:
|
|
64
|
+
"""Analyze file type and provide relevant suggestions.
|
|
65
|
+
|
|
66
|
+
Args:
|
|
67
|
+
filename: Name of the file
|
|
68
|
+
mime_type: MIME type of the file
|
|
69
|
+
content: File content as bytes
|
|
70
|
+
|
|
71
|
+
Returns:
|
|
72
|
+
Dictionary with category, suggestions, and file_size
|
|
73
|
+
"""
|
|
74
|
+
file_size = len(content)
|
|
75
|
+
|
|
76
|
+
# Get case-insensitive filename for macOS/Windows compatibility
|
|
77
|
+
filename_lower = filename.lower()
|
|
78
|
+
|
|
79
|
+
# Determine file category based on MIME type and extension
|
|
80
|
+
if mime_type.startswith('image/'):
|
|
81
|
+
category = 'images'
|
|
82
|
+
suggestions = [
|
|
83
|
+
"Can be used for thumbnails and previews",
|
|
84
|
+
"Consider image optimization for web use"
|
|
85
|
+
]
|
|
86
|
+
elif mime_type.startswith('video/') or mime_type.startswith('audio/'):
|
|
87
|
+
category = 'media'
|
|
88
|
+
suggestions = [
|
|
89
|
+
"Large files may need streaming support",
|
|
90
|
+
"Consider format compatibility for playback"
|
|
91
|
+
]
|
|
92
|
+
elif mime_type in ['application/pdf', 'application/msword',
|
|
93
|
+
'application/vnd.openxmlformats-officedocument.wordprocessingml.document']:
|
|
94
|
+
category = 'documents'
|
|
95
|
+
suggestions = [
|
|
96
|
+
"PDF files support full-text search",
|
|
97
|
+
">> Consider using text extraction for searchable content",
|
|
98
|
+
"Can be previewed in most browsers"
|
|
99
|
+
]
|
|
100
|
+
elif mime_type in ['application/vnd.ms-excel',
|
|
101
|
+
'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet']:
|
|
102
|
+
category = 'documents'
|
|
103
|
+
suggestions = [
|
|
104
|
+
">> Excel spreadsheet - can be opened with Excel or LibreOffice Calc",
|
|
105
|
+
">> May contain tracked changes or comments"
|
|
106
|
+
]
|
|
107
|
+
elif mime_type in ['application/vnd.ms-powerpoint',
|
|
108
|
+
'application/vnd.openxmlformats-officedocument.presentationml.presentation']:
|
|
109
|
+
category = 'documents'
|
|
110
|
+
suggestions = [
|
|
111
|
+
"Presentation files support slide previews",
|
|
112
|
+
"Consider extracting slide content for search"
|
|
113
|
+
]
|
|
114
|
+
elif mime_type.startswith('text/') or 'javascript' in mime_type or 'json' in mime_type:
|
|
115
|
+
if filename_lower.endswith(('.py', '.js', '.java', '.cpp', '.c', '.cs', '.php', '.rb')):
|
|
116
|
+
category = 'code'
|
|
117
|
+
suggestions = [
|
|
118
|
+
"Source code files support syntax highlighting",
|
|
119
|
+
">> Check contents before extraction for security",
|
|
120
|
+
"Can be indexed for code search"
|
|
121
|
+
]
|
|
122
|
+
else:
|
|
123
|
+
category = 'documents'
|
|
124
|
+
suggestions = [
|
|
125
|
+
">> Review for security before execution",
|
|
126
|
+
">> May require specific runtime environment"
|
|
127
|
+
]
|
|
128
|
+
elif mime_type in ['application/zip', 'application/x-rar-compressed', 'application/gzip']:
|
|
129
|
+
category = 'archives'
|
|
130
|
+
suggestions = [
|
|
131
|
+
"Archive contents can be extracted and indexed",
|
|
132
|
+
">> Check contents before extraction for security",
|
|
133
|
+
"May contain multiple file types"
|
|
134
|
+
]
|
|
135
|
+
else:
|
|
136
|
+
category = 'other'
|
|
137
|
+
suggestions = [
|
|
138
|
+
"Unknown file type - review content manually",
|
|
139
|
+
"Consider file format documentation"
|
|
140
|
+
]
|
|
141
|
+
|
|
142
|
+
# Add size-based suggestions
|
|
143
|
+
if file_size > 100 * 1024 * 1024: # > 100MB
|
|
144
|
+
suggestions.append("WARNING: Large file - consider network and storage impact")
|
|
145
|
+
|
|
146
|
+
# Add security suggestions for executable files (case-insensitive for cross-platform)
|
|
147
|
+
executable_extensions = (
|
|
148
|
+
'.exe', '.bat', '.sh', '.com', '.scr', # Windows & shell scripts
|
|
149
|
+
'.app', '.dmg', '.pkg', # macOS
|
|
150
|
+
'.deb', '.rpm', '.run', '.bin', '.appimage' # Linux
|
|
151
|
+
)
|
|
152
|
+
if filename_lower.endswith(executable_extensions):
|
|
153
|
+
suggestions = [
|
|
154
|
+
"WARNING: Executable file - scan for security before running",
|
|
155
|
+
"Consider sandboxed execution environment"
|
|
156
|
+
]
|
|
157
|
+
category = 'executable'
|
|
158
|
+
|
|
159
|
+
return {
|
|
160
|
+
'category': category,
|
|
161
|
+
'suggestions': suggestions,
|
|
162
|
+
'file_size': file_size
|
|
163
|
+
}
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""
|
|
2
|
+
JSON utilities for Alfresco MCP Server.
|
|
3
|
+
Handles proper Unicode emoji encoding for MCP protocol transport.
|
|
4
|
+
"""
|
|
5
|
+
import json
|
|
6
|
+
import logging
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
logger = logging.getLogger(__name__)
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def make_json_safe(text: str) -> str:
|
|
13
|
+
"""
|
|
14
|
+
Make text JSON-safe for MCP protocol transport.
|
|
15
|
+
Properly encodes Unicode emojis to prevent character map errors.
|
|
16
|
+
|
|
17
|
+
Args:
|
|
18
|
+
text: Input text that may contain Unicode emojis
|
|
19
|
+
|
|
20
|
+
Returns:
|
|
21
|
+
JSON-safe text with properly encoded Unicode characters
|
|
22
|
+
"""
|
|
23
|
+
if not text:
|
|
24
|
+
return text
|
|
25
|
+
|
|
26
|
+
try:
|
|
27
|
+
# Ensure proper Unicode normalization
|
|
28
|
+
import unicodedata
|
|
29
|
+
normalized = unicodedata.normalize('NFC', text)
|
|
30
|
+
|
|
31
|
+
# Test if it can be safely JSON serialized
|
|
32
|
+
json.dumps(normalized)
|
|
33
|
+
return normalized
|
|
34
|
+
|
|
35
|
+
except (UnicodeError, TypeError) as e:
|
|
36
|
+
logger.warning(f"Unicode encoding issue, falling back to ASCII: {e}")
|
|
37
|
+
# Fall back to ASCII-safe version with emoji descriptions
|
|
38
|
+
return text.encode('ascii', errors='ignore').decode('ascii')
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def safe_format_output(text: str) -> str:
|
|
42
|
+
"""
|
|
43
|
+
Format output text to be safe for MCP JSON transport.
|
|
44
|
+
Replaces emojis with text equivalents to prevent character map errors.
|
|
45
|
+
|
|
46
|
+
Args:
|
|
47
|
+
text: Text to format
|
|
48
|
+
|
|
49
|
+
Returns:
|
|
50
|
+
Safely formatted text with emojis replaced
|
|
51
|
+
"""
|
|
52
|
+
if not text:
|
|
53
|
+
return text
|
|
54
|
+
|
|
55
|
+
try:
|
|
56
|
+
# Define emoji replacements for common ones used in the tools
|
|
57
|
+
emoji_replacements = {
|
|
58
|
+
'๐': '[LINK]',
|
|
59
|
+
'๐': '[UNLOCKED]',
|
|
60
|
+
'๐': '[DOCUMENT]',
|
|
61
|
+
'๐': '[ID]',
|
|
62
|
+
'๐': '[SIZE]',
|
|
63
|
+
'๐พ': '[SAVED]',
|
|
64
|
+
'๐': '[LOCKED]',
|
|
65
|
+
'๐': '[TIME]',
|
|
66
|
+
'๐ฅ': '[DOWNLOAD]',
|
|
67
|
+
'โน๏ธ': '[INFO]',
|
|
68
|
+
'โ ๏ธ': '[WARNING]',
|
|
69
|
+
'๐ค': '[USER]',
|
|
70
|
+
'โ
': '[SUCCESS]',
|
|
71
|
+
'โ': '[ERROR]',
|
|
72
|
+
'๐ท๏ธ': '[TAG]',
|
|
73
|
+
'๐งฉ': '[MODULE]',
|
|
74
|
+
'๐': '[FOLDER]',
|
|
75
|
+
'๐': '[LOCATION]',
|
|
76
|
+
'๐
': '[DATE]',
|
|
77
|
+
'๐': '[NOTE]',
|
|
78
|
+
'๐ข': '[VERSION]',
|
|
79
|
+
'๐': '[SIZE]',
|
|
80
|
+
'๐๏ธ': '[DELETE]',
|
|
81
|
+
'๐': '[SEARCH]',
|
|
82
|
+
'๐ค': '[UPLOAD]',
|
|
83
|
+
'๐งน': '[CLEANUP]',
|
|
84
|
+
'๐ข': '[REPOSITORY]',
|
|
85
|
+
'๐ง': '[TOOL]',
|
|
86
|
+
'๐ฆ': '[PACKAGE]'
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
# Replace emojis with text equivalents
|
|
90
|
+
safe_text = text
|
|
91
|
+
for emoji, replacement in emoji_replacements.items():
|
|
92
|
+
safe_text = safe_text.replace(emoji, replacement)
|
|
93
|
+
|
|
94
|
+
# Test if the result is JSON-safe
|
|
95
|
+
test_json = json.dumps(safe_text, ensure_ascii=True)
|
|
96
|
+
json.loads(test_json)
|
|
97
|
+
|
|
98
|
+
return safe_text
|
|
99
|
+
|
|
100
|
+
except Exception as e:
|
|
101
|
+
logger.warning(f"JSON formatting issue: {e}")
|
|
102
|
+
try:
|
|
103
|
+
# Ultimate fallback: remove all non-ASCII characters
|
|
104
|
+
ascii_text = text.encode('ascii', errors='ignore').decode('ascii')
|
|
105
|
+
return ascii_text
|
|
106
|
+
except Exception as fallback_error:
|
|
107
|
+
logger.error(f"ASCII fallback failed: {fallback_error}")
|
|
108
|
+
return "Error: Text encoding failed"
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def escape_unicode_for_json(text: str) -> str:
|
|
112
|
+
"""
|
|
113
|
+
Alternative approach: explicitly escape Unicode characters for JSON.
|
|
114
|
+
Use this if the regular approach doesn't work.
|
|
115
|
+
|
|
116
|
+
Args:
|
|
117
|
+
text: Input text with Unicode characters
|
|
118
|
+
|
|
119
|
+
Returns:
|
|
120
|
+
Text with Unicode characters escaped for JSON
|
|
121
|
+
"""
|
|
122
|
+
if not text:
|
|
123
|
+
return text
|
|
124
|
+
|
|
125
|
+
try:
|
|
126
|
+
# Use json.dumps to properly escape Unicode, then remove the quotes
|
|
127
|
+
escaped = json.dumps(text, ensure_ascii=False)
|
|
128
|
+
# Remove the surrounding quotes added by json.dumps
|
|
129
|
+
if escaped.startswith('"') and escaped.endswith('"'):
|
|
130
|
+
escaped = escaped[1:-1]
|
|
131
|
+
return escaped
|
|
132
|
+
|
|
133
|
+
except Exception as e:
|
|
134
|
+
logger.warning(f"Unicode escaping failed: {e}")
|
|
135
|
+
return text
|