supernote 0.13.6__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- supernote/__init__.py +28 -0
- supernote/alembic/__init__.py +0 -0
- supernote/alembic/env.py +121 -0
- supernote/alembic/script.py.mako +28 -0
- supernote/alembic/versions/0543a383957b_initial_schema.py +312 -0
- supernote/alembic.ini +149 -0
- supernote/cli/__init__.py +1 -0
- supernote/cli/admin.py +178 -0
- supernote/cli/client.py +434 -0
- supernote/cli/main.py +61 -0
- supernote/cli/notebook.py +313 -0
- supernote/cli/server.py +89 -0
- supernote/client/__init__.py +28 -0
- supernote/client/admin.py +76 -0
- supernote/client/api.py +135 -0
- supernote/client/api_model.py +231 -0
- supernote/client/auth.py +84 -0
- supernote/client/client.py +365 -0
- supernote/client/device.py +361 -0
- supernote/client/exceptions.py +33 -0
- supernote/client/extended.py +20 -0
- supernote/client/hashing.py +50 -0
- supernote/client/login_client.py +154 -0
- supernote/client/schedule.py +149 -0
- supernote/client/summary.py +156 -0
- supernote/client/web.py +213 -0
- supernote/models/__init__.py +14 -0
- supernote/models/auth.py +380 -0
- supernote/models/base.py +266 -0
- supernote/models/equipment.py +276 -0
- supernote/models/extended.py +124 -0
- supernote/models/file_common.py +95 -0
- supernote/models/file_device.py +571 -0
- supernote/models/file_web.py +602 -0
- supernote/models/schedule.py +663 -0
- supernote/models/summary.py +830 -0
- supernote/models/system.py +431 -0
- supernote/models/user.py +638 -0
- supernote/notebook/__init__.py +67 -0
- supernote/notebook/color.py +93 -0
- supernote/notebook/converter.py +560 -0
- supernote/notebook/decoder.py +398 -0
- supernote/notebook/exceptions.py +43 -0
- supernote/notebook/fileformat.py +463 -0
- supernote/notebook/manipulator.py +420 -0
- supernote/notebook/parser.py +737 -0
- supernote/notebook/utils.py +49 -0
- supernote/py.typed +0 -0
- supernote/server/__init__.py +12 -0
- supernote/server/app.py +428 -0
- supernote/server/config.py +265 -0
- supernote/server/constants.py +26 -0
- supernote/server/db/__init__.py +9 -0
- supernote/server/db/base.py +8 -0
- supernote/server/db/migrations.py +44 -0
- supernote/server/db/models/__init__.py +23 -0
- supernote/server/db/models/device.py +17 -0
- supernote/server/db/models/file.py +107 -0
- supernote/server/db/models/kv.py +17 -0
- supernote/server/db/models/login_record.py +22 -0
- supernote/server/db/models/note_processing.py +100 -0
- supernote/server/db/models/schedule.py +83 -0
- supernote/server/db/models/summary.py +122 -0
- supernote/server/db/models/user.py +30 -0
- supernote/server/db/session.py +88 -0
- supernote/server/events.py +63 -0
- supernote/server/exceptions.py +153 -0
- supernote/server/mcp/__init__.py +1 -0
- supernote/server/mcp/models.py +109 -0
- supernote/server/mcp/server.py +176 -0
- supernote/server/resources/__init__.py +0 -0
- supernote/server/resources/prompts/__init__.py +0 -0
- supernote/server/resources/prompts/ocr/common/context.md +19 -0
- supernote/server/resources/prompts/ocr/common/legend.md +11 -0
- supernote/server/resources/prompts/ocr/daily/prompt.md +12 -0
- supernote/server/resources/prompts/ocr/default/system.md +14 -0
- supernote/server/resources/prompts/ocr/monthly/prompt.md +22 -0
- supernote/server/resources/prompts/ocr/weekly/prompt.md +16 -0
- supernote/server/resources/prompts/summary/common/instruction.md +8 -0
- supernote/server/resources/prompts/summary/daily/prompt.md +6 -0
- supernote/server/resources/prompts/summary/default/prompt.md +4 -0
- supernote/server/resources/prompts/summary/monthly/prompt.md +6 -0
- supernote/server/resources/prompts/summary/weekly/prompt.md +6 -0
- supernote/server/routes/admin.py +109 -0
- supernote/server/routes/auth.py +337 -0
- supernote/server/routes/decorators.py +13 -0
- supernote/server/routes/extended.py +139 -0
- supernote/server/routes/file_device.py +593 -0
- supernote/server/routes/file_web.py +597 -0
- supernote/server/routes/oss.py +317 -0
- supernote/server/routes/schedule.py +223 -0
- supernote/server/routes/summary.py +390 -0
- supernote/server/routes/system.py +57 -0
- supernote/server/services/__init__.py +15 -0
- supernote/server/services/blob.py +214 -0
- supernote/server/services/coordination.py +171 -0
- supernote/server/services/file.py +970 -0
- supernote/server/services/gemini.py +66 -0
- supernote/server/services/integrity.py +112 -0
- supernote/server/services/processor.py +331 -0
- supernote/server/services/processor_modules/__init__.py +145 -0
- supernote/server/services/processor_modules/gemini_embedding.py +112 -0
- supernote/server/services/processor_modules/gemini_ocr.py +149 -0
- supernote/server/services/processor_modules/page_hashing.py +220 -0
- supernote/server/services/processor_modules/png_conversion.py +107 -0
- supernote/server/services/processor_modules/summary.py +302 -0
- supernote/server/services/prompt_loader.py +133 -0
- supernote/server/services/schedule.py +161 -0
- supernote/server/services/search.py +250 -0
- supernote/server/services/summary.py +419 -0
- supernote/server/services/user.py +480 -0
- supernote/server/services/vfs.py +476 -0
- supernote/server/static/favicon.ico +0 -0
- supernote/server/static/index.html +186 -0
- supernote/server/static/js/api/client.js +463 -0
- supernote/server/static/js/components/FileCard.js +55 -0
- supernote/server/static/js/components/FileViewer.js +137 -0
- supernote/server/static/js/components/LoginCard.js +64 -0
- supernote/server/static/js/components/MoveModal.js +74 -0
- supernote/server/static/js/components/RenameModal.js +39 -0
- supernote/server/static/js/components/SummaryPanel.js +101 -0
- supernote/server/static/js/components/SystemPanel.js +158 -0
- supernote/server/static/js/composables/useFileSystem.js +116 -0
- supernote/server/static/js/main.js +241 -0
- supernote/server/static/style.css +14 -0
- supernote/server/utils/hashing.py +18 -0
- supernote/server/utils/note_content.py +76 -0
- supernote/server/utils/paths.py +58 -0
- supernote/server/utils/rate_limit.py +52 -0
- supernote/server/utils/tasks.py +71 -0
- supernote/server/utils/unique_id.py +48 -0
- supernote/server/utils/url_signer.py +218 -0
- supernote-0.13.6.dist-info/METADATA +183 -0
- supernote-0.13.6.dist-info/RECORD +138 -0
- supernote-0.13.6.dist-info/WHEEL +5 -0
- supernote-0.13.6.dist-info/entry_points.txt +3 -0
- supernote-0.13.6.dist-info/licenses/LICENSE +201 -0
- supernote-0.13.6.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import logging
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from supernote.server.config import ServerConfig
|
|
6
|
+
from supernote.server.db.session import DatabaseSessionManager
|
|
7
|
+
from supernote.server.services.file import FileService
|
|
8
|
+
from supernote.server.services.gemini import GeminiService
|
|
9
|
+
from supernote.server.services.processor_modules import ProcessorModule
|
|
10
|
+
from supernote.server.utils.note_content import get_page_content_by_id
|
|
11
|
+
|
|
12
|
+
logger = logging.getLogger(__name__)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class GeminiEmbeddingModule(ProcessorModule):
|
|
16
|
+
"""Module responsible for generating embeddings for note pages using Gemini."""
|
|
17
|
+
|
|
18
|
+
def __init__(
|
|
19
|
+
self,
|
|
20
|
+
file_service: FileService,
|
|
21
|
+
config: ServerConfig,
|
|
22
|
+
gemini_service: GeminiService,
|
|
23
|
+
) -> None:
|
|
24
|
+
self.file_service = file_service
|
|
25
|
+
self.config = config
|
|
26
|
+
self.gemini_service = gemini_service
|
|
27
|
+
|
|
28
|
+
@property
|
|
29
|
+
def name(self) -> str:
|
|
30
|
+
return "GeminiEmbeddingModule"
|
|
31
|
+
|
|
32
|
+
@property
|
|
33
|
+
def task_type(self) -> str:
|
|
34
|
+
return "EMBEDDING_GENERATION"
|
|
35
|
+
|
|
36
|
+
async def run_if_needed(
|
|
37
|
+
self,
|
|
38
|
+
file_id: int,
|
|
39
|
+
session_manager: DatabaseSessionManager,
|
|
40
|
+
page_index: Optional[int] = None,
|
|
41
|
+
page_id: Optional[str] = None,
|
|
42
|
+
) -> bool:
|
|
43
|
+
if not page_id:
|
|
44
|
+
return False
|
|
45
|
+
|
|
46
|
+
if not self.gemini_service.is_configured:
|
|
47
|
+
return False
|
|
48
|
+
|
|
49
|
+
if not await super().run_if_needed(
|
|
50
|
+
file_id, session_manager, page_index, page_id
|
|
51
|
+
):
|
|
52
|
+
return False
|
|
53
|
+
|
|
54
|
+
async with session_manager.session() as session:
|
|
55
|
+
# Check Prerequisites (Text Content must exist)
|
|
56
|
+
content = await get_page_content_by_id(session, file_id, page_id)
|
|
57
|
+
|
|
58
|
+
if not content or not content.text_content:
|
|
59
|
+
# Dependency not met yet (OCR not done or empty page)
|
|
60
|
+
return False
|
|
61
|
+
|
|
62
|
+
return True
|
|
63
|
+
|
|
64
|
+
async def process(
|
|
65
|
+
self,
|
|
66
|
+
file_id: int,
|
|
67
|
+
session_manager: DatabaseSessionManager,
|
|
68
|
+
page_index: Optional[int] = None,
|
|
69
|
+
page_id: Optional[str] = None,
|
|
70
|
+
**kwargs: object,
|
|
71
|
+
) -> None:
|
|
72
|
+
if not page_id:
|
|
73
|
+
return
|
|
74
|
+
|
|
75
|
+
# Get Text Content
|
|
76
|
+
text_content = ""
|
|
77
|
+
async with session_manager.session() as session:
|
|
78
|
+
content = await get_page_content_by_id(session, file_id, page_id)
|
|
79
|
+
|
|
80
|
+
if not content or not content.text_content:
|
|
81
|
+
logger.warning(
|
|
82
|
+
f"No text content found for embedding: file {file_id} page {page_id} (idx {page_index})"
|
|
83
|
+
)
|
|
84
|
+
return
|
|
85
|
+
text_content = content.text_content
|
|
86
|
+
|
|
87
|
+
# Call Gemini API
|
|
88
|
+
if not self.gemini_service.is_configured:
|
|
89
|
+
raise ValueError("Gemini API key not configured")
|
|
90
|
+
|
|
91
|
+
model_id = self.config.gemini_embedding_model
|
|
92
|
+
response = await self.gemini_service.embed_content(
|
|
93
|
+
model=model_id,
|
|
94
|
+
contents=text_content,
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
if not response.embeddings:
|
|
98
|
+
raise ValueError("No embeddings returned from Gemini API")
|
|
99
|
+
|
|
100
|
+
# Assuming single embedding for the whole text block for now
|
|
101
|
+
embedding_values = response.embeddings[0].values
|
|
102
|
+
embedding_json = json.dumps(embedding_values)
|
|
103
|
+
|
|
104
|
+
# Save Result
|
|
105
|
+
async with session_manager.session() as session:
|
|
106
|
+
content = await get_page_content_by_id(session, file_id, page_id)
|
|
107
|
+
|
|
108
|
+
if content:
|
|
109
|
+
content.embedding = embedding_json
|
|
110
|
+
await session.commit()
|
|
111
|
+
|
|
112
|
+
logger.info(f"Completed Embedding for file {file_id} page {page_index}")
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from google.genai import types
|
|
6
|
+
|
|
7
|
+
from supernote.server.config import ServerConfig
|
|
8
|
+
from supernote.server.constants import CACHE_BUCKET
|
|
9
|
+
from supernote.server.db.models.file import UserFileDO
|
|
10
|
+
from supernote.server.db.session import DatabaseSessionManager
|
|
11
|
+
from supernote.server.services.file import FileService
|
|
12
|
+
from supernote.server.services.gemini import GeminiService
|
|
13
|
+
from supernote.server.services.processor_modules import ProcessorModule
|
|
14
|
+
from supernote.server.services.prompt_loader import PROMPT_LOADER, PromptId
|
|
15
|
+
from supernote.server.utils.note_content import (
|
|
16
|
+
format_page_metadata,
|
|
17
|
+
get_page_content_by_id,
|
|
18
|
+
)
|
|
19
|
+
from supernote.server.utils.paths import get_page_png_path
|
|
20
|
+
|
|
21
|
+
logger = logging.getLogger(__name__)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class GeminiOcrModule(ProcessorModule):
|
|
25
|
+
"""Module responsible for extracting text from note pages using Gemini OCR."""
|
|
26
|
+
|
|
27
|
+
def __init__(
|
|
28
|
+
self,
|
|
29
|
+
file_service: FileService,
|
|
30
|
+
config: ServerConfig,
|
|
31
|
+
gemini_service: GeminiService,
|
|
32
|
+
) -> None:
|
|
33
|
+
self.file_service = file_service
|
|
34
|
+
self.config = config
|
|
35
|
+
self.gemini_service = gemini_service
|
|
36
|
+
|
|
37
|
+
@property
|
|
38
|
+
def name(self) -> str:
|
|
39
|
+
return "GeminiOcrModule"
|
|
40
|
+
|
|
41
|
+
@property
|
|
42
|
+
def task_type(self) -> str:
|
|
43
|
+
return "OCR_EXTRACTION"
|
|
44
|
+
|
|
45
|
+
async def run_if_needed(
|
|
46
|
+
self,
|
|
47
|
+
file_id: int,
|
|
48
|
+
session_manager: DatabaseSessionManager,
|
|
49
|
+
page_index: Optional[int] = None,
|
|
50
|
+
page_id: Optional[str] = None,
|
|
51
|
+
) -> bool:
|
|
52
|
+
if page_index is None:
|
|
53
|
+
return False
|
|
54
|
+
|
|
55
|
+
if not self.gemini_service.is_configured:
|
|
56
|
+
return False
|
|
57
|
+
|
|
58
|
+
if not await super().run_if_needed(
|
|
59
|
+
file_id, session_manager, page_index, page_id
|
|
60
|
+
):
|
|
61
|
+
return False
|
|
62
|
+
|
|
63
|
+
if not page_id:
|
|
64
|
+
return False
|
|
65
|
+
|
|
66
|
+
# Check if PNG exists (Prerequisite)
|
|
67
|
+
png_path = get_page_png_path(file_id, page_id)
|
|
68
|
+
if not await self.file_service.blob_storage.exists(CACHE_BUCKET, png_path):
|
|
69
|
+
logger.warning(
|
|
70
|
+
f"PNG prerequisite not met for OCR of {file_id} page {page_id}"
|
|
71
|
+
)
|
|
72
|
+
return False
|
|
73
|
+
|
|
74
|
+
return True
|
|
75
|
+
|
|
76
|
+
async def process(
|
|
77
|
+
self,
|
|
78
|
+
file_id: int,
|
|
79
|
+
session_manager: DatabaseSessionManager,
|
|
80
|
+
page_index: Optional[int] = None,
|
|
81
|
+
page_id: Optional[str] = None,
|
|
82
|
+
**kwargs: object,
|
|
83
|
+
) -> None:
|
|
84
|
+
if page_id is None:
|
|
85
|
+
logger.error(f"Page ID required for OCR processing of file {file_id}")
|
|
86
|
+
return
|
|
87
|
+
|
|
88
|
+
# Get PNG Content
|
|
89
|
+
png_path = get_page_png_path(file_id, page_id)
|
|
90
|
+
png_data = b""
|
|
91
|
+
async for chunk in self.file_service.blob_storage.get(CACHE_BUCKET, png_path):
|
|
92
|
+
png_data += chunk
|
|
93
|
+
|
|
94
|
+
# Call Gemini API
|
|
95
|
+
if not self.gemini_service.is_configured:
|
|
96
|
+
raise ValueError("Gemini API key not configured")
|
|
97
|
+
|
|
98
|
+
# Get File Info for custom prompt and metadata
|
|
99
|
+
file_name_basis: Optional[str] = None
|
|
100
|
+
file_name: Optional[str] = None
|
|
101
|
+
notebook_create_time: Optional[int] = None
|
|
102
|
+
async with session_manager.session() as session:
|
|
103
|
+
file_do = await session.get(UserFileDO, file_id)
|
|
104
|
+
if file_do:
|
|
105
|
+
file_name = file_do.file_name
|
|
106
|
+
notebook_create_time = file_do.create_time
|
|
107
|
+
file_name_basis = Path(file_name).stem.lower()
|
|
108
|
+
|
|
109
|
+
model_id = self.config.gemini_ocr_model
|
|
110
|
+
prompt = PROMPT_LOADER.get_prompt(
|
|
111
|
+
PromptId.OCR_TRANSCRIPTION, custom_type=file_name_basis
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
metadata_block = format_page_metadata(
|
|
115
|
+
page_index=page_index or 0,
|
|
116
|
+
page_id=page_id,
|
|
117
|
+
file_name=file_name,
|
|
118
|
+
notebook_create_time=notebook_create_time,
|
|
119
|
+
include_section_divider=True,
|
|
120
|
+
)
|
|
121
|
+
prompt = f"{metadata_block}\n\n{prompt}"
|
|
122
|
+
|
|
123
|
+
response = await self.gemini_service.generate_content(
|
|
124
|
+
model=model_id,
|
|
125
|
+
contents=[
|
|
126
|
+
types.Content(
|
|
127
|
+
parts=[
|
|
128
|
+
types.Part.from_text(text=prompt),
|
|
129
|
+
types.Part.from_bytes(data=png_data, mime_type="image/png"),
|
|
130
|
+
]
|
|
131
|
+
)
|
|
132
|
+
],
|
|
133
|
+
config={"media_resolution": "media_resolution_high"}, # type: ignore[arg-type]
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
text_content = response.text if response.text else ""
|
|
137
|
+
|
|
138
|
+
# Save Result
|
|
139
|
+
async with session_manager.session() as session:
|
|
140
|
+
content = await get_page_content_by_id(session, file_id, page_id)
|
|
141
|
+
if content:
|
|
142
|
+
content.text_content = text_content
|
|
143
|
+
else:
|
|
144
|
+
logger.warning(
|
|
145
|
+
f"NotePageContentDO missing for {file_id} page {page_id} during OCR"
|
|
146
|
+
)
|
|
147
|
+
await session.commit()
|
|
148
|
+
|
|
149
|
+
logger.info(f"Completed OCR for file {file_id} page {page_id}")
|
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
import logging
|
|
3
|
+
from functools import partial
|
|
4
|
+
from typing import Any, Optional
|
|
5
|
+
|
|
6
|
+
from sqlalchemy import delete, select
|
|
7
|
+
|
|
8
|
+
from supernote.notebook.parser import parse_metadata
|
|
9
|
+
from supernote.server.constants import CACHE_BUCKET, USER_DATA_BUCKET
|
|
10
|
+
from supernote.server.db.models.file import UserFileDO
|
|
11
|
+
from supernote.server.db.models.note_processing import NotePageContentDO, SystemTaskDO
|
|
12
|
+
from supernote.server.db.session import DatabaseSessionManager
|
|
13
|
+
from supernote.server.services.file import FileService
|
|
14
|
+
from supernote.server.services.processor_modules import ProcessorModule
|
|
15
|
+
from supernote.server.utils.hashing import get_md5_hash
|
|
16
|
+
from supernote.server.utils.paths import get_page_png_path
|
|
17
|
+
|
|
18
|
+
logger = logging.getLogger(__name__)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _parse_helper(path: str) -> Any:
|
|
22
|
+
with open(path, "rb") as f:
|
|
23
|
+
return parse_metadata(f) # type: ignore[arg-type]
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class PageHashingModule(ProcessorModule):
|
|
27
|
+
"""Module responsible for detecting changes in .note files at the page level.
|
|
28
|
+
|
|
29
|
+
This module performs the following:
|
|
30
|
+
- Parses the binary .note file using `SupernoteParser`.
|
|
31
|
+
- Computes a unique MD5 hash for each page based on its layer metadata.
|
|
32
|
+
- Updates the `NotePageContentDO` table:
|
|
33
|
+
- Creates entries for new pages.
|
|
34
|
+
- Updates hashes for changed pages and invalidates downstream data (OCR, Embeddings) to trigger reprocessing.
|
|
35
|
+
- Removes entries for deleted pages.
|
|
36
|
+
- Updates the `SystemTaskDO` status for the 'HASHING' task (handled by base class).
|
|
37
|
+
|
|
38
|
+
This module acts as the entry point and "change detector" for the incremental processing pipeline.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
def __init__(self, file_service: FileService) -> None:
|
|
42
|
+
self.file_service = file_service
|
|
43
|
+
|
|
44
|
+
@property
|
|
45
|
+
def name(self) -> str:
|
|
46
|
+
return "PageHashingModule"
|
|
47
|
+
|
|
48
|
+
@property
|
|
49
|
+
def task_type(self) -> str:
|
|
50
|
+
return "HASHING"
|
|
51
|
+
|
|
52
|
+
async def run_if_needed(
|
|
53
|
+
self,
|
|
54
|
+
file_id: int,
|
|
55
|
+
session_manager: DatabaseSessionManager,
|
|
56
|
+
page_index: Optional[int] = None,
|
|
57
|
+
page_id: Optional[str] = None,
|
|
58
|
+
) -> bool:
|
|
59
|
+
"""Hashing acts as the change detector, so it MUST run every time the file is processed."""
|
|
60
|
+
return True
|
|
61
|
+
|
|
62
|
+
async def process(
|
|
63
|
+
self,
|
|
64
|
+
file_id: int,
|
|
65
|
+
session_manager: DatabaseSessionManager,
|
|
66
|
+
page_index: Optional[int] = None,
|
|
67
|
+
page_id: Optional[str] = None,
|
|
68
|
+
**kwargs: object,
|
|
69
|
+
) -> None:
|
|
70
|
+
"""Parses the .note file, computes page hashes, and updates NotePageContentDO."""
|
|
71
|
+
logger.info(f"Starting PageHashingModule for file_id={file_id}")
|
|
72
|
+
|
|
73
|
+
# Resolve file path
|
|
74
|
+
async with session_manager.session() as session:
|
|
75
|
+
# Get UserFileDO to find owner & storage key
|
|
76
|
+
result = await session.execute(
|
|
77
|
+
select(UserFileDO).where(UserFileDO.id == file_id)
|
|
78
|
+
)
|
|
79
|
+
user_file = result.scalars().first()
|
|
80
|
+
if not user_file:
|
|
81
|
+
logger.error(f"File {file_id} not found in DB")
|
|
82
|
+
return
|
|
83
|
+
|
|
84
|
+
storage_key = user_file.storage_key
|
|
85
|
+
if not storage_key:
|
|
86
|
+
logger.error(f"File {file_id} has no storage_key")
|
|
87
|
+
return
|
|
88
|
+
|
|
89
|
+
# Construct real OS path via BlobStorage (No DB access needed here)
|
|
90
|
+
try:
|
|
91
|
+
# This assumes LocalBlobStorage. For S3, we'd need to stream,
|
|
92
|
+
# but SupernoteParser currently expects a file path.
|
|
93
|
+
abs_path = self.file_service.blob_storage.get_blob_path(
|
|
94
|
+
USER_DATA_BUCKET, storage_key
|
|
95
|
+
)
|
|
96
|
+
except Exception as e:
|
|
97
|
+
logger.error(f"Failed to resolve blob path for {file_id}: {e}")
|
|
98
|
+
return
|
|
99
|
+
|
|
100
|
+
if not abs_path.exists():
|
|
101
|
+
logger.error(f"File {abs_path} does not exist on disk")
|
|
102
|
+
return
|
|
103
|
+
|
|
104
|
+
# Parse .note file
|
|
105
|
+
try:
|
|
106
|
+
# Run parser in thread pool
|
|
107
|
+
loop = asyncio.get_running_loop()
|
|
108
|
+
metadata = await loop.run_in_executor(
|
|
109
|
+
None, partial(_parse_helper, str(abs_path))
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
except Exception as e:
|
|
113
|
+
logger.error(f"Failed to parse .note file {file_id}: {e}")
|
|
114
|
+
return
|
|
115
|
+
|
|
116
|
+
# Iterate pages and update DB
|
|
117
|
+
total_pages = metadata.get_total_pages()
|
|
118
|
+
if not metadata.pages:
|
|
119
|
+
return
|
|
120
|
+
|
|
121
|
+
async with session_manager.session() as session:
|
|
122
|
+
# 1. Fetch all existing pages for this file and map by page_id
|
|
123
|
+
existing_pages = await session.execute(
|
|
124
|
+
select(NotePageContentDO).where(NotePageContentDO.file_id == file_id)
|
|
125
|
+
)
|
|
126
|
+
existing_map = {p.page_id: p for p in existing_pages.scalars().all()}
|
|
127
|
+
|
|
128
|
+
# Track which page_ids are present in the current file
|
|
129
|
+
current_page_ids = set()
|
|
130
|
+
|
|
131
|
+
for i in range(total_pages):
|
|
132
|
+
page_info = metadata.pages[i]
|
|
133
|
+
page_id = page_info.get("PAGEID")
|
|
134
|
+
|
|
135
|
+
if not page_id:
|
|
136
|
+
logger.warning(f"Page {i} of file {file_id} has no PAGEID.")
|
|
137
|
+
continue
|
|
138
|
+
|
|
139
|
+
current_page_ids.add(page_id)
|
|
140
|
+
|
|
141
|
+
# Canonical string representation of page metadata for hashing
|
|
142
|
+
page_hash_input = str(page_info)
|
|
143
|
+
current_hash = get_md5_hash(page_hash_input)
|
|
144
|
+
|
|
145
|
+
if page_id in existing_map:
|
|
146
|
+
# UPDATE path
|
|
147
|
+
row = existing_map[page_id]
|
|
148
|
+
|
|
149
|
+
# Check for Move (Index Changed)
|
|
150
|
+
if row.page_index != i:
|
|
151
|
+
logger.info(
|
|
152
|
+
f"Page {page_id} moved from {row.page_index} to {i}"
|
|
153
|
+
)
|
|
154
|
+
row.page_index = i
|
|
155
|
+
|
|
156
|
+
# Check for Content Change
|
|
157
|
+
if row.content_hash != current_hash:
|
|
158
|
+
logger.info(
|
|
159
|
+
f"Page {page_id} (idx {i}) changed. Resetting content."
|
|
160
|
+
)
|
|
161
|
+
row.content_hash = current_hash
|
|
162
|
+
row.text_content = None # Clear OCR
|
|
163
|
+
row.embedding = None # Clear Embedding
|
|
164
|
+
|
|
165
|
+
# Invalidate downstream tasks
|
|
166
|
+
# Note: Tasks are keyed by page_index usually?
|
|
167
|
+
# We should probably migrate tasks to use page_id too,
|
|
168
|
+
# but for now let's invalidate based on the NEW index
|
|
169
|
+
# or just all tasks for this file/page logic.
|
|
170
|
+
# Ideally, SystemTaskDO key should be page_id.
|
|
171
|
+
# For now, let's assume keys are "page_{page_index}".
|
|
172
|
+
# If page moved, the old task key "page_{old_idx}" is irrelevant,
|
|
173
|
+
# and we need to trigger "page_{new_idx}".
|
|
174
|
+
# Refactoring SystemTaskDO is out of scope?
|
|
175
|
+
# Let's clean up tasks for this "logical" page.
|
|
176
|
+
# Actually, keeping it simple: invalidate tasks for the CURRENT index.
|
|
177
|
+
# Update: Use page_id for task keys now.
|
|
178
|
+
page_task_key = f"page_{page_id}"
|
|
179
|
+
await session.execute(
|
|
180
|
+
delete(SystemTaskDO)
|
|
181
|
+
.where(SystemTaskDO.file_id == file_id)
|
|
182
|
+
.where(SystemTaskDO.key == page_task_key)
|
|
183
|
+
)
|
|
184
|
+
else:
|
|
185
|
+
# INSERT path
|
|
186
|
+
logger.info(f"New page {page_id} at index {i} detected.")
|
|
187
|
+
new_content = NotePageContentDO(
|
|
188
|
+
file_id=file_id,
|
|
189
|
+
page_index=i,
|
|
190
|
+
page_id=page_id,
|
|
191
|
+
content_hash=current_hash,
|
|
192
|
+
text_content=None,
|
|
193
|
+
embedding=None,
|
|
194
|
+
)
|
|
195
|
+
session.add(new_content)
|
|
196
|
+
|
|
197
|
+
# 2. Handle Deletions
|
|
198
|
+
# Any page_id in existing_map but NOT in current_page_ids is deleted
|
|
199
|
+
for pid, row in existing_map.items():
|
|
200
|
+
if pid not in current_page_ids:
|
|
201
|
+
logger.info(f"Page {pid} deleted (was at {row.page_index}).")
|
|
202
|
+
await session.delete(row)
|
|
203
|
+
|
|
204
|
+
# Cleanup Blobs
|
|
205
|
+
png_path = get_page_png_path(file_id, pid)
|
|
206
|
+
try:
|
|
207
|
+
await self.file_service.blob_storage.delete(
|
|
208
|
+
CACHE_BUCKET, png_path
|
|
209
|
+
)
|
|
210
|
+
except Exception as e:
|
|
211
|
+
logger.warning(f"Failed to delete orphan PNG {pid}: {e}")
|
|
212
|
+
|
|
213
|
+
# Cleanup Tasks
|
|
214
|
+
await session.execute(
|
|
215
|
+
delete(SystemTaskDO)
|
|
216
|
+
.where(SystemTaskDO.file_id == file_id)
|
|
217
|
+
.where(SystemTaskDO.key == f"page_{pid}")
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
await session.commit()
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
import io
|
|
3
|
+
import logging
|
|
4
|
+
from functools import partial
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
from sqlalchemy import select
|
|
8
|
+
|
|
9
|
+
from supernote.notebook.converter import ImageConverter
|
|
10
|
+
from supernote.notebook.parser import load_notebook
|
|
11
|
+
from supernote.server.constants import CACHE_BUCKET, USER_DATA_BUCKET
|
|
12
|
+
from supernote.server.db.models.file import UserFileDO
|
|
13
|
+
from supernote.server.db.models.note_processing import NotePageContentDO
|
|
14
|
+
from supernote.server.db.session import DatabaseSessionManager
|
|
15
|
+
from supernote.server.services.file import FileService
|
|
16
|
+
from supernote.server.services.processor_modules import ProcessorModule
|
|
17
|
+
from supernote.server.utils.paths import get_page_png_path
|
|
18
|
+
|
|
19
|
+
logger = logging.getLogger(__name__)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _convert_helper(path: str, page_index: int) -> bytes:
|
|
23
|
+
# Use loose policy to attempt parsing even if signature is unknown
|
|
24
|
+
notebook = load_notebook(path, policy="loose") # type: ignore[no-untyped-call]
|
|
25
|
+
converter = ImageConverter(notebook) # type: ignore[no-untyped-call]
|
|
26
|
+
img = converter.convert(page_index) # type: ignore[no-untyped-call]
|
|
27
|
+
buf = io.BytesIO()
|
|
28
|
+
img.save(buf, format="PNG")
|
|
29
|
+
return buf.getvalue()
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class PngConversionModule(ProcessorModule):
|
|
33
|
+
def __init__(self, file_service: FileService) -> None:
|
|
34
|
+
self.file_service = file_service
|
|
35
|
+
|
|
36
|
+
@property
|
|
37
|
+
def name(self) -> str:
|
|
38
|
+
return "PngConversionModule"
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def task_type(self) -> str:
|
|
42
|
+
return "PNG_CONVERSION"
|
|
43
|
+
|
|
44
|
+
async def process(
|
|
45
|
+
self,
|
|
46
|
+
file_id: int,
|
|
47
|
+
session_manager: DatabaseSessionManager,
|
|
48
|
+
page_index: Optional[int] = None,
|
|
49
|
+
page_id: Optional[str] = None,
|
|
50
|
+
**kwargs: object,
|
|
51
|
+
) -> None:
|
|
52
|
+
"""
|
|
53
|
+
Converts a specific page of the .note file to PNG and saves it to blob storage.
|
|
54
|
+
"""
|
|
55
|
+
if page_index is None:
|
|
56
|
+
logger.warning(
|
|
57
|
+
f"PngConversionModule requires page_index for file {file_id}"
|
|
58
|
+
)
|
|
59
|
+
return
|
|
60
|
+
|
|
61
|
+
# Resolve file path
|
|
62
|
+
async with session_manager.session() as session:
|
|
63
|
+
result = await session.execute(
|
|
64
|
+
select(UserFileDO).where(UserFileDO.id == file_id)
|
|
65
|
+
)
|
|
66
|
+
user_file = result.scalars().first()
|
|
67
|
+
if not user_file or not user_file.storage_key:
|
|
68
|
+
logger.error(f"File {file_id} not found or missing storage_key")
|
|
69
|
+
return
|
|
70
|
+
storage_key = user_file.storage_key
|
|
71
|
+
|
|
72
|
+
# Get Page ID for stable path
|
|
73
|
+
page_result = await session.execute(
|
|
74
|
+
select(NotePageContentDO.page_id).where(
|
|
75
|
+
NotePageContentDO.file_id == file_id,
|
|
76
|
+
NotePageContentDO.page_index == page_index,
|
|
77
|
+
)
|
|
78
|
+
)
|
|
79
|
+
page_id = page_result.scalars().first()
|
|
80
|
+
if not page_id:
|
|
81
|
+
logger.error(f"Page ID not found for {file_id} index {page_index}")
|
|
82
|
+
return
|
|
83
|
+
|
|
84
|
+
try:
|
|
85
|
+
abs_path = self.file_service.blob_storage.get_blob_path(
|
|
86
|
+
USER_DATA_BUCKET, storage_key
|
|
87
|
+
)
|
|
88
|
+
except Exception as e:
|
|
89
|
+
logger.error(f"Failed to resolve blob path for {file_id}: {e}")
|
|
90
|
+
raise
|
|
91
|
+
|
|
92
|
+
if not abs_path.exists():
|
|
93
|
+
logger.error(f"File {abs_path} does not exist on disk")
|
|
94
|
+
raise FileNotFoundError(f"File {abs_path} does not exist on disk")
|
|
95
|
+
|
|
96
|
+
# Run Conversion in Thread Pool
|
|
97
|
+
loop = asyncio.get_running_loop()
|
|
98
|
+
png_data = await loop.run_in_executor(
|
|
99
|
+
None, partial(_convert_helper, str(abs_path), page_index)
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
# Upload to Blob Storage
|
|
103
|
+
blob_path = get_page_png_path(file_id, page_id)
|
|
104
|
+
await self.file_service.blob_storage.put(CACHE_BUCKET, blob_path, png_data)
|
|
105
|
+
logger.info(
|
|
106
|
+
f"Successfully converted page {page_index} ({page_id}) of {file_id} to PNG"
|
|
107
|
+
)
|