gllm-misc-binary 0.9.6__cp313-cp313-win_amd64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
gllm_misc/__init__.pyi ADDED
File without changes
@@ -0,0 +1,3 @@
1
+ from gllm_misc.cache_manager.cache_manager import CacheManager as CacheManager, CacheOperationType as CacheOperationType
2
+
3
+ __all__ = ['CacheManager', 'CacheOperationType']
@@ -0,0 +1,59 @@
1
+ from _typeshed import Incomplete
2
+ from enum import StrEnum
3
+ from gllm_core.schema import Component
4
+ from gllm_datastore.cache.hybrid_cache.hybrid_cache import BaseHybridCache
5
+ from typing import Any
6
+
7
+ class CacheOperationType(StrEnum):
8
+ """The type of operation for the cache manager.
9
+
10
+ Attribute:
11
+ RETRIEVE (str): The operation type for retrieving the cache.
12
+ STORE (str): The operation type for storing the cache.
13
+ """
14
+ RETRIEVE: str
15
+ STORE: str
16
+
17
+ class CacheManager(Component):
18
+ """A class for managing cache in Gen AI applications.
19
+
20
+ This class provides functionality for storing and retrieving cache.
21
+
22
+ Attributes:
23
+ data_store (BaseHybridCache): The cache store to use for storing and retrieving the cache.
24
+ """
25
+ data_store: Incomplete
26
+ def __init__(self, data_store: BaseHybridCache) -> None:
27
+ """Initializes a new instance of the CacheManager class.
28
+
29
+ Args:
30
+ data_store (BaseCacheDataStore): The data store to use for the cache manager.
31
+ """
32
+ async def retrieve(self, key: str | dict[str, Any] | tuple[Any, ...]) -> tuple[Any, bool]:
33
+ """Retrieves the cache of a given key.
34
+
35
+ This method stringifies the key and then calls the `retrieve` method of the data store to retrieve the cache.
36
+ It returns the retrieved cache and a boolean indicating if the cache was found.
37
+
38
+ Args:
39
+ key (str | dict[str, Any] | tuple[Any, ...]): The key of the cache to retrieve.
40
+
41
+ Returns:
42
+ tuple[Any, bool]: The retrieved cache and a boolean indicating if the cache was found.
43
+ """
44
+ async def store(self, key: str | dict[str, Any] | tuple[Any, ...], value: Any, ttl: int | str | None = None) -> None:
45
+ '''Stores the cache of a given key.
46
+
47
+ This method stringifies the key and then calls the `store` method of the data store to store the cache.
48
+ If the value is None, the method will skip the storage process.
49
+
50
+ Args:
51
+ key (str | dict[str, Any] | tuple[Any, ...]): The key of the cache to store.
52
+ value (Any): The value of the cache to store.
53
+ ttl (int | str | None, optional): The time-to-live (TTL) for the cache data.
54
+ Must be an integer in seconds or a string in a valid time format (e.g. "1h", "1d", "1w", "1m", "1y").
55
+ If None, the cache data will not expire.
56
+
57
+ Returns:
58
+ None
59
+ '''
@@ -0,0 +1,5 @@
1
+ from gllm_misc.graph_transformer.lm_based_graph_transformer import LMBasedGraphTransformer as LMBasedGraphTransformer
2
+ from gllm_misc.graph_transformer.lm_based_mindmap_transformer import LMBasedMindMapTransformer as LMBasedMindMapTransformer
3
+ from gllm_misc.graph_transformer.schema import GraphDocument as GraphDocument, Node as Node, Relationship as Relationship
4
+
5
+ __all__ = ['GraphDocument', 'LMBasedGraphTransformer', 'LMBasedMindMapTransformer', 'Node', 'Relationship']
@@ -0,0 +1,10 @@
1
+ from _typeshed import Incomplete
2
+
3
+ DEFAULT_NODE_TYPE: str
4
+ MINDMAP_DEFAULT_ALLOWED_NODES: Incomplete
5
+ MINDMAP_DEFAULT_ALLOWED_RELATIONSHIPS: Incomplete
6
+ MINDMAP_DEFAULT_ROOT_NODE_TYPE: str
7
+ MINDMAP_DEFAULT_ROOT_RELATIONSHIP_TYPE: str
8
+ MINDMAP_DEFAULT_TRANSFORMER_PROMPT: str
9
+ MINDMAP_DEFAULT_ROOT_NODE_PROMPT: str
10
+ RELATIONSHIP_TUPLE_LENGTH: int
@@ -0,0 +1,137 @@
1
+ from _typeshed import Incomplete
2
+ from gllm_core.schema import Chunk
3
+ from gllm_inference.lm_invoker.lm_invoker import BaseLMInvoker
4
+ from gllm_inference.prompt_builder import PromptBuilder
5
+ from gllm_inference.schema import ModelId
6
+ from gllm_misc.graph_transformer.schema import GraphDocument as GraphDocument, GraphResponse as GraphResponse, RelationshipTypeFormat as RelationshipTypeFormat
7
+ from gllm_misc.graph_transformer.utils import convert_to_graph_document as convert_to_graph_document, create_simple_model as create_simple_model, filter_graph_document_by_allowed_nodes_and_relationships as filter_graph_document_by_allowed_nodes_and_relationships, filter_relationships_by_existing_nodes as filter_relationships_by_existing_nodes, get_default_prompt as get_default_prompt, validate_and_get_relationship_type as validate_and_get_relationship_type
8
+ from typing import Any
9
+
10
+ class LMBasedGraphTransformer:
11
+ '''Transform documents into graph-based documents using a language model.
12
+
13
+ This class orchestrates the extraction of knowledge graphs from text documents
14
+ using a language model. It handles prompt construction, model invocation, and
15
+ parsing of results into a standardized graph format.
16
+
17
+ The transformer can be configured with constraints on node and relationship types,
18
+ and supports both structured and unstructured output formats from the language model.
19
+
20
+ Attributes:
21
+ lm_invoker: The language model invoker used to generate graph data.
22
+ allowed_nodes: List of allowed node types to constrain the output.
23
+ allowed_relationships: List of allowed relationship types to constrain the output.
24
+ strict_mode: Whether to strictly enforce node and relationship type constraints.
25
+ use_structured_output: Whether to use the LM\'s structured output capabilities.
26
+ prompt_builder: The prompt builder used to generate prompts for the LM.
27
+
28
+ Example:
29
+ Using an existing LM invoker:
30
+
31
+ ```python
32
+ from gllm_inference.lm_invoker import OpenAILMInvoker
33
+ from gllm_misc.graph_transformer import LMBasedGraphTransformer
34
+ from gllm_core.schema import Chunk
35
+
36
+ # Create an LM invoker
37
+ invoker = OpenAILMInvoker(model_name="gpt-4o-mini", api_key="sk-proj-123")
38
+
39
+ # Create a graph transformer with constraints
40
+ transformer = LMBasedGraphTransformer(
41
+ lm_invoker=invoker,
42
+ allowed_nodes=["Person", "Organization", "Event"],
43
+ allowed_relationships=["WORKS_AT", "PARTICIPATES_IN"]
44
+ )
45
+
46
+ # Extract graph from text
47
+ chunks = [Chunk(content="Elon Musk is the CEO of SpaceX and Tesla.")]
48
+ graph_docs = await transformer.convert_to_graph_documents(chunks)
49
+ ```
50
+
51
+ Building LM invoker from model ID:
52
+
53
+ ```python
54
+ from gllm_misc.graph_transformer import LMBasedGraphTransformer
55
+ from gllm_core.schema import Chunk
56
+
57
+ # Create a graph transformer by specifying model ID
58
+ transformer = LMBasedGraphTransformer(
59
+ model_id="openai/gpt-4o-mini",
60
+ allowed_nodes=["Person", "Organization", "Event"],
61
+ allowed_relationships=["WORKS_AT", "PARTICIPATES_IN"]
62
+ )
63
+
64
+ # Extract graph from text
65
+ chunks = [Chunk(content="Elon Musk is the CEO of SpaceX and Tesla.")]
66
+ graph_docs = await transformer.convert_to_graph_documents(chunks)
67
+ ```
68
+ '''
69
+ lm_invoker: Incomplete
70
+ def __init__(self, lm_invoker: BaseLMInvoker | None = None, model_id: str | ModelId | None = None, credentials: str | dict[str, Any] | None = None, config: dict[str, Any] | None = None, allowed_nodes: list[str] | None = None, allowed_relationships: list[str] | list[tuple[str, str, str]] | None = None, prompt_builder: PromptBuilder | None = None, strict_mode: bool = True, use_structured_output: bool = False) -> None:
71
+ """Initialize the LMBasedGraphTransformer.
72
+
73
+ Args:
74
+ lm_invoker (BaseLMInvoker | None, optional): The language model invoker to use for generating graph data.
75
+ Either lm_invoker or model_id must be provided. Defaults to None.
76
+ model_id (str | ModelId | None, optional): The model ID to use for building the LM invoker.
77
+ Required if lm_invoker is not provided. Defaults to None.
78
+ credentials (str | dict[str, Any] | None, optional): Credentials for the LM model.
79
+ Used when building the LM invoker. Defaults to None.
80
+ config (dict[str, Any] | None, optional): Configuration for the LM model.
81
+ Used when building the LM invoker. Defaults to None.
82
+ allowed_nodes (list[str] | None, optional): Optional list of allowed node types.
83
+ If provided, the transformer will constrain output to only use these node types.
84
+ Defaults to None.
85
+ allowed_relationships (list[str] | list[tuple[str, str, str]] | None, optional): Optional list of allowed
86
+ relationship types. Can be either a list of strings or a list of tuples in the format
87
+ (source_type, relationship_type, target_type).
88
+ Defaults to None.
89
+ prompt_builder (PromptBuilder | None, optional): Optional custom prompt builder. Defaults to a
90
+ prompt builder created using get_default_prompt().
91
+ strict_mode (bool, optional): Determines whether the transformer should apply filtering to
92
+ strictly adhere to allowed_nodes and allowed_relationships.
93
+ Defaults to True.
94
+ use_structured_output (bool, optional): Indicates whether the transformer should use the
95
+ language model's native structured output functionality.
96
+ Defaults to False.
97
+ """
98
+ async def process_response(self, chunk: Chunk) -> GraphDocument:
99
+ """Process a single text chunk and transform it into a graph document.
100
+
101
+ This method handles the core transformation logic, including:
102
+ 1. Formatting the prompt with the chunk content
103
+ 2. Invoking the language model
104
+ 3. Parsing the response into nodes and relationships
105
+ 4. Applying filtering based on allowed node and relationship types if strict_mode is enabled
106
+
107
+ Args:
108
+ chunk (Chunk): A Chunk object containing the text content to transform into a graph
109
+
110
+ Returns:
111
+ GraphDocument: A GraphDocument containing the extracted nodes and relationships
112
+ """
113
+ async def convert_to_graph_documents(self, documents: list[Chunk]) -> list[GraphDocument]:
114
+ '''Asynchronously convert a sequence of text chunks into graph documents.
115
+
116
+ This method processes multiple chunks in parallel by creating asyncio tasks
117
+ for each document and gathering their results. Each chunk is processed
118
+ independently using the process_response method.
119
+
120
+ Args:
121
+ documents (list[Chunk]): A list of Chunk objects containing the text content to
122
+ transform into graphs.
123
+
124
+ Returns:
125
+ list[GraphDocument]: A list of GraphDocument objects, each containing nodes and relationships
126
+ extracted from the corresponding input chunk.
127
+
128
+ Example:
129
+ ```python
130
+ chunks = [
131
+ Chunk(content="Alice works at Acme Corp."),
132
+ Chunk(content="Bob is the CEO of TechStart.")
133
+ ]
134
+ graph_docs = await transformer.convert_to_graph_documents(chunks)
135
+ # Returns a list of two GraphDocument objects
136
+ ```
137
+ '''
@@ -0,0 +1,127 @@
1
+ from _typeshed import Incomplete
2
+ from gllm_core.schema import Chunk as Chunk
3
+ from gllm_inference.lm_invoker.lm_invoker import BaseLMInvoker
4
+ from gllm_inference.prompt_builder import PromptBuilder
5
+ from gllm_inference.schema import ModelId
6
+ from gllm_misc.graph_transformer.constant import MINDMAP_DEFAULT_ALLOWED_NODES as MINDMAP_DEFAULT_ALLOWED_NODES, MINDMAP_DEFAULT_ALLOWED_RELATIONSHIPS as MINDMAP_DEFAULT_ALLOWED_RELATIONSHIPS, MINDMAP_DEFAULT_ROOT_NODE_PROMPT as MINDMAP_DEFAULT_ROOT_NODE_PROMPT, MINDMAP_DEFAULT_ROOT_NODE_TYPE as MINDMAP_DEFAULT_ROOT_NODE_TYPE, MINDMAP_DEFAULT_ROOT_RELATIONSHIP_TYPE as MINDMAP_DEFAULT_ROOT_RELATIONSHIP_TYPE, MINDMAP_DEFAULT_TRANSFORMER_PROMPT as MINDMAP_DEFAULT_TRANSFORMER_PROMPT
7
+ from gllm_misc.graph_transformer.lm_based_graph_transformer import LMBasedGraphTransformer as LMBasedGraphTransformer
8
+ from gllm_misc.graph_transformer.schema import GraphDocument as GraphDocument, GraphResponse as GraphResponse, Node as Node, Relationship as Relationship
9
+ from typing import Any
10
+
11
+ class LMBasedMindMapTransformer(LMBasedGraphTransformer):
12
+ '''Transform documents into mindmap-based knowledge representations using a language model.
13
+
14
+ This class extends LMBasedGraphTransformer to specifically handle the extraction of hierarchical
15
+ mind map structures from text documents. It uses a specialized prompt and default node/relationship
16
+ types designed for mind mapping, organizing content into central themes, main ideas, and sub-ideas.
17
+
18
+ The transformer maintains the hierarchical structure of ideas while converting text into a
19
+ graph representation that can be visualized as a mind map.
20
+
21
+ The transformer extends LMBasedGraphTransformer by adding specialized functionality for
22
+ merging root nodes and pruning disconnected graph documents to ensure the mind map forms
23
+ a connected graph stemming from a single root node.
24
+
25
+ Attributes:
26
+ lm_invoker (BaseLMInvoker): The language model invoker used to generate graph data.
27
+ allowed_nodes (list[str]): List of allowed node types to constrain the output.
28
+ Defaults to MINDMAP_DEFAULT_ALLOWED_NODES (CentralTheme, MainIdea, SubIdea).
29
+ allowed_relationships (list[str] | list[tuple[str, str, str]]): List of allowed
30
+ relationship types to constrain the output. Defaults to MINDMAP_DEFAULT_ALLOWED_RELATIONSHIPS
31
+ which define the hierarchical structure of a mind map.
32
+ root_node_type (str): The type of the root node. Defaults to
33
+ MINDMAP_DEFAULT_ROOT_NODE_TYPE (CentralTheme).
34
+ strict_mode (bool): Whether to strictly enforce node and relationship type constraints.
35
+ Defaults to True.
36
+ use_structured_output (bool): Whether to use the LM\'s structured output capabilities.
37
+ Defaults to True.
38
+ prompt_builder (PromptBuilder, optional): The prompt builder used to generate prompts for the LM.
39
+ If None, creates a default prompt builder with the mind map extraction prompt.
40
+
41
+ Example:
42
+ Using an existing LM invoker:
43
+
44
+ ```python
45
+ from gllm_inference.lm_invoker import OpenAILMInvoker
46
+ from gllm_misc.graph_transformer import LMBasedMindMapTransformer
47
+ from gllm_core.schema import Chunk
48
+
49
+ # Create an LM invoker
50
+ invoker = OpenAILMInvoker(model_name="gpt-4o-mini", api_key="sk-proj-123")
51
+
52
+ # Create a mind map transformer
53
+ transformer = LMBasedMindMapTransformer(lm_invoker=invoker)
54
+
55
+ # Extract mind map from text
56
+ chunks = [Chunk(content="Artificial Intelligence is transforming industries...")]
57
+ mind_map_docs = await transformer.convert_to_graph_documents(chunks)
58
+ ```
59
+
60
+ Building LM invoker from model ID:
61
+
62
+ ```python
63
+ from gllm_misc.graph_transformer import LMBasedMindMapTransformer
64
+ from gllm_core.schema import Chunk
65
+
66
+ # Create a mind map transformer by specifying model ID
67
+ transformer = LMBasedMindMapTransformer(
68
+ model_id="openai/gpt-4o-mini",
69
+ )
70
+
71
+ # Extract mind map from text
72
+ chunks = [Chunk(content="Artificial Intelligence is transforming industries...")]
73
+ mind_map_docs = await transformer.convert_to_graph_documents(chunks)
74
+ ```
75
+ '''
76
+ root_node_type: Incomplete
77
+ def __init__(self, lm_invoker: BaseLMInvoker | None = None, model_id: str | ModelId | None = None, credentials: str | dict[str, Any] | None = None, config: dict[str, Any] | None = None, allowed_nodes: list[str] | None = None, allowed_relationships: list[str] | list[tuple[str, str, str]] | None = None, root_node_type: str | None = None, prompt_builder: PromptBuilder | None = None, strict_mode: bool = True, use_structured_output: bool = True) -> None:
78
+ """Initialize the LMBasedMindMapTransformer with the specified configuration.
79
+
80
+ This constructor sets up the mind map transformer with appropriate defaults
81
+ for mind map extraction if not explicitly provided.
82
+
83
+ Args:
84
+ lm_invoker (BaseLMInvoker | None, optional): The language model invoker to use for generating graph data.
85
+ Either lm_invoker or model_id must be provided. Defaults to None.
86
+ model_id (str | ModelId | None, optional): The model ID to use for building the LM invoker.
87
+ Required if lm_invoker is not provided. Defaults to None.
88
+ credentials (str | dict[str, Any] | None, optional): Credentials for the LM model.
89
+ Used when building the LM invoker. Defaults to None.
90
+ config (dict[str, Any] | None, optional): Configuration for the LM model.
91
+ Used when building the LM invoker. Defaults to None.
92
+ allowed_nodes (list[str] | None, optional): Optional list of allowed node types. If None,
93
+ defaults to MINDMAP_DEFAULT_ALLOWED_NODES (CentralTheme, MainIdea, SubIdea).
94
+ Defaults to None.
95
+ allowed_relationships (list[str] | list[tuple[str, str, str]] | None, optional): Optional list of
96
+ allowed relationship types. If None, defaults to MINDMAP_DEFAULT_ALLOWED_RELATIONSHIPS
97
+ which define the hierarchical structure of a mind map.
98
+ Defaults to None.
99
+ root_node_type (str | None, optional): Optional root node type. If None,
100
+ uses the default root node type MINDMAP_DEFAULT_ROOT_NODE_TYPE.
101
+ Defaults to None.
102
+ prompt_builder (PromptBuilder | None, optional): Optional prompt builder. If None,
103
+ creates a default prompt builder with the mind map extraction prompt.
104
+ Defaults to None.
105
+ strict_mode (bool, optional): Whether to strictly enforce node and relationship type constraints.
106
+ Defaults to True.
107
+ use_structured_output (bool, optional): Whether to use the LM's structured output capabilities.
108
+ Defaults to True.
109
+
110
+ Raises:
111
+ ValueError: If root_node_type is not in allowed_nodes
112
+ or if both/none of lm_invoker and lm_model_id are provided.
113
+ """
114
+ async def convert_to_graph_documents(self, documents: list[Chunk]) -> list[GraphDocument]:
115
+ """Asynchronously convert a sequence of text chunks into graph documents.
116
+
117
+ This method overrides the parent class to ensure that each graph document
118
+ has its root nodes merged after processing.
119
+
120
+ Args:
121
+ documents (list[Chunk]): A list of Chunk objects containing the text content to
122
+ transform into graphs.
123
+
124
+ Returns:
125
+ list[GraphDocument]: A list of GraphDocument objects, each containing nodes and relationships
126
+ extracted from the corresponding input chunk, with merged root nodes.
127
+ """
@@ -0,0 +1,19 @@
1
+ from gllm_misc.graph_transformer.schema import PrimitiveNodeType as PrimitiveNodeType, PrimitiveRelationshipType as PrimitiveRelationshipType
2
+
3
+ def get_default_graph_transformer_system_prompt(allowed_nodes: list[PrimitiveNodeType] | None = None, allowed_relationships: list[PrimitiveRelationshipType] | None = None) -> str:
4
+ """Generate system prompt with optional allowed nodes and relationships.
5
+
6
+ This is a public utility function that creates a formatted system prompt for LLMs
7
+ to guide knowledge graph extraction. The prompt includes instructions for the model
8
+ on how to identify and structure entities and relationships.
9
+
10
+ Args:
11
+ allowed_nodes (list[PrimitiveNodeType] | None, optional): Optional list of allowed node types.
12
+ If provided, the prompt will instruct the model to only use these specific node types. Defaults to None.
13
+ allowed_relationships (list[PrimitiveRelationshipType] | None, optional): Optional list of allowed
14
+ relationship types. If provided, the prompt will instruct the model to only use these specific
15
+ relationship types. Defaults to None.
16
+
17
+ Returns:
18
+ str: Formatted system prompt string ready to be used with an LLM.
19
+ """
@@ -0,0 +1,100 @@
1
+ from enum import StrEnum
2
+ from gllm_core.schema import Chunk
3
+ from pydantic import BaseModel
4
+
5
+ class InputType(StrEnum):
6
+ """Enum for input types used in graph element descriptions."""
7
+ NODE: str
8
+ RELATIONSHIP: str
9
+ PROPERTY: str
10
+
11
+ class RelationshipTypeFormat(StrEnum):
12
+ '''Enum for relationship type formats.
13
+
14
+ This enum defines the valid formats for specifying relationship types:
15
+ - STRING: Relationship types are specified as simple strings (e.g., "WORKS_AT")
16
+ - TUPLE: Relationship types are specified as 3-tuples in the format
17
+ (source_type, relationship_type, target_type)
18
+ '''
19
+ STRING: str
20
+ TUPLE: str
21
+
22
+ class Node(BaseModel):
23
+ '''Represents a node in a graph with associated properties.
24
+
25
+ Attributes:
26
+ id (str): A unique identifier for the node.
27
+ type (str): The type or label of the node. Defaults to "Node".
28
+ properties (dict): Additional properties and metadata associated with the node.
29
+ '''
30
+ id: str
31
+ type: str
32
+ properties: dict
33
+
34
+ class Relationship(BaseModel):
35
+ '''Represents a directed relationship between two nodes in a graph.
36
+
37
+ Attributes:
38
+ source (Node): The source node of the relationship.
39
+ target (Node): The target node of the relationship.
40
+ type (str): The type of the relationship. Defaults to "Relationship".
41
+ properties (dict): Additional properties associated with the relationship.
42
+ '''
43
+ source: Node
44
+ target: Node
45
+ type: str
46
+ properties: dict
47
+
48
+ class GraphDocument(BaseModel):
49
+ """Represents a graph document consisting of nodes and relationships.
50
+
51
+ Attributes:
52
+ nodes (list[Node]): A list of nodes in the graph.
53
+ relationships (list[Relationship]): A list of relationships in the graph.
54
+ source (Chunk | None): The document from which the graph information is derived.
55
+ """
56
+ nodes: list[Node]
57
+ relationships: list[Relationship]
58
+ source: Chunk | None
59
+
60
+ class SimpleNode(BaseModel):
61
+ '''Represents a node in a graph with associated properties.
62
+
63
+ This class is to be used as the schema for the LMInvoker\'s structured output.
64
+
65
+ Attributes:
66
+ id (str): A unique identifier for the node.
67
+ type (str): The type or label of the node. Defaults to "Node".
68
+ '''
69
+ id: str
70
+ type: str
71
+
72
+ class SimpleRelationship(BaseModel):
73
+ """Represents a simple directed relationship between two nodes in a graph.
74
+
75
+ This class is to be used as the schema for the LMInvoker's structured output.
76
+
77
+ Attributes:
78
+ source_node_id (str): The ID of the source node.
79
+ source_node_type (str): The type of the source node.
80
+ target_node_id (str): The ID of the target node.
81
+ target_node_type (str): The type of the target node.
82
+ type (str): The type of the relationship.
83
+ """
84
+ source_node_id: str
85
+ source_node_type: str
86
+ target_node_id: str
87
+ target_node_type: str
88
+ type: str
89
+
90
+ class GraphResponse(BaseModel):
91
+ """Represents a graph response generated by LM Invoker containing nodes and relationships.
92
+
93
+ Attributes:
94
+ nodes (list[SimpleNode] | None): A list of nodes generated by LM Invoker.
95
+ relationships (list[SimpleRelationship] | None): A list of relationships generated by LM Invoker.
96
+ """
97
+ nodes: list[SimpleNode] | None
98
+ relationships: list[SimpleRelationship] | None
99
+ PrimitiveNodeType = str
100
+ PrimitiveRelationshipType = str | tuple[str, str, str]
@@ -0,0 +1,176 @@
1
+ from gllm_inference.prompt_builder import PromptBuilder
2
+ from gllm_inference.schema import LMOutput
3
+ from gllm_misc.graph_transformer.constant import DEFAULT_NODE_TYPE as DEFAULT_NODE_TYPE, RELATIONSHIP_TUPLE_LENGTH as RELATIONSHIP_TUPLE_LENGTH
4
+ from gllm_misc.graph_transformer.schema import GraphDocument as GraphDocument, GraphResponse as GraphResponse, InputType as InputType, Node as Node, PrimitiveNodeType as PrimitiveNodeType, PrimitiveRelationshipType as PrimitiveRelationshipType, Relationship as Relationship, RelationshipTypeFormat as RelationshipTypeFormat, SimpleNode as SimpleNode, SimpleRelationship as SimpleRelationship
5
+ from typing import Any
6
+
7
+ def get_default_prompt(allowed_nodes: list[PrimitiveNodeType] | None = None, allowed_relationships: list[PrimitiveRelationshipType] | None = None, relationship_type: RelationshipTypeFormat | None = None) -> PromptBuilder:
8
+ """Create a prompt for LMBasedGraphTransformer that does not use structured output.
9
+
10
+ Args:
11
+ allowed_nodes (list[PrimitiveNodeType] | None, optional): Optional list of allowed node types. Defaults to None.
12
+ allowed_relationships (list[PrimitiveRelationshipType] | None, optional): Optional list of allowed
13
+ relationship types. Defaults to None.
14
+ relationship_type (str | None, optional): Optional string indicating the type of relationship. Defaults to None.
15
+
16
+ Returns:
17
+ A PromptBuilder instance with the prompt template.
18
+ """
19
+ def optional_enum_field(enum_values: list[str] | list[tuple[str, str, str]] | None = None, description: str = '', input_type: str | InputType = ..., relationship_type: RelationshipTypeFormat | None = None, **field_kwargs: Any) -> Any:
20
+ """Utility function to conditionally create a field with an enum constraint.
21
+
22
+ This function creates a Pydantic Field with optional enum constraints based on the
23
+ provided parameters. It handles different LLM types and relationship formats.
24
+
25
+ Args:
26
+ enum_values (list[str] | list[tuple[str, str, str]] | None, optional): List of allowed values for the field.
27
+ Can be a list of strings or a list of tuples (relationship_type, source_node, target_node).
28
+ Defaults to None.
29
+ description (str, optional): Description of the field. Defaults to an empty string.
30
+ input_type (InputType, optional): The type of input to get additional information for.
31
+ Defaults to InputType.NODE.
32
+ relationship_type (RelationshipTypeFormat | None, optional): The type of relationship for relationship fields.
33
+ Defaults to None.
34
+ **field_kwargs (Any, optional): Additional keyword arguments to pass to the Field constructor.
35
+
36
+ Returns:
37
+ Any: A Pydantic Field with the specified constraints and descriptions.
38
+ """
39
+ def create_simple_model(node_labels: list[PrimitiveNodeType] | None = None, rel_types: list[PrimitiveRelationshipType] | None = None, relationship_type: RelationshipTypeFormat | None = None) -> type[GraphResponse]:
40
+ """Create a simple graph model with optional constraints on node and relationship types.
41
+
42
+ This public utility function dynamically creates a Pydantic model for graph data
43
+ with optional constraints on node and relationship types. The model includes
44
+ fields for nodes and relationships with appropriate validation rules.
45
+
46
+ Args:
47
+ node_labels (list[PrimitiveNodeType] | None, optional): Specifies the allowed node types.
48
+ If None, all node types are allowed. Defaults to None.
49
+ rel_types (list[PrimitiveRelationshipType] | None, optional): Specifies the allowed relationship types.
50
+ Can be either a list of strings or a list of tuples in the format (source_type, relationship_type,
51
+ target_type). If None, all relationship types are allowed. Defaults to None.
52
+ relationship_type (str | None, optional): Type of relationship format. If 'tuple', will extract
53
+ relationship types from the tuples in rel_types. Defaults to None.
54
+
55
+ Returns:
56
+ Type[GraphResponse]: A dynamically created Pydantic model class with the specified constraints.
57
+
58
+ Raises:
59
+ ValueError: If 'id' is included in the node or relationship properties list.
60
+ """
61
+ def map_to_base_node(node: SimpleNode) -> Node:
62
+ """Map the SimpleNode to the base Node.
63
+
64
+ This internal helper function converts a dynamically created SimpleNode instance
65
+ to the standard Node class used throughout the application.
66
+
67
+ Args:
68
+ node (SimpleNode): A SimpleNode instance from the dynamically created model.
69
+
70
+ Returns:
71
+ Node: A standard Node instance with properties copied from the SimpleNode.
72
+ """
73
+ def map_to_base_relationship(rel: SimpleRelationship) -> Relationship:
74
+ """Map the SimpleRelationship to the base Relationship.
75
+
76
+ This internal helper function converts a dynamically created SimpleRelationship instance
77
+ to the standard Relationship class used throughout the application.
78
+
79
+ Args:
80
+ rel (SimpleRelationship): A SimpleRelationship instance from the dynamically created model.
81
+
82
+ Returns:
83
+ Relationship: A standard Relationship instance with properties copied from the SimpleRelationship.
84
+ """
85
+ def format_property_key(s: str) -> str:
86
+ '''Format property keys in camelCase style.
87
+
88
+ This utility function converts a space-separated string into camelCase format,
89
+ which is commonly used for property keys in graph databases and JSON.
90
+
91
+ Args:
92
+ s (str): The input string to format as a property key.
93
+
94
+ Returns:
95
+ str: The formatted property key in camelCase (first word lowercase,
96
+ subsequent words capitalized with no spaces).
97
+
98
+ Examples:
99
+ >>> format_property_key("date of birth")
100
+ \'dateOfBirth\'
101
+ >>> format_property_key("name")
102
+ \'name\'
103
+ '''
104
+ def convert_to_graph_document(output: LMOutput | str) -> GraphDocument:
105
+ """Convert LLM output to formatted nodes and relationships.
106
+
107
+ This internal helper function processes the output from a language model
108
+ (either as a string or structured LMOutput) and converts it into properly
109
+ formatted lists of nodes and relationships.
110
+
111
+ The function handles both structured output (Pydantic model) and unstructured
112
+ output (JSON string) formats, applying appropriate parsing and formatting.
113
+
114
+ Args:
115
+ output (LMOutput | str): Either a structured LMOutput object or a JSON string containing
116
+ nodes and relationships data.
117
+
118
+ Returns:
119
+ GraphDocument: A GraphDocument containing the extracted nodes and relationships
120
+
121
+ Note:
122
+ If parsing fails for string output, empty lists will be returned.
123
+ For structured output, nodes without IDs and relationships without type,
124
+ source_node_id, or target_node_id will be filtered out.
125
+ """
126
+ def validate_and_get_relationship_type(allowed_relationships: list[PrimitiveRelationshipType] | None, allowed_nodes: list[PrimitiveNodeType] | None) -> RelationshipTypeFormat | None:
127
+ '''Validate relationship type format and return the format type.
128
+
129
+ This utility function validates that the allowed_relationships parameter
130
+ is in one of the expected formats (list of strings or list of tuples)
131
+ and returns the detected format type as an enum.
132
+
133
+ Args:
134
+ allowed_relationships (list[PrimitiveRelationshipType] | None, optional): List of allowed relationship types,
135
+ either as:
136
+ - A list of strings (e.g., ["WORKS_AT", "FRIEND_OF"])
137
+ - A list of 3-tuples in the format (source_type, relationship_type, target_type)
138
+ (e.g., [("Person", "WORKS_AT", "Company")])
139
+ allowed_nodes (list[PrimitiveNodeType] | None, optional): Optional list of allowed node types. Required when
140
+ allowed_relationships is a list of tuples to validate that source and
141
+ target node types are in the allowed_nodes list. Defaults to None.
142
+
143
+ Returns:
144
+ RelationshipTypeFormat | None: The detected format type:
145
+ - RelationshipTypeFormat.STRING if allowed_relationships is a list of strings
146
+ - RelationshipTypeFormat.TUPLE if allowed_relationships is a list of 3-tuples
147
+ - None if allowed_relationships is empty or None
148
+
149
+ Raises:
150
+ ValueError: If allowed_relationships is not a list, or if it contains invalid
151
+ formats, or if tuple source/target types are not in allowed_nodes.
152
+ '''
153
+ def filter_relationships_by_existing_nodes(nodes: list[Node], relationships: list[Relationship]) -> list[Relationship]:
154
+ """Remove relationships where source or target node is not in nodes.
155
+
156
+ Args:
157
+ nodes (list[Node]): List of nodes.
158
+ relationships (list[Relationship]): List of relationships.
159
+
160
+ Returns:
161
+ list[Relationship]: List of relationships where source and target node is in nodes.
162
+ """
163
+ def filter_graph_document_by_allowed_nodes_and_relationships(graph_document: GraphDocument, allowed_nodes: list[str] | None, allowed_relationships: list[str] | list[tuple[str, str, str]] | None, relationship_type: RelationshipTypeFormat | None) -> GraphDocument:
164
+ """Filter graph document by allowed nodes and relationships.
165
+
166
+ This implementation performs filtering with case-insensitive comparison.
167
+
168
+ Args:
169
+ graph_document (GraphDocument): Graph document to filter.
170
+ allowed_nodes (list[str] | None): List of allowed node types.
171
+ allowed_relationships (list[str] | list[tuple[str, str, str]] | None): List of allowed relationship types.
172
+ relationship_type (str | None): Type of relationship.
173
+
174
+ Returns:
175
+ GraphDocument: Filtered graph document.
176
+ """
Binary file
gllm_misc.pyi ADDED
@@ -0,0 +1,35 @@
1
+ # This file was generated by Nuitka
2
+
3
+ # Stubs included by default
4
+ from __future__ import annotations
5
+
6
+
7
+ __name__ = ...
8
+
9
+
10
+
11
+ # Modules used internally, to allow implicit dependencies to be seen:
12
+ import os
13
+ import hashlib
14
+ import json
15
+ import enum
16
+ import typing
17
+ import gllm_core
18
+ import gllm_core.schema
19
+ import gllm_datastore
20
+ import gllm_datastore.cache
21
+ import gllm_datastore.cache.hybrid_cache
22
+ import gllm_datastore.cache.hybrid_cache.hybrid_cache
23
+ import asyncio
24
+ import gllm_core.utils
25
+ import gllm_core.utils.logger_manager
26
+ import gllm_inference
27
+ import gllm_inference.lm_invoker
28
+ import gllm_inference.lm_invoker.lm_invoker
29
+ import gllm_inference.prompt_builder
30
+ import gllm_inference.schema
31
+ import json_repair
32
+ import collections
33
+ import collections.defaultdict
34
+ import collections.deque
35
+ import pydantic
@@ -0,0 +1,152 @@
1
+ Metadata-Version: 2.2
2
+ Name: gllm-misc-binary
3
+ Version: 0.9.6
4
+ Summary: A library containing miscellaneous components for Gen AI applications.
5
+ Author-email: Dimitrij Ray <dimitrij.ray@gdplabs.id>, Henry Wicaksono <henry.wicaksono@gdplabs.id>, Resti Febriana <resti.febriana@gdplabs.id>, Kadek Denaya <kadek.d.r.diana@gdplabs.id>
6
+ Requires-Python: <3.14,>=3.11
7
+ Description-Content-Type: text/markdown
8
+ Requires-Dist: gllm-core-binary<0.5.0,>=0.3.0
9
+ Requires-Dist: gllm-inference-binary[google]<0.7.0,>=0.5.0
10
+ Requires-Dist: gllm-datastore-binary[chroma]<0.6.0,>=0.5.0
11
+ Provides-Extra: dev
12
+ Requires-Dist: coverage<8.0.0,>=7.4.4; extra == "dev"
13
+ Requires-Dist: mypy<2.0.0,>=1.15.0; extra == "dev"
14
+ Requires-Dist: pre-commit<4.0.0,>=3.7.0; extra == "dev"
15
+ Requires-Dist: pytest<10.0.0,>=9.0.3; extra == "dev"
16
+ Requires-Dist: pytest-asyncio<2.0.0,>=1.0.0; extra == "dev"
17
+ Requires-Dist: pytest-cov<6.0.0,>=5.0.0; extra == "dev"
18
+ Requires-Dist: ruff<0.7.0,>=0.6.7; extra == "dev"
19
+ Provides-Extra: json-repair
20
+ Requires-Dist: json-repair<1.0.0,>=0.47.6; extra == "json-repair"
21
+
22
+ # GLLM Misc
23
+
24
+ ## Description
25
+ A library containing miscellaneous utilities and helper functions for Generative AI applications.
26
+
27
+ ---
28
+
29
+ ## Installation
30
+
31
+ ### Prerequisites
32
+
33
+ Mandatory:
34
+ 1. Python 3.11+ — [Install here](https://www.python.org/downloads/)
35
+ 2. pip — [Install here](https://pip.pypa.io/en/stable/installation/)
36
+ 3. uv — [Install here](https://docs.astral.sh/uv/getting-started/installation/)
37
+ 4. gcloud CLI (for authentication) — [Install here](https://cloud.google.com/sdk/docs/install), then log in using:
38
+ ```bash
39
+ gcloud auth login
40
+ ```
41
+
42
+ ---
43
+
44
+ ### Install from Artifact Registry
45
+
46
+ This requires authentication via the `gcloud` CLI.
47
+
48
+ ```bash
49
+ uv pip install \
50
+ --extra-index-url "https://oauth2accesstoken:$(gcloud auth print-access-token)@glsdk.gdplabs.id/gen-ai-internal/simple/" \
51
+ gllm-misc
52
+ ```
53
+
54
+ ---
55
+
56
+ ## Local Development Setup
57
+
58
+ ### Prerequisites
59
+
60
+ 1. Python 3.11+ — [Install here](https://www.python.org/downloads/)
61
+ 2. pip — [Install here](https://pip.pypa.io/en/stable/installation/)
62
+ 3. uv — [Install here](https://docs.astral.sh/uv/getting-started/installation/)
63
+ 4. gcloud CLI — [Install here](https://cloud.google.com/sdk/docs/install), then log in using:
64
+
65
+ ```bash
66
+ gcloud auth login
67
+ ```
68
+ 5. Git — [Install here](https://git-scm.com/downloads)
69
+ 6. Access to the [GDP Labs SDK GitHub repository](https://github.com/GDP-ADMIN/gl-sdk)
70
+
71
+ ---
72
+
73
+ ### 1. Clone Repository
74
+
75
+ ```bash
76
+ git clone git@github.com:GDP-ADMIN/gl-sdk.git
77
+ cd gl-sdk/libs/gllm-misc
78
+ ```
79
+
80
+ ---
81
+
82
+ ### 2. Setup Authentication
83
+
84
+ Set the following environment variables to authenticate with internal package indexes:
85
+
86
+ ```bash
87
+ export UV_INDEX_GEN_AI_INTERNAL_USERNAME=oauth2accesstoken
88
+ export UV_INDEX_GEN_AI_INTERNAL_PASSWORD="$(gcloud auth print-access-token)"
89
+ export UV_INDEX_GEN_AI_USERNAME=oauth2accesstoken
90
+ export UV_INDEX_GEN_AI_PASSWORD="$(gcloud auth print-access-token)"
91
+ ```
92
+
93
+ ---
94
+
95
+ ### 3. Quick Setup
96
+
97
+ Run:
98
+
99
+ ```bash
100
+ make setup
101
+ ```
102
+
103
+ ---
104
+
105
+ ### 4. Activate Virtual Environment
106
+
107
+ ```bash
108
+ source .venv/bin/activate
109
+ ```
110
+
111
+ ---
112
+
113
+ ## Local Development Utilities
114
+
115
+ The following Makefile commands are available for quick operations:
116
+
117
+ ### Install uv
118
+
119
+ ```bash
120
+ make install-uv
121
+ ```
122
+
123
+ ### Install Pre-Commit
124
+
125
+ ```bash
126
+ make install-pre-commit
127
+ ```
128
+
129
+ ### Install Dependencies
130
+
131
+ ```bash
132
+ make install
133
+ ```
134
+
135
+ ### Update Dependencies
136
+
137
+ ```bash
138
+ make update
139
+ ```
140
+
141
+ ### Run Tests
142
+
143
+ ```bash
144
+ make test
145
+ ```
146
+
147
+ ---
148
+
149
+ ## Contributing
150
+
151
+ Please refer to the [Python Style Guide](https://docs.google.com/document/d/1uRggCrHnVfDPBnG641FyQBwUwLoFw0kTzNqRm92vUwM/edit?usp=sharing)
152
+ for information about code style, documentation standards, and SCA requirements.
@@ -0,0 +1,16 @@
1
+ gllm_misc.cp313-win_amd64.pyd,sha256=1HsOL2e8t8vxNCr5Cu9rklmDyihcWfgvxFp7cMazC_w,627200
2
+ gllm_misc.pyi,sha256=TzFLLF1rrJDtKBMH_SY8l0xnahr12pHfKdetY0FoaQI,789
3
+ gllm_misc/__init__.pyi,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
4
+ gllm_misc/cache_manager/__init__.pyi,sha256=xlTbERx84PBt7weWfQ_ef2mLpHtUDjvbVVeBXcFCOUA,174
5
+ gllm_misc/cache_manager/cache_manager.pyi,sha256=BOc_6WahW9sESXHf3c3dURirro6TeOu2qm6ncSVpoF0,2497
6
+ gllm_misc/graph_transformer/__init__.pyi,sha256=rxzbZy51KZntL5VuX3VPv5tD-Llnu8lOPMf6jTN4G-U,478
7
+ gllm_misc/graph_transformer/constant.pyi,sha256=2YtEYPJAhLzTVBLMhAqUQ7gUTkww1hirtO4pT7PO2fc,348
8
+ gllm_misc/graph_transformer/lm_based_graph_transformer.pyi,sha256=TOJMDvCejPX2MxuwmebfIij5fY-btCIImMo6dB7blok,7712
9
+ gllm_misc/graph_transformer/lm_based_mindmap_transformer.pyi,sha256=Y8j4WJ3ay0AR-avideS8OJSAi1SWZFgS4pti8FxPN5M,7924
10
+ gllm_misc/graph_transformer/prompt.pyi,sha256=NhiOmUC5Ya37OabxBADfb6XLgB2vMdHbd8hnkBO0sM4,1250
11
+ gllm_misc/graph_transformer/schema.pyi,sha256=SQttxH_YnVGH6DZbVInUgFPiVwEkLwcBsjP2yxzTz0o,3524
12
+ gllm_misc/graph_transformer/utils.pyi,sha256=gXzW-kXpKM4x8Pi9n0uKlaf4aQ67kO8xXZ-_Kja2PYg,9849
13
+ gllm_misc_binary-0.9.6.dist-info/METADATA,sha256=a2x9kfbjyCZo0o1pfePD_cLl4hrovsu5oZ4JXyrmD6M,3815
14
+ gllm_misc_binary-0.9.6.dist-info/WHEEL,sha256=ikXeVivtRq-oqOZfN0jwm7M68yNYx-wxpXplGP7L4Is,97
15
+ gllm_misc_binary-0.9.6.dist-info/top_level.txt,sha256=qauofm35pMgS3PoJTXiZF-eZWLfbegOgp5d9-tCUa6w,10
16
+ gllm_misc_binary-0.9.6.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: Nuitka (2.8.10)
3
+ Root-Is-Purelib: false
4
+ Tag: cp313-cp313-win_amd64
5
+
@@ -0,0 +1 @@
1
+ gllm_misc