PyPI - datachain - Versions diffs - 0.2.11__py3-none-any.whl → 0.2.12__py3-none-any.whl - Mend

datachain 0.2.11py3-none-any.whl → 0.2.12py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.

This version of datachain might be problematic. Click here for more details.

Files changed (46) hide show

datachain/__init__.py +3 -4
datachain/cache.py +10 -4
datachain/catalog/catalog.py +35 -15
datachain/cli.py +37 -32
datachain/data_storage/metastore.py +24 -0
datachain/data_storage/warehouse.py +3 -1
datachain/job.py +56 -0
datachain/lib/arrow.py +19 -7
datachain/lib/clip.py +89 -66
datachain/lib/convert/{type_converter.py → python_to_sql.py} +6 -6
datachain/lib/convert/sql_to_python.py +23 -0
datachain/lib/convert/values_to_tuples.py +51 -33
datachain/lib/data_model.py +6 -27
datachain/lib/dataset_info.py +70 -0
datachain/lib/dc.py +618 -156
datachain/lib/file.py +117 -15
datachain/lib/image.py +1 -1
datachain/lib/meta_formats.py +14 -2
datachain/lib/model_store.py +3 -2
datachain/lib/pytorch.py +10 -7
datachain/lib/signal_schema.py +19 -11
datachain/lib/text.py +2 -1
datachain/lib/udf.py +56 -5
datachain/lib/udf_signature.py +1 -1
datachain/node.py +11 -8
datachain/query/dataset.py +52 -26
datachain/query/schema.py +2 -0
datachain/query/session.py +4 -4
datachain/sql/functions/array.py +12 -0
datachain/sql/functions/string.py +8 -0
datachain/torch/__init__.py +1 -1
datachain/utils.py +6 -0
datachain-0.2.12.dist-info/METADATA +412 -0
{datachain-0.2.11.dist-info → datachain-0.2.12.dist-info}/RECORD +38 -42
{datachain-0.2.11.dist-info → datachain-0.2.12.dist-info}/WHEEL +1 -1
datachain/lib/gpt4_vision.py +0 -97
datachain/lib/hf_image_to_text.py +0 -97
datachain/lib/hf_pipeline.py +0 -90
datachain/lib/image_transform.py +0 -103
datachain/lib/iptc_exif_xmp.py +0 -76
datachain/lib/unstructured.py +0 -41
datachain/text/__init__.py +0 -3
datachain-0.2.11.dist-info/METADATA +0 -431
{datachain-0.2.11.dist-info → datachain-0.2.12.dist-info}/LICENSE +0 -0
{datachain-0.2.11.dist-info → datachain-0.2.12.dist-info}/entry_points.txt +0 -0
{datachain-0.2.11.dist-info → datachain-0.2.12.dist-info}/top_level.txt +0 -0

datachain-0.2.12.dist-info/METADATA ADDED Viewed

@@ -0,0 +1,412 @@
+Metadata-Version: 2.1
+Name: datachain
+Version: 0.2.12
+Summary: Wrangle unstructured AI data at scale
+Author-email: Dmitry Petrov <support@dvc.org>
+License: Apache-2.0
+Project-URL: Documentation, https://datachain.dvc.ai
+Project-URL: Issues, https://github.com/iterative/datachain/issues
+Project-URL: Source, https://github.com/iterative/datachain
+Classifier: Programming Language :: Python :: 3
+Classifier: Programming Language :: Python :: 3.9
+Classifier: Programming Language :: Python :: 3.10
+Classifier: Programming Language :: Python :: 3.11
+Classifier: Programming Language :: Python :: 3.12
+Classifier: Development Status :: 2 - Pre-Alpha
+Requires-Python: >=3.9
+Description-Content-Type: text/x-rst
+License-File: LICENSE
+Requires-Dist: pyyaml
+Requires-Dist: tomlkit
+Requires-Dist: tqdm
+Requires-Dist: numpy
+Requires-Dist: pandas >=2.0.0
+Requires-Dist: pyarrow
+Requires-Dist: typing-extensions
+Requires-Dist: python-dateutil >=2
+Requires-Dist: attrs >=21.3.0
+Requires-Dist: s3fs >=2024.2.0
+Requires-Dist: gcsfs >=2024.2.0
+Requires-Dist: adlfs >=2024.2.0
+Requires-Dist: dvc-data <4,>=3.10
+Requires-Dist: dvc-objects <6,>=4
+Requires-Dist: shtab <2,>=1.3.4
+Requires-Dist: sqlalchemy >=2
+Requires-Dist: multiprocess ==0.70.16
+Requires-Dist: dill ==0.3.8
+Requires-Dist: cloudpickle
+Requires-Dist: ujson >=5.9.0
+Requires-Dist: pydantic <3,>=2
+Requires-Dist: jmespath >=1.0
+Requires-Dist: datamodel-code-generator >=0.25
+Requires-Dist: Pillow <11,>=10.0.0
+Requires-Dist: numpy <2,>=1 ; sys_platform == "win32"
+Provides-Extra: dev
+Requires-Dist: datachain[docs,tests] ; extra == 'dev'
+Requires-Dist: mypy ==1.10.1 ; extra == 'dev'
+Requires-Dist: types-python-dateutil ; extra == 'dev'
+Requires-Dist: types-PyYAML ; extra == 'dev'
+Requires-Dist: types-requests ; extra == 'dev'
+Requires-Dist: types-ujson ; extra == 'dev'
+Provides-Extra: docs
+Requires-Dist: mkdocs >=1.5.2 ; extra == 'docs'
+Requires-Dist: mkdocs-gen-files >=0.5.0 ; extra == 'docs'
+Requires-Dist: mkdocs-material >=9.3.1 ; extra == 'docs'
+Requires-Dist: mkdocs-section-index >=0.3.6 ; extra == 'docs'
+Requires-Dist: mkdocstrings-python >=1.6.3 ; extra == 'docs'
+Requires-Dist: mkdocs-literate-nav >=0.6.1 ; extra == 'docs'
+Provides-Extra: remote
+Requires-Dist: lz4 ; extra == 'remote'
+Requires-Dist: msgpack <2,>=1.0.4 ; extra == 'remote'
+Requires-Dist: requests >=2.22.0 ; extra == 'remote'
+Provides-Extra: tests
+Requires-Dist: datachain[remote,torch,vector] ; extra == 'tests'
+Requires-Dist: pytest <9,>=8 ; extra == 'tests'
+Requires-Dist: pytest-sugar >=0.9.6 ; extra == 'tests'
+Requires-Dist: pytest-cov >=4.1.0 ; extra == 'tests'
+Requires-Dist: pytest-mock >=3.12.0 ; extra == 'tests'
+Requires-Dist: pytest-servers[all] >=0.5.5 ; extra == 'tests'
+Requires-Dist: pytest-benchmark[histogram] ; extra == 'tests'
+Requires-Dist: pytest-asyncio >=0.23.2 ; extra == 'tests'
+Requires-Dist: pytest-xdist >=3.3.1 ; extra == 'tests'
+Requires-Dist: virtualenv ; extra == 'tests'
+Requires-Dist: dulwich ; extra == 'tests'
+Requires-Dist: hypothesis ; extra == 'tests'
+Requires-Dist: open-clip-torch ; extra == 'tests'
+Requires-Dist: aiotools >=1.7.0 ; extra == 'tests'
+Requires-Dist: requests-mock ; extra == 'tests'
+Provides-Extra: torch
+Requires-Dist: torch >=2.1.0 ; extra == 'torch'
+Requires-Dist: torchvision ; extra == 'torch'
+Requires-Dist: transformers >=4.36.0 ; extra == 'torch'
+Provides-Extra: vector
+Requires-Dist: usearch ; extra == 'vector'
+|PyPI| |Python Version| |Codecov| |Tests|
+.. |PyPI| image:: https://img.shields.io/pypi/v/datachain.svg
+   :target: https://pypi.org/project/datachain/
+   :alt: PyPI
+.. |Python Version| image:: https://img.shields.io/pypi/pyversions/datachain
+   :target: https://pypi.org/project/datachain
+   :alt: Python Version
+.. |Codecov| image:: https://codecov.io/gh/iterative/datachain/graph/badge.svg?token=byliXGGyGB
+   :target: https://codecov.io/gh/iterative/datachain
+   :alt: Codecov
+.. |Tests| image:: https://github.com/iterative/datachain/actions/workflows/tests.yml/badge.svg
+   :target: https://github.com/iterative/datachain/actions/workflows/tests.yml
+   :alt: Tests
+AI 🔗 DataChain
+----------------
+DataChain is an open-source Python library for processing and curating unstructured
+data at scale.
+🤖 AI-Driven Data Curation: Use local ML models, LLM APIs calls to enrich your data.
+🚀 GenAI Dataset scale: Handle 10s of milions of files or file snippets.
+🐍 Python-friendly: Use strictly typed `Pydantic`_ objects instead of JSON.
+To ensure efficiency, Datachain supports parallel processing, parallel data
+downloads, and out-of-memory computing. It excels at optimizing batch operations.
+While most GenAI tools focus on online applications and realtime, DataChain is designed
+for offline data processing, data curation and ETL.
+The typical use cases are Computer Vision data curation, LLM analytics
+and validation.
+.. code:: console
+   $ pip install datachain
+|Flowchart|
+Quick Start
+-----------
+Basic evaluation
+================
+We will evaluate chatbot dialogs stored as text files in Google Cloud Storage
+- 50 files total in the example.
+These dialogs involve users looking for better wireless plans chatting with bot.
+Our goal is to identify successful dialogs.
+The data used in the examples is publicly available. Please feel free to run this code.
+First, we'll use a simple sentiment analysis model. Please install transformers.
+.. code:: shell
+    pip install transformers
+The code below downloads files the cloud, applies function
+`is_positive_dialogue_ending()` to each. All files with a positive sentiment
+are copied to local directory `output/`.
+.. code:: py
+    from transformers import pipeline
+    from datachain import DataChain, Column
+    classifier = pipeline("sentiment-analysis", device="cpu",
+                    model="distilbert/distilbert-base-uncased-finetuned-sst-2-english")
+    def is_positive_dialogue_ending(file) -> bool:
+        dialogue_ending = file.read()[-512:]
+        return classifier(dialogue_ending)[0]["label"] == "POSITIVE"
+    chain = (
+       DataChain.from_storage("gs://datachain-demo/chatbot-KiT/",
+                              object_name="file", type="text")
+       .settings(parallel=8, cache=True)
+       .map(is_positive=is_positive_dialogue_ending)
+       .save("file_response")
+    )
+    positive_chain = chain.filter(Column("is_positive") == True)
+    positive_chain.export_files("./output1")
+    print(f"{positive_chain.count()} files were exported")
+13 files were exported
+.. code:: shell
+    $ ls output/datachain-demo/chatbot-KiT/
+    15.txt 20.txt 24.txt 27.txt 28.txt 29.txt 33.txt 37.txt 38.txt 43.txt ...
+    $ ls output/datachain-demo/chatbot-KiT/ | wc -l
+    13
+LLM judging LLMs dialogs
+==========================
+Finding good dialogs using an LLM can be more efficient. In this example,
+we use Mistral with a free API. Please install the package and get a free
+Mistral API key at https://console.mistral.ai
+.. code:: shell
+    $ pip install mistralai
+    $ export MISTRAL_API_KEY=_your_key_
+Below is a similar code example, but this time using an LLM to evaluate the dialogs.
+Note, only 4 threads were used in this example `parallel=4` due to a limitation of
+the free LLM service.
+.. code:: py
+    from mistralai.client import MistralClient
+    from mistralai.models.chat_completion import ChatMessage
+    from datachain import File, DataChain, Column
+    PROMPT = "Was this dialog successful? Answer in a single word: Success or Failure."
+    def eval_dialogue(file: File) -> bool:
+         client = MistralClient()
+         response = client.chat(
+             model="open-mixtral-8x22b",
+             messages=[ChatMessage(role="system", content=PROMPT),
+                       ChatMessage(role="user", content=file.read())])
+         result = response.choices[0].message.content
+         return result.lower().startswith("success")
+    chain = (
+       DataChain.from_storage("gs://datachain-demo/chatbot-KiT/", object_name="file")
+       .settings(parallel=4, cache=True)
+       .map(is_success=eval_dialogue)
+       .save("mistral_files")
+    )
+    successful_chain = chain.filter(Column("is_success") == True)
+    successful_chain.export_files("./output_mistral")
+    print(f"{successful_chain.count()} files were exported")
+With the current prompt, we found 31 files considered successful dialogs:
+.. code:: shell
+    $ ls output_mistral/datachain-demo/chatbot-KiT/
+    1.txt  15.txt 18.txt 2.txt  22.txt 25.txt 28.txt 33.txt 37.txt 4.txt  41.txt ...
+    $ ls output_mistral/datachain-demo/chatbot-KiT/ | wc -l
+    31
+Serializing Python-objects
+==========================
+LLM responses contain valuable information for analytics, such as tokens used and the
+model. Preserving this information can be beneficial.
+Instead of extracting this information from the Mistral data structure (class
+`ChatCompletionResponse`), we serialize the entire Python object to the internal DB.
+.. code:: py
+    from mistralai.client import MistralClient
+    from mistralai.models.chat_completion import ChatMessage, ChatCompletionResponse
+    from datachain import File, DataChain, Column
+    PROMPT = "Was this dialog successful? Answer in a single word: Success or Failure."
+    def eval_dialog(file: File) -> ChatCompletionResponse:
+         client = MistralClient()
+         return client.chat(
+             model="open-mixtral-8x22b",
+             messages=[ChatMessage(role="system", content=PROMPT),
+                       ChatMessage(role="user", content=file.read())])
+    chain = (
+       DataChain.from_storage("gs://datachain-demo/chatbot-KiT/", object_name="file")
+       .settings(parallel=4, cache=True)
+       .map(response=eval_dialog)
+       .map(status=lambda response: response.choices[0].message.content.lower()[:7])
+       .save("response")
+    )
+    chain.select("file.name", "status", "response.usage").show(5)
+    success_rate = chain.filter(Column("status") == "success").count() / chain.count()
+    print(f"{100*success_rate:.1f}% dialogs were successful")
+Output:
+.. code:: shell
+         file   status      response     response          response
+         name                  usage        usage             usage
+                       prompt_tokens total_tokens completion_tokens
+    0   1.txt  success           547          548                 1
+    1  10.txt  failure          3576         3578                 2
+    2  11.txt  failure           626          628                 2
+    3  12.txt  failure          1144         1182                38
+    4  13.txt  success          1100         1101                 1
+    [Limited by 5 rows]
+    64.0% dialogs were successful
+Complex Python data structures
+=============================================
+In the previous examples, a few dataset were saved in the embedded database
+(`SQLite`_ in directory `.datachain`).
+These datasets are versioned, and can be accessed using
+`DataChain.from_dataset("dataset_name")`.
+.. code:: py
+    chain = DataChain.from_dataset("response")
+    # Iterating one-by-one: out of memory
+    for file, response in chain.limit(5).collect("file", "response"):
+        # You work with Python objects
+        assert isinstance(response, ChatCompletionResponse)
+        status = response.choices[0].message.content[:7]
+        tokens = response.usage.total_tokens
+        print(f"{file.get_uri()}: {status}, file size: {file.size}, tokens: {tokens}")
+Output:
+.. code:: shell
+    gs://datachain-demo/chatbot-KiT/1.txt: Success, file size: 1776, tokens: 548
+    gs://datachain-demo/chatbot-KiT/10.txt: Failure, file size: 11576, tokens: 3578
+    gs://datachain-demo/chatbot-KiT/11.txt: Failure, file size: 2045, tokens: 628
+    gs://datachain-demo/chatbot-KiT/12.txt: Failure, file size: 3833, tokens: 1207
+    gs://datachain-demo/chatbot-KiT/13.txt: Success, file size: 3657, tokens: 1101
+Vectorized analytics over Python objects
+========================================
+Some operations can be efficiently run inside the DB without deserializing Python objects.
+Let's calculate the cost of using LLM APIs in a vectorized way.
+Mistral calls cost $2 per 1M input tokens and $6 per 1M output tokens:
+.. code:: py
+    chain = DataChain.from_dataset("mistral_dataset")
+    cost = chain.sum("response.usage.prompt_tokens")*0.000002 \
+               + chain.sum("response.usage.completion_tokens")*0.000006
+    print(f"Spent ${cost:.2f} on {chain.count()} calls")
+Output:
+.. code:: shell
+    Spent $0.08 on 50 calls
+PyTorch data loader
+===================
+Chain results can be exported or passed directly to PyTorch dataloader.
+For example, if we are interested in passing image and a label based on file
+name suffix, the following code will do it:
+.. code:: py
+    from torch.utils.data import DataLoader
+    from transformers import CLIPProcessor
+    from datachain import C, DataChain
+    processor = CLIPProcessor.from_pretrained("openai/clip-vit-base-patch32")
+    chain = (
+        DataChain.from_storage("gs://datachain-demo/dogs-and-cats/", type="image")
+        .map(label=lambda name: name.split(".")[0], params=["file.name"])
+        .select("file", "label").to_pytorch(
+            transform=processor.image_processor,
+            tokenizer=processor.tokenizer,
+        )
+    )
+    loader = DataLoader(chain, batch_size=1)
+Tutorials
+---------
+* `Getting Started`_
+* `Multimodal <examples/multimodal/clip_fine_tuning.ipynb>`_ (try in `Colab <https://colab.research.google.com/github/iterative/datachain/blob/main/examples/multimodal/clip_fine_tuning.ipynb>`__)
+Contributions
+-------------
+Contributions are very welcome.
+To learn more, see the `Contributor Guide`_.
+Community and Support
+---------------------
+* `Docs <https://datachain.dvc.ai/>`_
+* `File an issue`_ if you encounter any problems
+* `Discord Chat <https://dvc.org/chat>`_
+* `Email <mailto:support@dvc.org>`_
+* `Twitter <https://twitter.com/DVCorg>`_
+.. _PyPI: https://pypi.org/
+.. _file an issue: https://github.com/iterative/datachain/issues
+.. github-only
+.. _Contributor Guide: CONTRIBUTING.rst
+.. _Pydantic: https://github.com/pydantic/pydantic
+.. _SQLite: https://www.sqlite.org/
+.. _Getting Started: https://datachain.dvc.ai/
+.. |Flowchart| image:: https://github.com/iterative/datachain/blob/main/docs/assets/flowchart.png?raw=true
+   :alt: DataChain FlowChart

{datachain-0.2.11.dist-info → datachain-0.2.12.dist-info}/RECORD RENAMED Viewed

@@ -1,22 +1,23 @@
-datachain/__init__.py,sha256=L5IlHOD4AaHkV7P5dbUwdq90I3bGFLtOghoZ1WVFGcs,841
+datachain/__init__.py,sha256=GeyhE-5LgfJav2OKYGaieP2lBvf2Gm-ihj7thnK9zjI,800
 datachain/__main__.py,sha256=hG3Y4ARGEqe1AWwNMd259rBlqtphx1Wk39YbueQ0yV8,91
 datachain/asyn.py,sha256=CKCFQJ0CbB3r04S7mUTXxriKzPnOvdUaVPXjM8vCtJw,7644
-datachain/cache.py,sha256=FaPWrqWznPffmskTb1pdPkt2jAMMf__9FC2zEnP0vDU,4022
-datachain/cli.py,sha256=gikzwEXTDKyzY1xOAUziXN2-OVqnOhDMJTd7SHq0Jxc,32406
+datachain/cache.py,sha256=N6PCEFJlWRpq7f_zeBNoaURFCJFAV7ibsLJqyiMHbBg,4207
+datachain/cli.py,sha256=MSOID2t-kesk5Z80uoepN63rqvB7iZxaWYLqkiWehkQ,32628
 datachain/cli_utils.py,sha256=jrn9ejGXjybeO1ur3fjdSiAyCHZrX0qsLLbJzN9ErPM,2418
 datachain/config.py,sha256=PfC7W5yO6HFO6-iMB4YB-0RR88LPiGmD6sS_SfVbGso,1979
 datachain/dataset.py,sha256=MZezyuJWNj_3PEtzr0epPMNyWAOTrhTSPI5FmemV6L4,14470
 datachain/error.py,sha256=GY9KYTmb7GHXn2gGHV9X-PBhgwLj3i7VpK7tGHtAoGM,1279
+datachain/job.py,sha256=bk25bIqClhgRPzlXAhxpTtDeewibQe5l3S8Cf7db0gM,1229
 datachain/listing.py,sha256=sX8vZNzAzoTel1li6VJiYeHUJwseUERVEoW9D5P7tII,8192
-datachain/node.py,sha256=fsQDJUmRMSRHhL1u6qQlWgreHbH760Ls-yDzFLhbW-U,5724
+datachain/node.py,sha256=LwzSOSM9SbPLI5RvYDsiEkk7d5rbMX8huzM_m7uWKx4,5917
 datachain/nodes_fetcher.py,sha256=kca19yvu11JxoVY1t4_ydp1FmchiV88GnNicNBQ9NIA,831
 datachain/nodes_thread_pool.py,sha256=ZyzBvUImIPmi4WlKC2SW2msA0UhtembbTdcs2nx29A0,3191
 datachain/progress.py,sha256=7_8FtJs770ITK9sMq-Lt4k4k18QmYl4yIG_kCoWID3o,4559
 datachain/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
 datachain/storage.py,sha256=RiSJLYdHUjnrEWkLBKPcETHpAxld_B2WxLg711t0aZI,3733
-datachain/utils.py,sha256=AWUXRk7yvDpHcqzzPWwzv8HtF1-jDVEBHKxAgT7u02E,12288
+datachain/utils.py,sha256=kgH5NPj47eC_KrFTd6ZS206lKVhnJVFt5XsqkK6ppTc,12483
 datachain/catalog/__init__.py,sha256=g2iAAFx_gEIrqshXlhSEbrc8qDaEH11cjU40n3CHDz4,409
-datachain/catalog/catalog.py,sha256=A5W9Ffoz1lZkzl6A3igaMC5jrus8VIYVLJLX8JTVKrk,79603
+datachain/catalog/catalog.py,sha256=u8tvWooIon9ju59q8-Re_iqflgbCB-JMZD8n2UC4iag,80397
 datachain/catalog/datasource.py,sha256=D-VWIVDCM10A8sQavLhRXdYSCG7F4o4ifswEF80_NAQ,1412
 datachain/catalog/loader.py,sha256=GJ8zhEYkC7TuaPzCsjJQ4LtTdECu-wwYzC12MikPOMQ,7307
 datachain/catalog/subclass.py,sha256=B5R0qxeTYEyVAAPM1RutBPSoXZc8L5mVVZeSGXki9Sw,2096
@@ -31,50 +32,46 @@ datachain/data_storage/__init__.py,sha256=cEOJpyu1JDZtfUupYucCDNFI6e5Wmp_Oyzq6rZ
 datachain/data_storage/db_engine.py,sha256=rgBuqJ-M1j5QyqiUQuJRewctuvRRj8LBDL54-aPEFxE,3287
 datachain/data_storage/id_generator.py,sha256=VlDALKijggegAnNMJwuMETJgnLoPYxpkrkld5DNTPQw,3839
 datachain/data_storage/job.py,sha256=w-7spowjkOa1P5fUVtJou3OltT0L48P0RYWZ9rSJ9-s,383
-datachain/data_storage/metastore.py,sha256=y-4fYvuOPnWeYxAvqhDnw6CdlTvQiurg0Gg4TaG9LR0,54074
+datachain/data_storage/metastore.py,sha256=R1Jj8dOTAex8fjehewV2vUO4VhBSjj8JQI5mM3YhVEQ,54989
 datachain/data_storage/schema.py,sha256=hUykqT-As-__WffMdWTrSZwv9k5EYYowRke3OENQ3aY,8102
 datachain/data_storage/serializer.py,sha256=6G2YtOFqqDzJf1KbvZraKGXl2XHZyVml2krunWUum5o,927
 datachain/data_storage/sqlite.py,sha256=cIYobczfH72c4l-iMkxpkgcTuuvvT8Xi64iP7Zr3Skw,25084
-datachain/data_storage/warehouse.py,sha256=UbD37_jqaM4BY2SsQaTiJre-eSa7HcPejrTp936L080,33170
+datachain/data_storage/warehouse.py,sha256=FedcsvkAphpi2tUnlcrxO4mYumiCQAcrB5XRAK9tfXQ,33288
 datachain/lib/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
-datachain/lib/arrow.py,sha256=ttSiH8Xr08zxypAa3-BNTxMO2NBuZfYICwmG1qQwvWU,3268
-datachain/lib/clip.py,sha256=YRa15Whnn6C8BMA-OAu0mYjc4h9i_n7pffRGdtfrTBA,5222
-datachain/lib/data_model.py,sha256=DpV_-1JqJptCf0w4cnzPlHm5Yl4FQaveRgVCDZFaHXs,2012
-datachain/lib/dc.py,sha256=rd-7gVcMRZ2M-O8aQhNx85H31w-kRQHpXSwtf26dSk4,35849
-datachain/lib/file.py,sha256=Uik1sq2l-uknpikH4Gdm7ZR0EcQYP2TrNg-urECjbW4,8304
-datachain/lib/gpt4_vision.py,sha256=CZ-a64olZNp9TNmLGngmbN6b02UYImzwK3dPClnjxTI,2716
-datachain/lib/hf_image_to_text.py,sha256=uVl4mnUl8gnHrJ3wfSZlxBevH-cxqOswxLArLAHxRrE,3077
-datachain/lib/hf_pipeline.py,sha256=MBFzixVa25_6QVR9RyOq8Rr9UIQ-sFVcBHducx_sZcY,2069
-datachain/lib/image.py,sha256=K0n_P7kmobWTgxe-rDbr5yY3vBrOPnseziE3DXwFFVo,2325
-datachain/lib/image_transform.py,sha256=hfgvIrSMGBx_MEXECyvrFoO1NyPBHoDb28j2lT2dsf8,2953
-datachain/lib/iptc_exif_xmp.py,sha256=rmlxjOmAP31OCgbGBAwIgd1F_6QVBoSWsOPG6UsBg_w,2007
-datachain/lib/meta_formats.py,sha256=SF7UPPe-U-1HL6DBO1NfwZLIChjkHrHasgHf5ztCUoU,6436
-datachain/lib/model_store.py,sha256=JFpI1P0WFpsO6eAU49AdWmff5T8azqLrqOMB08pYJjg,2331
-datachain/lib/pytorch.py,sha256=7fd2g0dI9zrMfRl3IVwIvXRH0v6TwSAyZGAbqKdEjcI,5505
+datachain/lib/arrow.py,sha256=WBZ4iVU0CcmCgog1wS-Nrtqhzvf2I4_QqDJtzhaECeA,3641
+datachain/lib/clip.py,sha256=16u4b_y2Y15nUS2UN_8ximMo6r_-_4IQpmct2ol-e-g,5730
+datachain/lib/data_model.py,sha256=jPYDmTYbixy4LhdToOyvldYGYZxblhp6Tn4MF-VAd-o,1495
+datachain/lib/dataset_info.py,sha256=lONGr71ozo1DS4CQEhnpKORaU4qFb6Ketv8Xm8CVm2U,2188
+datachain/lib/dc.py,sha256=KboCSSyjZ69hIpyjgza4HindFwO7L1Usxa0769N57NA,50561
+datachain/lib/file.py,sha256=xiLHaqyl4rqcBLGD62YD3aBIAOmX4EBVucxIncpRi80,11916
+datachain/lib/image.py,sha256=TgYhRhzd4nkytfFMeykQkPyzqb5Le_-tU81unVMPn4Q,2328
+datachain/lib/meta_formats.py,sha256=Z2NVH5X4N2rrj5kFxKsHKq3zD4kaRHbDCx3oiUEKYUk,6920
+datachain/lib/model_store.py,sha256=c4USXsBBjrGH8VOh4seIgOiav-qHOwdoixtxfLgU63c,2409
+datachain/lib/pytorch.py,sha256=9PsypKseyKfIimTmTQOgb-pbNXgeeAHLdlWx0qRPULY,5660
 datachain/lib/settings.py,sha256=6Nkoh8riETrftYwDp3aniK53Dsjc07MdztL8N0cW1D8,2849
-datachain/lib/signal_schema.py,sha256=mRdq5qEGnFQgbSawzDPi2MCZ6PULTMigd51B2RuNxpg,14173
-datachain/lib/text.py,sha256=d2V-52cqzVm5OT68BcLYyHrglvFMVR5DPzsbtRRv3D0,1063
-datachain/lib/udf.py,sha256=RqCiGuNKL5P8eS84s_mmVYjK1gvkuRYdnIKm9qe-i2U,9698
-datachain/lib/udf_signature.py,sha256=R81QqZseG_xeBFzJSgt-wrTQeUU-1RrWkHckLm_HEUU,7135
-datachain/lib/unstructured.py,sha256=9Y6rAelXdYqkNbPaqz6DhXjhS8d6qXcP0ieIsWkzvkk,1143
+datachain/lib/signal_schema.py,sha256=lKGlpRRUHOUFLcpk-pLQd9kGAJ8FPy0Q2bk--UlVemU,14559
+datachain/lib/text.py,sha256=dVe2Ilc_gW2EV0kun0UwegiCkapWcd20cef7CgINWHU,1083
+datachain/lib/udf.py,sha256=mo3NoyYy7fY2UZtZOtAN_jR1e5a803b1dlnD5ztduzk,11454
+datachain/lib/udf_signature.py,sha256=gMStcEeYJka5M6cg50Z9orC6y6HzCAJ3MkFqqn1fjZg,7137
 datachain/lib/utils.py,sha256=5-kJlAZE0D9nXXweAjo7-SP_AWGo28feaDByONYaooQ,463
 datachain/lib/vfile.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
 datachain/lib/webdataset.py,sha256=nIa6ubv94CwnATeeSdE7f_F9Zkz9LuBTfbXvFg3_-Ak,8295
 datachain/lib/webdataset_laion.py,sha256=PQP6tQmUP7Xu9fPuAGK1JDBYA6T5UufYMUTGaxgspJA,2118
 datachain/lib/convert/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
 datachain/lib/convert/flatten.py,sha256=XdAj0f9W32ABjOo8UyYm0y0H_yHDn3qEHERTyXuhJxk,1592
-datachain/lib/convert/type_converter.py,sha256=W-wvCIcb6OwWjRJ3EWJE4-LbpoqxsRBd6gYNpFlm8qo,2643
+datachain/lib/convert/python_to_sql.py,sha256=54G6dsMhxo1GKCzPziOqCKo2d4VRWmsJhJYRJxt1Thw,2615
+datachain/lib/convert/sql_to_python.py,sha256=HK414fexSQ4Ur-OY7_pKvDKEGdtos1CeeAFa4RxH4nU,532
 datachain/lib/convert/unflatten.py,sha256=Ogvh_5wg2f38_At_1lN0D_e2uZOOpYEvwvB2xdq56Tw,2012
-datachain/lib/convert/values_to_tuples.py,sha256=MWz9pHT-AaPQN8hNMUYfuOHstyuNv0QEckwXlKgFbLA,3088
+datachain/lib/convert/values_to_tuples.py,sha256=Bh8L4zA66XRhQxmONvLvn94_i8MBMYgfJ6A2i7l_6Jo,3592
 datachain/query/__init__.py,sha256=tv-spkjUCYamMN9ys_90scYrZ8kJ7C7d1MTYVmxGtk4,325
 datachain/query/batch.py,sha256=j-_ZcuQra2Ro3Wj4crtqQCg-7xuv-p84hr4QHdvT7as,3479
 datachain/query/builtins.py,sha256=ZKNs49t8Oa_OaboCBIEqtXZt7c1Qe9OR_C_HpoDriIU,2781
-datachain/query/dataset.py,sha256=P1KBv_R0YnKjNDHzOJwAx9qhwI08l0dLgaXfak3ps7k,60578
+datachain/query/dataset.py,sha256=m0bDQK_xXB85KPdJpH3OHdW6WJd1_PMgi01GRcWiiSg,61280
 datachain/query/dispatch.py,sha256=oGX9ZuoKWPB_EyqAZD_eULcO3OejY44_keSmFS6SHT0,13315
 datachain/query/metrics.py,sha256=vsECqbZfoSDBnvC3GQlziKXmISVYDLgHP1fMPEOtKyo,640
 datachain/query/params.py,sha256=O_j89mjYRLOwWNhYZl-z7mi-rkdP7WyFmaDufsdTryE,863
-datachain/query/schema.py,sha256=n1NBOj6JO2I26mZD4vSURmVC2rk3mjIkJQheeLogoy4,7748
-datachain/query/session.py,sha256=e4_vv4RqAjU-g3KK0avgLd9MEsmJBzRVEj1w8v7fP1k,3663
+datachain/query/schema.py,sha256=hAvux_GxUmuG_PwtnKkkizld9f0Gvt2JBzbu3m74fvE,7840
+datachain/query/session.py,sha256=am4XCNj8NlZPAYJSvh43C13dQ5NsfzzuyVDjPgYAgJE,3655
 datachain/query/udf.py,sha256=c0IOTkcedpOQEmX-Idlrrl1__1IecNXL0N9oUO9Dtkg,7755
 datachain/remote/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
 datachain/remote/studio.py,sha256=f5s6qSZ9uB4URGUoU_8_W1KZRRQQVSm6cgEBkBUEfuE,7226
@@ -85,20 +82,19 @@ datachain/sql/utils.py,sha256=rzlJw08etivdrcuQPqNVvVWhuVSyUPUQEEc6DOhu258,818
 datachain/sql/default/__init__.py,sha256=XQ2cEZpzWiABqjV-6yYHUBGI9vN_UHxbxZENESmVAWw,45
 datachain/sql/default/base.py,sha256=h44005q3qtMc9cjWmRufWwcBr5CfK_dnvG4IrcSQs_8,536
 datachain/sql/functions/__init__.py,sha256=PP8XV1CC1naIu87fiExbJRpV0Rww47EcDrDIKJb_xBQ,368
-datachain/sql/functions/array.py,sha256=vgTXFmBTq5-QW3Z8oDo4cFNi0B8zBqQnCRTQQKlp_VU,899
+datachain/sql/functions/array.py,sha256=rvH27SWN9gdh_mFnp0GIiXuCrNW6n8ZbY4I_JUS-_e0,1140
 datachain/sql/functions/conditional.py,sha256=q7YUKfunXeEldXaxgT-p5pUTcOEVU_tcQ2BJlquTRPs,207
 datachain/sql/functions/path.py,sha256=zixpERotTFP6LZ7I4TiGtyRA8kXOoZmH1yzH9oRW0mg,1294
 datachain/sql/functions/random.py,sha256=vBwEEj98VH4LjWixUCygQ5Bz1mv1nohsCG0-ZTELlVg,271
-datachain/sql/functions/string.py,sha256=DsyY6ZMAUqmZVRSla-BJLsLYNsIgLOh4XLR3yvYJUbE,505
+datachain/sql/functions/string.py,sha256=hIrF1fTvlPamDtm8UMnWDcnGfbbjCsHxZXS30U2Rzxo,651
 datachain/sql/sqlite/__init__.py,sha256=TAdJX0Bg28XdqPO-QwUVKy8rg78cgMileHvMNot7d04,166
 datachain/sql/sqlite/base.py,sha256=nPMF6_FF04hclDNZev_YfxMgbJAsWEdF-rU2pUhqBtc,12048
 datachain/sql/sqlite/types.py,sha256=oP93nLfTBaYnN0z_4Dsv-HZm8j9rrUf1esMM-z3JLbg,1754
 datachain/sql/sqlite/vector.py,sha256=ncW4eu2FlJhrP_CIpsvtkUabZlQdl2D5Lgwy_cbfqR0,469
-datachain/text/__init__.py,sha256=-yxHL2gVl3H0Zxam6iWUO6F1Mc4QAFHX6z-5fjHND74,72
-datachain/torch/__init__.py,sha256=9QJW8h0FevIXEykRsxQ7XzMDXvdIkv3kVf_UY95CTyg,600
-datachain-0.2.11.dist-info/LICENSE,sha256=8DnqK5yoPI_E50bEg_zsHKZHY2HqPy4rYN338BHQaRA,11344
-datachain-0.2.11.dist-info/METADATA,sha256=OVKgVc-Wc75AAQIY6hGL1CEBmnwksfgOXfiUen_xAOM,16759
-datachain-0.2.11.dist-info/WHEEL,sha256=FZ75kcLy9M91ncbIgG8dnpCncbiKXSRGJ_PFILs6SFg,91
-datachain-0.2.11.dist-info/entry_points.txt,sha256=0GMJS6B_KWq0m3VT98vQI2YZodAMkn4uReZ_okga9R4,49
-datachain-0.2.11.dist-info/top_level.txt,sha256=lZPpdU_2jJABLNIg2kvEOBi8PtsYikbN1OdMLHk8bTg,10
-datachain-0.2.11.dist-info/RECORD,,
+datachain/torch/__init__.py,sha256=gIS74PoEPy4TB3X6vx9nLO0Y3sLJzsA8ckn8pRWihJM,579
+datachain-0.2.12.dist-info/LICENSE,sha256=8DnqK5yoPI_E50bEg_zsHKZHY2HqPy4rYN338BHQaRA,11344
+datachain-0.2.12.dist-info/METADATA,sha256=QfDhY5jkblcb94A5CxT-ELhDcwDzZq1ju4cPQXHDEkY,14333
+datachain-0.2.12.dist-info/WHEEL,sha256=Wyh-_nZ0DJYolHNn1_hMa4lM7uDedD_RGVwbmTjyItk,91
+datachain-0.2.12.dist-info/entry_points.txt,sha256=0GMJS6B_KWq0m3VT98vQI2YZodAMkn4uReZ_okga9R4,49
+datachain-0.2.12.dist-info/top_level.txt,sha256=lZPpdU_2jJABLNIg2kvEOBi8PtsYikbN1OdMLHk8bTg,10
+datachain-0.2.12.dist-info/RECORD,,

{datachain-0.2.11.dist-info → datachain-0.2.12.dist-info}/WHEEL RENAMED Viewed

@@ -1,5 +1,5 @@
 Wheel-Version: 1.0
-Generator: setuptools (71.0.1)
+Generator: setuptools (71.1.0)
 Root-Is-Purelib: true
 Tag: py3-none-any

datachain/lib/gpt4_vision.py DELETED Viewed

@@ -1,97 +0,0 @@
-import base64
-import io
-import os
-import requests
-from PIL import Image, ImageOps, UnidentifiedImageError
-from datachain.query import Object, udf
-from datachain.sql.types import String
-DEFAULT_FIT_BOX = (500, 500)
-DEFAULT_TOKENS = 300
-def encode_image(raw):
-    try:
-        img = Image.open(raw)
-    except UnidentifiedImageError:
-        return None
-    img.load()
-    img = ImageOps.fit(img, DEFAULT_FIT_BOX)
-    output = io.BytesIO()
-    img.save(output, format="JPEG")
-    hex_data = output.getvalue()
-    return base64.b64encode(hex_data).decode("utf-8")
-@udf(
-    params=(Object(encode_image),),  # Columns consumed by the UDF.
-    output={
-        "description": String,
-        "error": String,
-    },  # Signals being returned by the UDF.
-    method="image_description",
-)
-class DescribeImage:
-    def __init__(
-        self,
-        prompt="What is in this image?",
-        max_tokens=DEFAULT_TOKENS,
-        key="",
-        timeout=30,
-    ):
-        if not key:
-            key = os.getenv("OPENAI_API_KEY", "")
-            if not key:
-                raise ValueError(
-                    "No key found. Please pass key or set the OPENAI_API_KEY "
-                    "environment variable."
-                )
-        self.prompt = prompt
-        self.max_tokens = max_tokens
-        self.headers = {
-            "Content-Type": "application/json",
-            "Authorization": f"Bearer {key}",
-        }
-        self.timeout = timeout
-    def image_description(self, base64_image):
-        if base64_image is None:
-            return ("", "Unknown image format")
-        payload = {
-            "model": "gpt-4-vision-preview",
-            "messages": [
-                {
-                    "role": "user",
-                    "content": [
-                        {"type": "text", "text": self.prompt},
-                        {
-                            "type": "image_url",
-                            "image_url": {
-                                "url": f"data:image/jpeg;base64,{base64_image}"
-                            },
-                        },
-                    ],
-                }
-            ],
-            "max_tokens": self.max_tokens,
-        }
-        response = requests.post(
-            "https://api.openai.com/v1/chat/completions",
-            headers=self.headers,
-            json=payload,
-            timeout=self.timeout,
-        )
-        json_response = response.json()
-        if "error" in json_response:
-            error = str(json_response["error"])
-            openai_description = ""
-        else:
-            error = ""
-            openai_description = json_response["choices"][0]["message"]["content"]
-        return (openai_description, error)

datachain 0.2.11__py3-none-any.whl → 0.2.12__py3-none-any.whl

Potentially problematic release.

datachain 0.2.11py3-none-any.whl → 0.2.12py3-none-any.whl