update sql in batch (#24801)

Co-authored-by: autofix-ci[bot] <114827586+autofix-ci[bot]@users.noreply.github.com> Co-authored-by: -LAN- <laipz8200@outlook.com>
2025-09-10 14:00:17 +09:00
parent b51c724a94
commit cbc0e639e4
49 changed files with 281 additions and 277 deletions
--- a/api/tasks/annotation/enable_annotation_reply_task.py
+++ b/api/tasks/annotation/enable_annotation_reply_task.py
@@ -3,6 +3,7 @@ import time

 import click
 from celery import shared_task
+from sqlalchemy import select

 from core.rag.datasource.vdb.vector_factory import Vector
 from core.rag.models.document import Document
@@ -39,7 +40,7 @@ def enable_annotation_reply_task(
        db.session.close()
        return

-    annotations = db.session.query(MessageAnnotation).where(MessageAnnotation.app_id == app_id).all()
+    annotations = db.session.scalars(select(MessageAnnotation).where(MessageAnnotation.app_id == app_id)).all()
    enable_app_annotation_key = f"enable_app_annotation_{str(app_id)}"
    enable_app_annotation_job_key = f"enable_app_annotation_job_{str(job_id)}"

--- a/api/tasks/batch_clean_document_task.py
+++ b/api/tasks/batch_clean_document_task.py
@@ -3,6 +3,7 @@ import time

 import click
 from celery import shared_task
+from sqlalchemy import select

 from core.rag.index_processor.index_processor_factory import IndexProcessorFactory
 from core.tools.utils.web_reader_tool import get_image_upload_file_ids
@@ -34,7 +35,9 @@ def batch_clean_document_task(document_ids: list[str], dataset_id: str, doc_form
        if not dataset:
            raise Exception("Document has no dataset")

-        segments = db.session.query(DocumentSegment).where(DocumentSegment.document_id.in_(document_ids)).all()
+        segments = db.session.scalars(
+            select(DocumentSegment).where(DocumentSegment.document_id.in_(document_ids))
+        ).all()
        # check segment is exist
        if segments:
            index_node_ids = [segment.index_node_id for segment in segments]
@@ -59,7 +62,7 @@ def batch_clean_document_task(document_ids: list[str], dataset_id: str, doc_form

            db.session.commit()
        if file_ids:
-            files = db.session.query(UploadFile).where(UploadFile.id.in_(file_ids)).all()
+            files = db.session.scalars(select(UploadFile).where(UploadFile.id.in_(file_ids))).all()
            for file in files:
                try:
                    storage.delete(file.key)
--- a/api/tasks/clean_dataset_task.py
+++ b/api/tasks/clean_dataset_task.py
@@ -3,6 +3,7 @@ import time

 import click
 from celery import shared_task
+from sqlalchemy import select

 from core.rag.index_processor.index_processor_factory import IndexProcessorFactory
 from core.tools.utils.web_reader_tool import get_image_upload_file_ids
@@ -55,8 +56,8 @@ def clean_dataset_task(
            index_struct=index_struct,
            collection_binding_id=collection_binding_id,
        )
-        documents = db.session.query(Document).where(Document.dataset_id == dataset_id).all()
-        segments = db.session.query(DocumentSegment).where(DocumentSegment.dataset_id == dataset_id).all()
+        documents = db.session.scalars(select(Document).where(Document.dataset_id == dataset_id)).all()
+        segments = db.session.scalars(select(DocumentSegment).where(DocumentSegment.dataset_id == dataset_id)).all()

        # Enhanced validation: Check if doc_form is None, empty string, or contains only whitespace
        # This ensures all invalid doc_form values are properly handled
--- a/api/tasks/clean_document_task.py
+++ b/api/tasks/clean_document_task.py
@@ -4,6 +4,7 @@ from typing import Optional

 import click
 from celery import shared_task
+from sqlalchemy import select

 from core.rag.index_processor.index_processor_factory import IndexProcessorFactory
 from core.tools.utils.web_reader_tool import get_image_upload_file_ids
@@ -35,7 +36,7 @@ def clean_document_task(document_id: str, dataset_id: str, doc_form: str, file_i
        if not dataset:
            raise Exception("Document has no dataset")

-        segments = db.session.query(DocumentSegment).where(DocumentSegment.document_id == document_id).all()
+        segments = db.session.scalars(select(DocumentSegment).where(DocumentSegment.document_id == document_id)).all()
        # check segment is exist
        if segments:
            index_node_ids = [segment.index_node_id for segment in segments]
--- a/api/tasks/clean_notion_document_task.py
+++ b/api/tasks/clean_notion_document_task.py
@@ -3,6 +3,7 @@ import time

 import click
 from celery import shared_task
+from sqlalchemy import select

 from core.rag.index_processor.index_processor_factory import IndexProcessorFactory
 from extensions.ext_database import db
@@ -34,7 +35,9 @@ def clean_notion_document_task(document_ids: list[str], dataset_id: str):
            document = db.session.query(Document).where(Document.id == document_id).first()
            db.session.delete(document)

-            segments = db.session.query(DocumentSegment).where(DocumentSegment.document_id == document_id).all()
+            segments = db.session.scalars(
+                select(DocumentSegment).where(DocumentSegment.document_id == document_id)
+            ).all()
            index_node_ids = [segment.index_node_id for segment in segments]

            index_processor.clean(dataset, index_node_ids, with_keywords=True, delete_child_chunks=True)
--- a/api/tasks/deal_dataset_vector_index_task.py
+++ b/api/tasks/deal_dataset_vector_index_task.py
@@ -4,6 +4,7 @@ from typing import Literal

 import click
 from celery import shared_task
+from sqlalchemy import select

 from core.rag.index_processor.constant.index_type import IndexType
 from core.rag.index_processor.index_processor_factory import IndexProcessorFactory
@@ -36,16 +37,14 @@ def deal_dataset_vector_index_task(dataset_id: str, action: Literal["remove", "a
        if action == "remove":
            index_processor.clean(dataset, None, with_keywords=False)
        elif action == "add":
-            dataset_documents = (
-                db.session.query(DatasetDocument)
-                .where(
+            dataset_documents = db.session.scalars(
+                select(DatasetDocument).where(
                    DatasetDocument.dataset_id == dataset_id,
                    DatasetDocument.indexing_status == "completed",
                    DatasetDocument.enabled == True,
                    DatasetDocument.archived == False,
                )
-                .all()
-            )
+            ).all()

            if dataset_documents:
                dataset_documents_ids = [doc.id for doc in dataset_documents]
@@ -89,16 +88,14 @@ def deal_dataset_vector_index_task(dataset_id: str, action: Literal["remove", "a
                        )
                        db.session.commit()
        elif action == "update":
-            dataset_documents = (
-                db.session.query(DatasetDocument)
-                .where(
+            dataset_documents = db.session.scalars(
+                select(DatasetDocument).where(
                    DatasetDocument.dataset_id == dataset_id,
                    DatasetDocument.indexing_status == "completed",
                    DatasetDocument.enabled == True,
                    DatasetDocument.archived == False,
                )
-                .all()
-            )
+            ).all()
            # add new index
            if dataset_documents:
                # update document status
--- a/api/tasks/disable_segments_from_index_task.py
+++ b/api/tasks/disable_segments_from_index_task.py
@@ -3,6 +3,7 @@ import time

 import click
 from celery import shared_task
+from sqlalchemy import select

 from core.rag.index_processor.index_processor_factory import IndexProcessorFactory
 from extensions.ext_database import db
@@ -44,15 +45,13 @@ def disable_segments_from_index_task(segment_ids: list, dataset_id: str, documen
    # sync index processor
    index_processor = IndexProcessorFactory(dataset_document.doc_form).init_index_processor()

-    segments = (
-        db.session.query(DocumentSegment)
-        .where(
+    segments = db.session.scalars(
+        select(DocumentSegment).where(
            DocumentSegment.id.in_(segment_ids),
            DocumentSegment.dataset_id == dataset_id,
            DocumentSegment.document_id == document_id,
        )
-        .all()
-    )
+    ).all()

    if not segments:
        db.session.close()
--- a/api/tasks/document_indexing_sync_task.py
+++ b/api/tasks/document_indexing_sync_task.py
@@ -3,6 +3,7 @@ import time

 import click
 from celery import shared_task
+from sqlalchemy import select

 from core.indexing_runner import DocumentIsPausedError, IndexingRunner
 from core.rag.extractor.notion_extractor import NotionExtractor
@@ -85,7 +86,9 @@ def document_indexing_sync_task(dataset_id: str, document_id: str):
                index_type = document.doc_form
                index_processor = IndexProcessorFactory(index_type).init_index_processor()

-                segments = db.session.query(DocumentSegment).where(DocumentSegment.document_id == document_id).all()
+                segments = db.session.scalars(
+                    select(DocumentSegment).where(DocumentSegment.document_id == document_id)
+                ).all()
                index_node_ids = [segment.index_node_id for segment in segments]

                # delete from vector index
--- a/api/tasks/document_indexing_update_task.py
+++ b/api/tasks/document_indexing_update_task.py
@@ -3,6 +3,7 @@ import time

 import click
 from celery import shared_task
+from sqlalchemy import select

 from core.indexing_runner import DocumentIsPausedError, IndexingRunner
 from core.rag.index_processor.index_processor_factory import IndexProcessorFactory
@@ -45,7 +46,7 @@ def document_indexing_update_task(dataset_id: str, document_id: str):
        index_type = document.doc_form
        index_processor = IndexProcessorFactory(index_type).init_index_processor()

-        segments = db.session.query(DocumentSegment).where(DocumentSegment.document_id == document_id).all()
+        segments = db.session.scalars(select(DocumentSegment).where(DocumentSegment.document_id == document_id)).all()
        if segments:
            index_node_ids = [segment.index_node_id for segment in segments]

--- a/api/tasks/duplicate_document_indexing_task.py
+++ b/api/tasks/duplicate_document_indexing_task.py
@@ -3,6 +3,7 @@ import time

 import click
 from celery import shared_task
+from sqlalchemy import select

 from configs import dify_config
 from core.indexing_runner import DocumentIsPausedError, IndexingRunner
@@ -79,7 +80,9 @@ def duplicate_document_indexing_task(dataset_id: str, document_ids: list):
                index_type = document.doc_form
                index_processor = IndexProcessorFactory(index_type).init_index_processor()

-                segments = db.session.query(DocumentSegment).where(DocumentSegment.document_id == document_id).all()
+                segments = db.session.scalars(
+                    select(DocumentSegment).where(DocumentSegment.document_id == document_id)
+                ).all()
                if segments:
                    index_node_ids = [segment.index_node_id for segment in segments]

--- a/api/tasks/enable_segments_to_index_task.py
+++ b/api/tasks/enable_segments_to_index_task.py
@@ -3,6 +3,7 @@ import time

 import click
 from celery import shared_task
+from sqlalchemy import select

 from core.rag.index_processor.constant.index_type import IndexType
 from core.rag.index_processor.index_processor_factory import IndexProcessorFactory
@@ -45,15 +46,13 @@ def enable_segments_to_index_task(segment_ids: list, dataset_id: str, document_i
    # sync index processor
    index_processor = IndexProcessorFactory(dataset_document.doc_form).init_index_processor()

-    segments = (
-        db.session.query(DocumentSegment)
-        .where(
+    segments = db.session.scalars(
+        select(DocumentSegment).where(
            DocumentSegment.id.in_(segment_ids),
            DocumentSegment.dataset_id == dataset_id,
            DocumentSegment.document_id == document_id,
        )
-        .all()
-    )
+    ).all()
    if not segments:
        logger.info(click.style(f"Segments not found: {segment_ids}", fg="cyan"))
        db.session.close()
--- a/api/tasks/remove_document_from_index_task.py
+++ b/api/tasks/remove_document_from_index_task.py
@@ -3,6 +3,7 @@ import time

 import click
 from celery import shared_task
+from sqlalchemy import select

 from core.rag.index_processor.index_processor_factory import IndexProcessorFactory
 from extensions.ext_database import db
@@ -45,7 +46,7 @@ def remove_document_from_index_task(document_id: str):

        index_processor = IndexProcessorFactory(document.doc_form).init_index_processor()

-        segments = db.session.query(DocumentSegment).where(DocumentSegment.document_id == document.id).all()
+        segments = db.session.scalars(select(DocumentSegment).where(DocumentSegment.document_id == document.id)).all()
        index_node_ids = [segment.index_node_id for segment in segments]
        if index_node_ids:
            try:
--- a/api/tasks/retry_document_indexing_task.py
+++ b/api/tasks/retry_document_indexing_task.py
@@ -3,6 +3,7 @@ import time

 import click
 from celery import shared_task
+from sqlalchemy import select

 from core.indexing_runner import IndexingRunner
 from core.rag.index_processor.index_processor_factory import IndexProcessorFactory
@@ -69,7 +70,9 @@ def retry_document_indexing_task(dataset_id: str, document_ids: list[str]):
                # clean old data
                index_processor = IndexProcessorFactory(document.doc_form).init_index_processor()

-                segments = db.session.query(DocumentSegment).where(DocumentSegment.document_id == document_id).all()
+                segments = db.session.scalars(
+                    select(DocumentSegment).where(DocumentSegment.document_id == document_id)
+                ).all()
                if segments:
                    index_node_ids = [segment.index_node_id for segment in segments]
                    # delete from vector index
--- a/api/tasks/sync_website_document_indexing_task.py
+++ b/api/tasks/sync_website_document_indexing_task.py
@@ -3,6 +3,7 @@ import time

 import click
 from celery import shared_task
+from sqlalchemy import select

 from core.indexing_runner import IndexingRunner
 from core.rag.index_processor.index_processor_factory import IndexProcessorFactory
@@ -63,7 +64,7 @@ def sync_website_document_indexing_task(dataset_id: str, document_id: str):
        # clean old data
        index_processor = IndexProcessorFactory(document.doc_form).init_index_processor()

-        segments = db.session.query(DocumentSegment).where(DocumentSegment.document_id == document_id).all()
+        segments = db.session.scalars(select(DocumentSegment).where(DocumentSegment.document_id == document_id)).all()
        if segments:
            index_node_ids = [segment.index_node_id for segment in segments]
            # delete from vector index