LightRAG/lightrag/lightrag.py

from __future__ import annotations

import asyncio
import configparser
import os
import warnings
from dataclasses import asdict, dataclass, field
from datetime import datetime
from functools import partial
from typing import Any, AsyncIterator, Callable, Iterator, cast, final, Literal

from lightrag.kg import (
    STORAGES,
    verify_storage_implementation,
)

from lightrag.kg.shared_storage import (
    get_namespace_data,
    get_pipeline_status_lock,
)

from .base import (
    BaseGraphStorage,
    BaseKVStorage,
    BaseVectorStorage,
    DocProcessingStatus,
    DocStatus,
    DocStatusStorage,
    QueryParam,
    StorageNameSpace,
    StoragesStatus,
)
from .namespace import NameSpace, make_namespace
from .operate import (
    chunking_by_token_size,
    extract_entities,
    kg_query,
    mix_kg_vector_query,
    naive_query,
    query_with_keywords,
)
from .prompt import GRAPH_FIELD_SEP, PROMPTS
from .utils import (
    EmbeddingFunc,
    always_get_an_event_loop,
    compute_mdhash_id,
    convert_response_to_json,
    encode_string_by_tiktoken,
    lazy_external_import,
    limit_async_func_call,
    get_content_summary,
    clean_text,
    check_storage_env_vars,
    logger,
)
from .types import KnowledgeGraph
from dotenv import load_dotenv

# use the .env that is inside the current folder
# allows to use different .env file for each lightrag instance
# the OS environment variables take precedence over the .env file
load_dotenv(dotenv_path=".env", override=False)

# TODO: TO REMOVE @Yannick
config = configparser.ConfigParser()
config.read("config.ini", "utf-8")


@final
@dataclass
class LightRAG:
    """LightRAG: Simple and Fast Retrieval-Augmented Generation."""

    # Directory
    # ---

    working_dir: str = field(
        default=f"./lightrag_cache_{datetime.now().strftime('%Y-%m-%d-%H:%M:%S')}"
    )
    """Directory where cache and temporary files are stored."""

    # Storage
    # ---

    kv_storage: str = field(default="JsonKVStorage")
    """Storage backend for key-value data."""

    vector_storage: str = field(default="NanoVectorDBStorage")
    """Storage backend for vector embeddings."""

    graph_storage: str = field(default="NetworkXStorage")
    """Storage backend for knowledge graphs."""

    doc_status_storage: str = field(default="JsonDocStatusStorage")
    """Storage type for tracking document processing statuses."""

    # Logging (Deprecated, use setup_logger in utils.py instead)
    # ---
    log_level: int | None = field(default=None)
    log_file_path: str | None = field(default=None)

    # Entity extraction
    # ---

    entity_extract_max_gleaning: int = field(default=1)
    """Maximum number of entity extraction attempts for ambiguous content."""

    summary_to_max_tokens: int = field(default=int(os.getenv("MAX_TOKEN_SUMMARY", 500)))

    force_llm_summary_on_merge: int = field(
        default=int(os.getenv("FORCE_LLM_SUMMARY_ON_MERGE", 6))
    )

    # Text chunking
    # ---

    chunk_token_size: int = field(default=int(os.getenv("CHUNK_SIZE", 1200)))
    """Maximum number of tokens per text chunk when splitting documents."""

    chunk_overlap_token_size: int = field(
        default=int(os.getenv("CHUNK_OVERLAP_SIZE", 100))
    )
    """Number of overlapping tokens between consecutive text chunks to preserve context."""

    tiktoken_model_name: str = field(default="gpt-4o-mini")
    """Model name used for tokenization when chunking text."""

    """Maximum number of tokens used for summarizing extracted entities."""

    chunking_func: Callable[
        [
            str,
            str | None,
            bool,
            int,
            int,
            str,
        ],
        list[dict[str, Any]],
    ] = field(default_factory=lambda: chunking_by_token_size)
    """
    Custom chunking function for splitting text into chunks before processing.

    The function should take the following parameters:

        - `content`: The text to be split into chunks.
        - `split_by_character`: The character to split the text on. If None, the text is split into chunks of `chunk_token_size` tokens.
        - `split_by_character_only`: If True, the text is split only on the specified character.
        - `chunk_token_size`: The maximum number of tokens per chunk.
        - `chunk_overlap_token_size`: The number of overlapping tokens between consecutive chunks.
        - `tiktoken_model_name`: The name of the tiktoken model to use for tokenization.

    The function should return a list of dictionaries, where each dictionary contains the following keys:
        - `tokens`: The number of tokens in the chunk.
        - `content`: The text content of the chunk.

    Defaults to `chunking_by_token_size` if not specified.
    """

    # Embedding
    # ---

    embedding_func: EmbeddingFunc | None = field(default=None)
    """Function for computing text embeddings. Must be set before use."""

    embedding_batch_num: int = field(default=int(os.getenv("EMBEDDING_BATCH_NUM", 32)))
    """Batch size for embedding computations."""

    embedding_func_max_async: int = field(
        default=int(os.getenv("EMBEDDING_FUNC_MAX_ASYNC", 16))
    )
    """Maximum number of concurrent embedding function calls."""

    embedding_cache_config: dict[str, Any] = field(
        default_factory=lambda: {
            "enabled": False,
            "similarity_threshold": 0.95,
            "use_llm_check": False,
        }
    )
    """Configuration for embedding cache.
    - enabled: If True, enables caching to avoid redundant computations.
    - similarity_threshold: Minimum similarity score to use cached embeddings.
    - use_llm_check: If True, validates cached embeddings using an LLM.
    """

    # LLM Configuration
    # ---

    llm_model_func: Callable[..., object] | None = field(default=None)
    """Function for interacting with the large language model (LLM). Must be set before use."""

    llm_model_name: str = field(default="gpt-4o-mini")
    """Name of the LLM model used for generating responses."""

    llm_model_max_token_size: int = field(default=int(os.getenv("MAX_TOKENS", 32768)))
    """Maximum number of tokens allowed per LLM response."""

    llm_model_max_async: int = field(default=int(os.getenv("MAX_ASYNC", 4)))
    """Maximum number of concurrent LLM calls."""

    llm_model_kwargs: dict[str, Any] = field(default_factory=dict)
    """Additional keyword arguments passed to the LLM model function."""

    # Storage
    # ---

    vector_db_storage_cls_kwargs: dict[str, Any] = field(default_factory=dict)
    """Additional parameters for vector database storage."""

    # TODO：deprecated, remove in the future, use WORKSPACE instead
    namespace_prefix: str = field(default="")
    """Prefix for namespacing stored data across different environments."""

    enable_llm_cache: bool = field(default=True)
    """Enables caching for LLM responses to avoid redundant computations."""

    enable_llm_cache_for_entity_extract: bool = field(default=True)
    """If True, enables caching for entity extraction steps to reduce LLM costs."""

    # Extensions
    # ---

    max_parallel_insert: int = field(default=int(os.getenv("MAX_PARALLEL_INSERT", 2)))
    """Maximum number of parallel insert operations."""

    addon_params: dict[str, Any] = field(
        default_factory=lambda: {
            "language": os.getenv("SUMMARY_LANGUAGE", PROMPTS["DEFAULT_LANGUAGE"])
        }
    )

    # Storages Management
    # ---

    auto_manage_storages_states: bool = field(default=True)
    """If True, lightrag will automatically calls initialize_storages and finalize_storages at the appropriate times."""

    # Storages Management
    # ---

    convert_response_to_json_func: Callable[[str], dict[str, Any]] = field(
        default_factory=lambda: convert_response_to_json
    )
    """
    Custom function for converting LLM responses to JSON format.

    The default function is :func:`.utils.convert_response_to_json`.
    """

    cosine_better_than_threshold: float = field(
        default=float(os.getenv("COSINE_THRESHOLD", 0.2))
    )

    _storages_status: StoragesStatus = field(default=StoragesStatus.NOT_CREATED)

    def __post_init__(self):
        from lightrag.kg.shared_storage import (
            initialize_share_data,
        )

        # Handle deprecated parameters
        if self.log_level is not None:
            warnings.warn(
                "WARNING: log_level parameter is deprecated, use setup_logger in utils.py instead",
                UserWarning,
                stacklevel=2,
            )
        if self.log_file_path is not None:
            warnings.warn(
                "WARNING: log_file_path parameter is deprecated, use setup_logger in utils.py instead",
                UserWarning,
                stacklevel=2,
            )

        # Remove these attributes to prevent their use
        if hasattr(self, "log_level"):
            delattr(self, "log_level")
        if hasattr(self, "log_file_path"):
            delattr(self, "log_file_path")

        initialize_share_data()

        if not os.path.exists(self.working_dir):
            logger.info(f"Creating working directory {self.working_dir}")
            os.makedirs(self.working_dir)

        # Verify storage implementation compatibility and environment variables
        storage_configs = [
            ("KV_STORAGE", self.kv_storage),
            ("VECTOR_STORAGE", self.vector_storage),
            ("GRAPH_STORAGE", self.graph_storage),
            ("DOC_STATUS_STORAGE", self.doc_status_storage),
        ]

        for storage_type, storage_name in storage_configs:
            # Verify storage implementation compatibility
            verify_storage_implementation(storage_type, storage_name)
            # Check environment variables
            check_storage_env_vars(storage_name)

        # Ensure vector_db_storage_cls_kwargs has required fields
        self.vector_db_storage_cls_kwargs = {
            "cosine_better_than_threshold": self.cosine_better_than_threshold,
            **self.vector_db_storage_cls_kwargs,
        }

        # Show config
        global_config = asdict(self)
        _print_config = ",\n  ".join([f"{k} = {v}" for k, v in global_config.items()])
        logger.debug(f"LightRAG init with param:\n  {_print_config}\n")

        # Init LLM
        self.embedding_func = limit_async_func_call(self.embedding_func_max_async)(  # type: ignore
            self.embedding_func
        )

        # Initialize all storages
        self.key_string_value_json_storage_cls: type[BaseKVStorage] = (
            self._get_storage_class(self.kv_storage)
        )  # type: ignore
        self.vector_db_storage_cls: type[BaseVectorStorage] = self._get_storage_class(
            self.vector_storage
        )  # type: ignore
        self.graph_storage_cls: type[BaseGraphStorage] = self._get_storage_class(
            self.graph_storage
        )  # type: ignore
        self.key_string_value_json_storage_cls = partial(  # type: ignore
            self.key_string_value_json_storage_cls, global_config=global_config
        )
        self.vector_db_storage_cls = partial(  # type: ignore
            self.vector_db_storage_cls, global_config=global_config
        )
        self.graph_storage_cls = partial(  # type: ignore
            self.graph_storage_cls, global_config=global_config
        )

        # Initialize document status storage
        self.doc_status_storage_cls = self._get_storage_class(self.doc_status_storage)

        self.llm_response_cache: BaseKVStorage = self.key_string_value_json_storage_cls(  # type: ignore
            namespace=make_namespace(
                self.namespace_prefix, NameSpace.KV_STORE_LLM_RESPONSE_CACHE
            ),
            global_config=asdict(
                self
            ),  # Add global_config to ensure cache works properly
            embedding_func=self.embedding_func,
        )

        self.full_docs: BaseKVStorage = self.key_string_value_json_storage_cls(  # type: ignore
            namespace=make_namespace(
                self.namespace_prefix, NameSpace.KV_STORE_FULL_DOCS
            ),
            embedding_func=self.embedding_func,
        )
        self.text_chunks: BaseKVStorage = self.key_string_value_json_storage_cls(  # type: ignore
            namespace=make_namespace(
                self.namespace_prefix, NameSpace.KV_STORE_TEXT_CHUNKS
            ),
            embedding_func=self.embedding_func,
        )
        self.chunk_entity_relation_graph: BaseGraphStorage = self.graph_storage_cls(  # type: ignore
            namespace=make_namespace(
                self.namespace_prefix, NameSpace.GRAPH_STORE_CHUNK_ENTITY_RELATION
            ),
            embedding_func=self.embedding_func,
        )

        self.entities_vdb: BaseVectorStorage = self.vector_db_storage_cls(  # type: ignore
            namespace=make_namespace(
                self.namespace_prefix, NameSpace.VECTOR_STORE_ENTITIES
            ),
            embedding_func=self.embedding_func,
            meta_fields={"entity_name", "source_id", "content", "file_path"},
        )
        self.relationships_vdb: BaseVectorStorage = self.vector_db_storage_cls(  # type: ignore
            namespace=make_namespace(
                self.namespace_prefix, NameSpace.VECTOR_STORE_RELATIONSHIPS
            ),
            embedding_func=self.embedding_func,
            meta_fields={"src_id", "tgt_id", "source_id", "content", "file_path"},
        )
        self.chunks_vdb: BaseVectorStorage = self.vector_db_storage_cls(  # type: ignore
            namespace=make_namespace(
                self.namespace_prefix, NameSpace.VECTOR_STORE_CHUNKS
            ),
            embedding_func=self.embedding_func,
            meta_fields={"full_doc_id", "content", "file_path"},
        )

        # Initialize document status storage
        self.doc_status: DocStatusStorage = self.doc_status_storage_cls(
            namespace=make_namespace(self.namespace_prefix, NameSpace.DOC_STATUS),
            global_config=global_config,
            embedding_func=None,
        )

        # Directly use llm_response_cache, don't create a new object
        hashing_kv = self.llm_response_cache

        self.llm_model_func = limit_async_func_call(self.llm_model_max_async)(
            partial(
                self.llm_model_func,  # type: ignore
                hashing_kv=hashing_kv,
                **self.llm_model_kwargs,
            )
        )

        self._storages_status = StoragesStatus.CREATED

        if self.auto_manage_storages_states:
            self._run_async_safely(self.initialize_storages, "Storage Initialization")

    def __del__(self):
        if self.auto_manage_storages_states:
            self._run_async_safely(self.finalize_storages, "Storage Finalization")

    def _run_async_safely(self, async_func, action_name=""):
        """Safely execute an async function, avoiding event loop conflicts."""
        try:
            loop = always_get_an_event_loop()
            if loop.is_running():
                task = loop.create_task(async_func())
                task.add_done_callback(
                    lambda t: logger.info(f"{action_name} completed!")
                )
            else:
                loop.run_until_complete(async_func())
        except RuntimeError:
            logger.warning(
                f"No running event loop, creating a new loop for {action_name}."
            )
            loop = asyncio.new_event_loop()
            loop.run_until_complete(async_func())
            loop.close()

    async def initialize_storages(self):
        """Asynchronously initialize the storages"""
        if self._storages_status == StoragesStatus.CREATED:
            tasks = []

            for storage in (
                self.full_docs,
                self.text_chunks,
                self.entities_vdb,
                self.relationships_vdb,
                self.chunks_vdb,
                self.chunk_entity_relation_graph,
                self.llm_response_cache,
                self.doc_status,
            ):
                if storage:
                    tasks.append(storage.initialize())

            await asyncio.gather(*tasks)

            self._storages_status = StoragesStatus.INITIALIZED
            logger.debug("Initialized Storages")

    async def finalize_storages(self):
        """Asynchronously finalize the storages"""
        if self._storages_status == StoragesStatus.INITIALIZED:
            tasks = []

            for storage in (
                self.full_docs,
                self.text_chunks,
                self.entities_vdb,
                self.relationships_vdb,
                self.chunks_vdb,
                self.chunk_entity_relation_graph,
                self.llm_response_cache,
                self.doc_status,
            ):
                if storage:
                    tasks.append(storage.finalize())

            await asyncio.gather(*tasks)

            self._storages_status = StoragesStatus.FINALIZED
            logger.debug("Finalized Storages")

    async def get_graph_labels(self):
        text = await self.chunk_entity_relation_graph.get_all_labels()
        return text

    async def get_knowledge_graph(
        self,
        node_label: str,
        max_depth: int = 3,
        max_nodes: int = 1000,
    ) -> KnowledgeGraph:
        """Get knowledge graph for a given label

        Args:
            node_label (str): Label to get knowledge graph for
            max_depth (int): Maximum depth of graph
            max_nodes (int, optional): Maximum number of nodes to return. Defaults to 1000.

        Returns:
            KnowledgeGraph: Knowledge graph containing nodes and edges
        """

        return await self.chunk_entity_relation_graph.get_knowledge_graph(
            node_label, max_depth, max_nodes
        )

    def _get_storage_class(self, storage_name: str) -> Callable[..., Any]:
        import_path = STORAGES[storage_name]
        storage_class = lazy_external_import(import_path, storage_name)
        return storage_class

    def insert(
        self,
        input: str | list[str],
        split_by_character: str | None = None,
        split_by_character_only: bool = False,
        ids: str | list[str] | None = None,
        file_paths: str | list[str] | None = None,
    ) -> None:
        """Sync Insert documents with checkpoint support

        Args:
            input: Single document string or list of document strings
            split_by_character: if split_by_character is not None, split the string by character, if chunk longer than
            chunk_token_size, it will be split again by token size.
            split_by_character_only: if split_by_character_only is True, split the string by character only, when
            split_by_character is None, this parameter is ignored.
            ids: single string of the document ID or list of unique document IDs, if not provided, MD5 hash IDs will be generated
            file_paths: single string of the file path or list of file paths, used for citation
        """
        loop = always_get_an_event_loop()
        loop.run_until_complete(
            self.ainsert(
                input, split_by_character, split_by_character_only, ids, file_paths
            )
        )

    async def ainsert(
        self,
        input: str | list[str],
        split_by_character: str | None = None,
        split_by_character_only: bool = False,
        ids: str | list[str] | None = None,
        file_paths: str | list[str] | None = None,
    ) -> None:
        """Async Insert documents with checkpoint support

        Args:
            input: Single document string or list of document strings
            split_by_character: if split_by_character is not None, split the string by character, if chunk longer than
            chunk_token_size, it will be split again by token size.
            split_by_character_only: if split_by_character_only is True, split the string by character only, when
            split_by_character is None, this parameter is ignored.
            ids: list of unique document IDs, if not provided, MD5 hash IDs will be generated
            file_paths: list of file paths corresponding to each document, used for citation
        """
        await self.apipeline_enqueue_documents(input, ids, file_paths)
        await self.apipeline_process_enqueue_documents(
            split_by_character, split_by_character_only
        )

    # TODO: deprecated, use insert instead
    def insert_custom_chunks(
        self,
        full_text: str,
        text_chunks: list[str],
        doc_id: str | list[str] | None = None,
    ) -> None:
        loop = always_get_an_event_loop()
        loop.run_until_complete(
            self.ainsert_custom_chunks(full_text, text_chunks, doc_id)
        )

    # TODO: deprecated, use ainsert instead
    async def ainsert_custom_chunks(
        self, full_text: str, text_chunks: list[str], doc_id: str | None = None
    ) -> None:
        update_storage = False
        try:
            # Clean input texts
            full_text = clean_text(full_text)
            text_chunks = [clean_text(chunk) for chunk in text_chunks]

            # Process cleaned texts
            if doc_id is None:
                doc_key = compute_mdhash_id(full_text, prefix="doc-")
            else:
                doc_key = doc_id
            new_docs = {doc_key: {"content": full_text}}

            _add_doc_keys = await self.full_docs.filter_keys({doc_key})
            new_docs = {k: v for k, v in new_docs.items() if k in _add_doc_keys}
            if not len(new_docs):
                logger.warning("This document is already in the storage.")
                return

            update_storage = True
            logger.info(f"Inserting {len(new_docs)} docs")

            inserting_chunks: dict[str, Any] = {}
            for chunk_text in text_chunks:
                chunk_key = compute_mdhash_id(chunk_text, prefix="chunk-")

                inserting_chunks[chunk_key] = {
                    "content": chunk_text,
                    "full_doc_id": doc_key,
                }

            doc_ids = set(inserting_chunks.keys())
            add_chunk_keys = await self.text_chunks.filter_keys(doc_ids)
            inserting_chunks = {
                k: v for k, v in inserting_chunks.items() if k in add_chunk_keys
            }
            if not len(inserting_chunks):
                logger.warning("All chunks are already in the storage.")
                return

            tasks = [
                self.chunks_vdb.upsert(inserting_chunks),
                self._process_entity_relation_graph(inserting_chunks),
                self.full_docs.upsert(new_docs),
                self.text_chunks.upsert(inserting_chunks),
            ]
            await asyncio.gather(*tasks)

        finally:
            if update_storage:
                await self._insert_done()

    async def apipeline_enqueue_documents(
        self,
        input: str | list[str],
        ids: list[str] | None = None,
        file_paths: str | list[str] | None = None,
    ) -> None:
        """
        Pipeline for Processing Documents

        1. Validate ids if provided or generate MD5 hash IDs
        2. Remove duplicate contents
        3. Generate document initial status
        4. Filter out already processed documents
        5. Enqueue document in status

        Args:
            input: Single document string or list of document strings
            ids: list of unique document IDs, if not provided, MD5 hash IDs will be generated
            file_paths: list of file paths corresponding to each document, used for citation
        """
        if isinstance(input, str):
            input = [input]
        if isinstance(ids, str):
            ids = [ids]
        if isinstance(file_paths, str):
            file_paths = [file_paths]

        # If file_paths is provided, ensure it matches the number of documents
        if file_paths is not None:
            if isinstance(file_paths, str):
                file_paths = [file_paths]
            if len(file_paths) != len(input):
                raise ValueError(
                    "Number of file paths must match the number of documents"
                )
        else:
            # If no file paths provided, use placeholder
            file_paths = ["unknown_source"] * len(input)

        # 1. Validate ids if provided or generate MD5 hash IDs
        if ids is not None:
            # Check if the number of IDs matches the number of documents
            if len(ids) != len(input):
                raise ValueError("Number of IDs must match the number of documents")

            # Check if IDs are unique
            if len(ids) != len(set(ids)):
                raise ValueError("IDs must be unique")

            # Generate contents dict of IDs provided by user and documents
            contents = {
                id_: {"content": doc, "file_path": path}
                for id_, doc, path in zip(ids, input, file_paths)
            }
        else:
            # Clean input text and remove duplicates
            cleaned_input = [
                (clean_text(doc), path) for doc, path in zip(input, file_paths)
            ]
            unique_content_with_paths = {}

            # Keep track of unique content and their paths
            for content, path in cleaned_input:
                if content not in unique_content_with_paths:
                    unique_content_with_paths[content] = path

            # Generate contents dict of MD5 hash IDs and documents with paths
            contents = {
                compute_mdhash_id(content, prefix="doc-"): {
                    "content": content,
                    "file_path": path,
                }
                for content, path in unique_content_with_paths.items()
            }

        # 2. Remove duplicate contents
        unique_contents = {}
        for id_, content_data in contents.items():
            content = content_data["content"]
            file_path = content_data["file_path"]
            if content not in unique_contents:
                unique_contents[content] = (id_, file_path)

        # Reconstruct contents with unique content
        contents = {
            id_: {"content": content, "file_path": file_path}
            for content, (id_, file_path) in unique_contents.items()
        }

        # 3. Generate document initial status
        new_docs: dict[str, Any] = {
            id_: {
                "status": DocStatus.PENDING,
                "content": content_data["content"],
                "content_summary": get_content_summary(content_data["content"]),
                "content_length": len(content_data["content"]),
                "created_at": datetime.now().isoformat(),
                "updated_at": datetime.now().isoformat(),
                "file_path": content_data[
                    "file_path"
                ],  # Store file path in document status
            }
            for id_, content_data in contents.items()
        }

        # 4. Filter out already processed documents
        # Get docs ids
        all_new_doc_ids = set(new_docs.keys())
        # Exclude IDs of documents that are already in progress
        unique_new_doc_ids = await self.doc_status.filter_keys(all_new_doc_ids)

        # Log ignored document IDs
        ignored_ids = [
            doc_id for doc_id in unique_new_doc_ids if doc_id not in new_docs
        ]
        if ignored_ids:
            logger.warning(
                f"Ignoring {len(ignored_ids)} document IDs not found in new_docs"
            )
            for doc_id in ignored_ids:
                logger.warning(f"Ignored document ID: {doc_id}")

        # Filter new_docs to only include documents with unique IDs
        new_docs = {
            doc_id: new_docs[doc_id]
            for doc_id in unique_new_doc_ids
            if doc_id in new_docs
        }

        if not new_docs:
            logger.info("No new unique documents were found.")
            return

        # 5. Store status document
        await self.doc_status.upsert(new_docs)
        logger.info(f"Stored {len(new_docs)} new unique documents")

    async def apipeline_process_enqueue_documents(
        self,
        split_by_character: str | None = None,
        split_by_character_only: bool = False,
    ) -> None:
        """
        Process pending documents by splitting them into chunks, processing
        each chunk for entity and relation extraction, and updating the
        document status.

        1. Get all pending, failed, and abnormally terminated processing documents.
        2. Split document content into chunks
        3. Process each chunk for entity and relation extraction
        4. Update the document status
        """

        # Get pipeline status shared data and lock
        pipeline_status = await get_namespace_data("pipeline_status")
        pipeline_status_lock = get_pipeline_status_lock()

        # Check if another process is already processing the queue
        async with pipeline_status_lock:
            # Ensure only one worker is processing documents
            if not pipeline_status.get("busy", False):
                processing_docs, failed_docs, pending_docs = await asyncio.gather(
                    self.doc_status.get_docs_by_status(DocStatus.PROCESSING),
                    self.doc_status.get_docs_by_status(DocStatus.FAILED),
                    self.doc_status.get_docs_by_status(DocStatus.PENDING),
                )

                to_process_docs: dict[str, DocProcessingStatus] = {}
                to_process_docs.update(processing_docs)
                to_process_docs.update(failed_docs)
                to_process_docs.update(pending_docs)

                if not to_process_docs:
                    logger.info("No documents to process")
                    return

                pipeline_status.update(
                    {
                        "busy": True,
                        "job_name": "Default Job",
                        "job_start": datetime.now().isoformat(),
                        "docs": 0,
                        "batchs": 0,
                        "cur_batch": 0,
                        "request_pending": False,  # Clear any previous request
                        "latest_message": "",
                    }
                )
                # Cleaning history_messages without breaking it as a shared list object
                del pipeline_status["history_messages"][:]
            else:
                # Another process is busy, just set request flag and return
                pipeline_status["request_pending"] = True
                logger.info(
                    "Another process is already processing the document queue. Request queued."
                )
                return

        try:
            # Process documents until no more documents or requests
            while True:
                if not to_process_docs:
                    log_message = "All documents have been processed or are duplicates"
                    logger.info(log_message)
                    pipeline_status["latest_message"] = log_message
                    pipeline_status["history_messages"].append(log_message)
                    break

                # 2. split docs into chunks, insert chunks, update doc status
                docs_batches = [
                    list(to_process_docs.items())[i : i + self.max_parallel_insert]
                    for i in range(0, len(to_process_docs), self.max_parallel_insert)
                ]

                log_message = f"Processing {len(to_process_docs)} document(s) in {len(docs_batches)} batches"
                logger.info(log_message)

                # Update pipeline status with current batch information
                pipeline_status["docs"] = len(to_process_docs)
                pipeline_status["batchs"] = len(docs_batches)
                pipeline_status["latest_message"] = log_message
                pipeline_status["history_messages"].append(log_message)

                # Get first document's file path and total count for job name
                first_doc_id, first_doc = next(iter(to_process_docs.items()))
                first_doc_path = first_doc.file_path
                path_prefix = first_doc_path[:20] + (
                    "..." if len(first_doc_path) > 20 else ""
                )
                total_files = len(to_process_docs)
                job_name = f"{path_prefix}[{total_files} files]"
                pipeline_status["job_name"] = job_name

                async def process_document(
                    doc_id: str,
                    status_doc: DocProcessingStatus,
                    split_by_character: str | None,
                    split_by_character_only: bool,
                    pipeline_status: dict,
                    pipeline_status_lock: asyncio.Lock,
                ) -> None:
                    """Process single document"""
                    try:
                        # Get file path from status document
                        file_path = getattr(status_doc, "file_path", "unknown_source")

                        async with pipeline_status_lock:
                            log_message = f"Processing file: {file_path}"
                            logger.info(log_message)
                            pipeline_status["history_messages"].append(log_message)
                            log_message = f"Processing d-id: {doc_id}"
                            logger.info(log_message)
                            pipeline_status["latest_message"] = log_message
                            pipeline_status["history_messages"].append(log_message)

                        # Generate chunks from document
                        chunks: dict[str, Any] = {
                            compute_mdhash_id(dp["content"], prefix="chunk-"): {
                                **dp,
                                "full_doc_id": doc_id,
                                "file_path": file_path,  # Add file path to each chunk
                            }
                            for dp in self.chunking_func(
                                status_doc.content,
                                split_by_character,
                                split_by_character_only,
                                self.chunk_overlap_token_size,
                                self.chunk_token_size,
                                self.tiktoken_model_name,
                            )
                        }

                        # Process document (text chunks and full docs) in parallel
                        # Create tasks with references for potential cancellation
                        doc_status_task = asyncio.create_task(
                            self.doc_status.upsert(
                                {
                                    doc_id: {
                                        "status": DocStatus.PROCESSING,
                                        "chunks_count": len(chunks),
                                        "content": status_doc.content,
                                        "content_summary": status_doc.content_summary,
                                        "content_length": status_doc.content_length,
                                        "created_at": status_doc.created_at,
                                        "updated_at": datetime.now().isoformat(),
                                        "file_path": file_path,
                                    }
                                }
                            )
                        )
                        chunks_vdb_task = asyncio.create_task(
                            self.chunks_vdb.upsert(chunks)
                        )
                        entity_relation_task = asyncio.create_task(
                            self._process_entity_relation_graph(
                                chunks, pipeline_status, pipeline_status_lock
                            )
                        )
                        full_docs_task = asyncio.create_task(
                            self.full_docs.upsert(
                                {doc_id: {"content": status_doc.content}}
                            )
                        )
                        text_chunks_task = asyncio.create_task(
                            self.text_chunks.upsert(chunks)
                        )
                        tasks = [
                            doc_status_task,
                            chunks_vdb_task,
                            entity_relation_task,
                            full_docs_task,
                            text_chunks_task,
                        ]
                        await asyncio.gather(*tasks)
                        await self.doc_status.upsert(
                            {
                                doc_id: {
                                    "status": DocStatus.PROCESSED,
                                    "chunks_count": len(chunks),
                                    "content": status_doc.content,
                                    "content_summary": status_doc.content_summary,
                                    "content_length": status_doc.content_length,
                                    "created_at": status_doc.created_at,
                                    "updated_at": datetime.now().isoformat(),
                                    "file_path": file_path,
                                }
                            }
                        )
                    except Exception as e:
                        # Log error and update pipeline status
                        error_msg = f"Failed to process document {doc_id}: {str(e)}"
                        logger.error(error_msg)
                        async with pipeline_status_lock:
                            pipeline_status["latest_message"] = error_msg
                            pipeline_status["history_messages"].append(error_msg)

                            # Cancel other tasks as they are no longer meaningful
                            for task in [
                                chunks_vdb_task,
                                entity_relation_task,
                                full_docs_task,
                                text_chunks_task,
                            ]:
                                if not task.done():
                                    task.cancel()
                        # Update document status to failed
                        await self.doc_status.upsert(
                            {
                                doc_id: {
                                    "status": DocStatus.FAILED,
                                    "error": str(e),
                                    "content": status_doc.content,
                                    "content_summary": status_doc.content_summary,
                                    "content_length": status_doc.content_length,
                                    "created_at": status_doc.created_at,
                                    "updated_at": datetime.now().isoformat(),
                                    "file_path": file_path,
                                }
                            }
                        )

                # 3. iterate over batches
                total_batches = len(docs_batches)
                for batch_idx, docs_batch in enumerate(docs_batches):
                    current_batch = batch_idx + 1
                    log_message = (
                        f"Start processing batch {current_batch} of {total_batches}."
                    )
                    logger.info(log_message)
                    pipeline_status["cur_batch"] = current_batch
                    pipeline_status["latest_message"] = log_message
                    pipeline_status["history_messages"].append(log_message)

                    doc_tasks = []
                    for doc_id, status_doc in docs_batch:
                        doc_tasks.append(
                            process_document(
                                doc_id,
                                status_doc,
                                split_by_character,
                                split_by_character_only,
                                pipeline_status,
                                pipeline_status_lock,
                            )
                        )

                    # Process documents in one batch parallelly
                    await asyncio.gather(*doc_tasks)
                    await self._insert_done()

                    log_message = f"Completed batch {current_batch} of {total_batches}."
                    logger.info(log_message)
                    pipeline_status["latest_message"] = log_message
                    pipeline_status["history_messages"].append(log_message)

                # Check if there's a pending request to process more documents (with lock)
                has_pending_request = False
                async with pipeline_status_lock:
                    has_pending_request = pipeline_status.get("request_pending", False)
                    if has_pending_request:
                        # Clear the request flag before checking for more documents
                        pipeline_status["request_pending"] = False

                if not has_pending_request:
                    break

                log_message = "Processing additional documents due to pending request"
                logger.info(log_message)
                pipeline_status["latest_message"] = log_message
                pipeline_status["history_messages"].append(log_message)

                # Check for pending documents again
                processing_docs, failed_docs, pending_docs = await asyncio.gather(
                    self.doc_status.get_docs_by_status(DocStatus.PROCESSING),
                    self.doc_status.get_docs_by_status(DocStatus.FAILED),
                    self.doc_status.get_docs_by_status(DocStatus.PENDING),
                )

                to_process_docs = {}
                to_process_docs.update(processing_docs)
                to_process_docs.update(failed_docs)
                to_process_docs.update(pending_docs)

        finally:
            log_message = "Document processing pipeline completed"
            logger.info(log_message)
            # Always reset busy status when done or if an exception occurs (with lock)
            async with pipeline_status_lock:
                pipeline_status["busy"] = False
                pipeline_status["latest_message"] = log_message
                pipeline_status["history_messages"].append(log_message)

    async def _process_entity_relation_graph(
        self, chunk: dict[str, Any], pipeline_status=None, pipeline_status_lock=None
    ) -> None:
        try:
            await extract_entities(
                chunk,
                knowledge_graph_inst=self.chunk_entity_relation_graph,
                entity_vdb=self.entities_vdb,
                relationships_vdb=self.relationships_vdb,
                global_config=asdict(self),
                pipeline_status=pipeline_status,
                pipeline_status_lock=pipeline_status_lock,
                llm_response_cache=self.llm_response_cache,
            )
        except Exception as e:
            logger.error("Failed to extract entities and relationships")
            raise e

    async def _insert_done(
        self, pipeline_status=None, pipeline_status_lock=None
    ) -> None:
        tasks = [
            cast(StorageNameSpace, storage_inst).index_done_callback()
            for storage_inst in [  # type: ignore
                self.full_docs,
                self.text_chunks,
                self.llm_response_cache,
                self.entities_vdb,
                self.relationships_vdb,
                self.chunks_vdb,
                self.chunk_entity_relation_graph,
            ]
            if storage_inst is not None
        ]
        await asyncio.gather(*tasks)

        log_message = "In memory DB persist to disk"
        logger.info(log_message)

        if pipeline_status is not None and pipeline_status_lock is not None:
            async with pipeline_status_lock:
                pipeline_status["latest_message"] = log_message
                pipeline_status["history_messages"].append(log_message)

    def insert_custom_kg(
        self, custom_kg: dict[str, Any], full_doc_id: str = None
    ) -> None:
        loop = always_get_an_event_loop()
        loop.run_until_complete(self.ainsert_custom_kg(custom_kg, full_doc_id))

    async def ainsert_custom_kg(
        self,
        custom_kg: dict[str, Any],
        full_doc_id: str = None,
        file_path: str = "custom_kg",
    ) -> None:
        update_storage = False
        try:
            # Insert chunks into vector storage
            all_chunks_data: dict[str, dict[str, str]] = {}
            chunk_to_source_map: dict[str, str] = {}
            for chunk_data in custom_kg.get("chunks", []):
                chunk_content = clean_text(chunk_data["content"])
                source_id = chunk_data["source_id"]
                tokens = len(
                    encode_string_by_tiktoken(
                        chunk_content, model_name=self.tiktoken_model_name
                    )
                )
                chunk_order_index = (
                    0
                    if "chunk_order_index" not in chunk_data.keys()
                    else chunk_data["chunk_order_index"]
                )
                chunk_id = compute_mdhash_id(chunk_content, prefix="chunk-")

                chunk_entry = {
                    "content": chunk_content,
                    "source_id": source_id,
                    "tokens": tokens,
                    "chunk_order_index": chunk_order_index,
                    "full_doc_id": full_doc_id
                    if full_doc_id is not None
                    else source_id,
                    "file_path": file_path,  # Add file path
                    "status": DocStatus.PROCESSED,
                }
                all_chunks_data[chunk_id] = chunk_entry
                chunk_to_source_map[source_id] = chunk_id
                update_storage = True

            if all_chunks_data:
                await asyncio.gather(
                    self.chunks_vdb.upsert(all_chunks_data),
                    self.text_chunks.upsert(all_chunks_data),
                )

            # Insert entities into knowledge graph
            all_entities_data: list[dict[str, str]] = []
            for entity_data in custom_kg.get("entities", []):
                entity_name = entity_data["entity_name"]
                entity_type = entity_data.get("entity_type", "UNKNOWN")
                description = entity_data.get("description", "No description provided")
                source_chunk_id = entity_data.get("source_id", "UNKNOWN")
                source_id = chunk_to_source_map.get(source_chunk_id, "UNKNOWN")

                # Log if source_id is UNKNOWN
                if source_id == "UNKNOWN":
                    logger.warning(
                        f"Entity '{entity_name}' has an UNKNOWN source_id. Please check the source mapping."
                    )

                # Prepare node data
                node_data: dict[str, str] = {
                    "entity_id": entity_name,
                    "entity_type": entity_type,
                    "description": description,
                    "source_id": source_id,
                }
                # Insert node data into the knowledge graph
                await self.chunk_entity_relation_graph.upsert_node(
                    entity_name, node_data=node_data
                )
                node_data["entity_name"] = entity_name
                all_entities_data.append(node_data)
                update_storage = True

            # Insert relationships into knowledge graph
            all_relationships_data: list[dict[str, str]] = []
            for relationship_data in custom_kg.get("relationships", []):
                src_id = relationship_data["src_id"]
                tgt_id = relationship_data["tgt_id"]
                description = relationship_data["description"]
                keywords = relationship_data["keywords"]
                weight = relationship_data.get("weight", 1.0)
                source_chunk_id = relationship_data.get("source_id", "UNKNOWN")
                source_id = chunk_to_source_map.get(source_chunk_id, "UNKNOWN")

                # Log if source_id is UNKNOWN
                if source_id == "UNKNOWN":
                    logger.warning(
                        f"Relationship from '{src_id}' to '{tgt_id}' has an UNKNOWN source_id. Please check the source mapping."
                    )

                # Check if nodes exist in the knowledge graph
                for need_insert_id in [src_id, tgt_id]:
                    if not (
                        await self.chunk_entity_relation_graph.has_node(need_insert_id)
                    ):
                        await self.chunk_entity_relation_graph.upsert_node(
                            need_insert_id,
                            node_data={
                                "entity_id": need_insert_id,
                                "source_id": source_id,
                                "description": "UNKNOWN",
                                "entity_type": "UNKNOWN",
                            },
                        )

                # Insert edge into the knowledge graph
                await self.chunk_entity_relation_graph.upsert_edge(
                    src_id,
                    tgt_id,
                    edge_data={
                        "weight": weight,
                        "description": description,
                        "keywords": keywords,
                        "source_id": source_id,
                    },
                )
                edge_data: dict[str, str] = {
                    "src_id": src_id,
                    "tgt_id": tgt_id,
                    "description": description,
                    "keywords": keywords,
                    "source_id": source_id,
                    "weight": weight,
                }
                all_relationships_data.append(edge_data)
                update_storage = True

            # Insert entities into vector storage with consistent format
            data_for_vdb = {
                compute_mdhash_id(dp["entity_name"], prefix="ent-"): {
                    "content": dp["entity_name"] + "\n" + dp["description"],
                    "entity_name": dp["entity_name"],
                    "source_id": dp["source_id"],
                    "description": dp["description"],
                    "entity_type": dp["entity_type"],
                    "file_path": file_path,  # Add file path
                }
                for dp in all_entities_data
            }
            await self.entities_vdb.upsert(data_for_vdb)

            # Insert relationships into vector storage with consistent format
            data_for_vdb = {
                compute_mdhash_id(dp["src_id"] + dp["tgt_id"], prefix="rel-"): {
                    "src_id": dp["src_id"],
                    "tgt_id": dp["tgt_id"],
                    "source_id": dp["source_id"],
                    "content": f"{dp['keywords']}\t{dp['src_id']}\n{dp['tgt_id']}\n{dp['description']}",
                    "keywords": dp["keywords"],
                    "description": dp["description"],
                    "weight": dp["weight"],
                    "file_path": file_path,  # Add file path
                }
                for dp in all_relationships_data
            }
            await self.relationships_vdb.upsert(data_for_vdb)

        except Exception as e:
            logger.error(f"Error in ainsert_custom_kg: {e}")
            raise
        finally:
            if update_storage:
                await self._insert_done()

    def query(
        self,
        query: str,
        param: QueryParam = QueryParam(),
        system_prompt: str | None = None,
    ) -> str | Iterator[str]:
        """
        Perform a sync query.

        Args:
            query (str): The query to be executed.
            param (QueryParam): Configuration parameters for query execution.
            prompt (Optional[str]): Custom prompts for fine-tuned control over the system's behavior. Defaults to None, which uses PROMPTS["rag_response"].

        Returns:
            str: The result of the query execution.
        """
        loop = always_get_an_event_loop()

        return loop.run_until_complete(self.aquery(query, param, system_prompt))  # type: ignore

    async def aquery(
        self,
        query: str,
        param: QueryParam = QueryParam(),
        system_prompt: str | None = None,
    ) -> str | AsyncIterator[str]:
        """
        Perform a async query.

        Args:
            query (str): The query to be executed.
            param (QueryParam): Configuration parameters for query execution.
                If param.model_func is provided, it will be used instead of the global model.
            prompt (Optional[str]): Custom prompts for fine-tuned control over the system's behavior. Defaults to None, which uses PROMPTS["rag_response"].

        Returns:
            str: The result of the query execution.
        """
        # If a custom model is provided in param, temporarily update global config
        global_config = asdict(self)

        if param.mode in ["local", "global", "hybrid"]:
            response = await kg_query(
                query.strip(),
                self.chunk_entity_relation_graph,
                self.entities_vdb,
                self.relationships_vdb,
                self.text_chunks,
                param,
                global_config,
                hashing_kv=self.llm_response_cache,  # Directly use llm_response_cache
                system_prompt=system_prompt,
            )
        elif param.mode == "naive":
            response = await naive_query(
                query.strip(),
                self.chunks_vdb,
                self.text_chunks,
                param,
                global_config,
                hashing_kv=self.llm_response_cache,  # Directly use llm_response_cache
                system_prompt=system_prompt,
            )
        elif param.mode == "mix":
            response = await mix_kg_vector_query(
                query.strip(),
                self.chunk_entity_relation_graph,
                self.entities_vdb,
                self.relationships_vdb,
                self.chunks_vdb,
                self.text_chunks,
                param,
                global_config,
                hashing_kv=self.llm_response_cache,  # Directly use llm_response_cache
                system_prompt=system_prompt,
            )
        elif param.mode == "bypass":
            # Bypass mode: directly use LLM without knowledge retrieval
            use_llm_func = param.model_func or global_config["llm_model_func"]
            param.stream = True if param.stream is None else param.stream
            response = await use_llm_func(
                query.strip(),
                system_prompt=system_prompt,
                history_messages=param.conversation_history,
                stream=param.stream,
            )
        else:
            raise ValueError(f"Unknown mode {param.mode}")
        await self._query_done()
        return response

    def query_with_separate_keyword_extraction(
        self, query: str, prompt: str, param: QueryParam = QueryParam()
    ):
        """
        Query with separate keyword extraction step.

        This method extracts keywords from the query first, then uses them for the query.

        Args:
            query: User query
            prompt: Additional prompt for the query
            param: Query parameters

        Returns:
            Query response
        """
        loop = always_get_an_event_loop()
        return loop.run_until_complete(
            self.aquery_with_separate_keyword_extraction(query, prompt, param)
        )

    async def aquery_with_separate_keyword_extraction(
        self, query: str, prompt: str, param: QueryParam = QueryParam()
    ) -> str | AsyncIterator[str]:
        """
        Async version of query_with_separate_keyword_extraction.

        Args:
            query: User query
            prompt: Additional prompt for the query
            param: Query parameters

        Returns:
            Query response or async iterator
        """
        response = await query_with_keywords(
            query=query,
            prompt=prompt,
            param=param,
            knowledge_graph_inst=self.chunk_entity_relation_graph,
            entities_vdb=self.entities_vdb,
            relationships_vdb=self.relationships_vdb,
            chunks_vdb=self.chunks_vdb,
            text_chunks_db=self.text_chunks,
            global_config=asdict(self),
            hashing_kv=self.llm_response_cache,
        )

        await self._query_done()
        return response

    async def _query_done(self):
        await self.llm_response_cache.index_done_callback()


    async def aclear_cache(self, modes: list[str] | None = None) -> None:
        """Clear cache data from the LLM response cache storage.

        Args:
            modes (list[str] | None): Modes of cache to clear. Options: ["default", "naive", "local", "global", "hybrid", "mix"].
                             "default" represents extraction cache.
                             If None, clears all cache.

        Example:
            # Clear all cache
            await rag.aclear_cache()

            # Clear local mode cache
            await rag.aclear_cache(modes=["local"])

            # Clear extraction cache
            await rag.aclear_cache(modes=["default"])
        """
        if not self.llm_response_cache:
            logger.warning("No cache storage configured")
            return

        valid_modes = ["default", "naive", "local", "global", "hybrid", "mix"]

        # Validate input
        if modes and not all(mode in valid_modes for mode in modes):
            raise ValueError(f"Invalid mode. Valid modes are: {valid_modes}")

        try:
            # Reset the cache storage for specified mode
            if modes:
                success = await self.llm_response_cache.drop_cache_by_modes(modes)
                if success:
                    logger.info(f"Cleared cache for modes: {modes}")
                else:
                    logger.warning(f"Failed to clear cache for modes: {modes}")
            else:
                # Clear all modes
                success = await self.llm_response_cache.drop_cache_by_modes(valid_modes)
                if success:
                    logger.info("Cleared all cache")
                else:
                    logger.warning("Failed to clear all cache")

            await self.llm_response_cache.index_done_callback()

        except Exception as e:
            logger.error(f"Error while clearing cache: {e}")

    def clear_cache(self, modes: list[str] | None = None) -> None:
        """Synchronous version of aclear_cache."""
        return always_get_an_event_loop().run_until_complete(self.aclear_cache(modes))


    async def get_docs_by_status(
        self, status: DocStatus
    ) -> dict[str, DocProcessingStatus]:
        """Get documents by status

        Returns:
            Dict with document id is keys and document status is values
        """
        return await self.doc_status.get_docs_by_status(status)

    # TODO: Deprecated (Deleting documents can cause hallucinations in RAG.)
    # Document delete is not working properly for most of the storage implementations.
    async def adelete_by_doc_id(self, doc_id: str) -> None:
        """Delete a document and all its related data

        Args:
            doc_id: Document ID to delete
        """
        try:
            # 1. Get the document status and related data
            if not await self.doc_status.get_by_id(doc_id):
                logger.warning(f"Document {doc_id} not found")
                return

            logger.debug(f"Starting deletion for document {doc_id}")

            # 2. Get all chunks related to this document
            # Find all chunks where full_doc_id equals the current doc_id
            all_chunks = await self.text_chunks.get_all()
            related_chunks = {
                chunk_id: chunk_data
                for chunk_id, chunk_data in all_chunks.items()
                if isinstance(chunk_data, dict)
                and chunk_data.get("full_doc_id") == doc_id
            }

            if not related_chunks:
                logger.warning(f"No chunks found for document {doc_id}")
                return

            # Get all related chunk IDs
            chunk_ids = set(related_chunks.keys())
            logger.debug(f"Found {len(chunk_ids)} chunks to delete")

            # TODO: self.entities_vdb.client_storage only works for local storage, need to fix this

            # 3. Before deleting, check the related entities and relationships for these chunks
            for chunk_id in chunk_ids:
                # Check entities
                entities_storage = await self.entities_vdb.client_storage
                entities = [
                    dp
                    for dp in entities_storage["data"]
                    if chunk_id in dp.get("source_id")
                ]
                logger.debug(f"Chunk {chunk_id} has {len(entities)} related entities")

                # Check relationships
                relationships_storage = await self.relationships_vdb.client_storage
                relations = [
                    dp
                    for dp in relationships_storage["data"]
                    if chunk_id in dp.get("source_id")
                ]
                logger.debug(f"Chunk {chunk_id} has {len(relations)} related relations")

            # Continue with the original deletion process...

            # 4. Delete chunks from vector database
            if chunk_ids:
                await self.chunks_vdb.delete(chunk_ids)
                await self.text_chunks.delete(chunk_ids)

            # 5. Find and process entities and relationships that have these chunks as source
            # Get all nodes and edges from the graph storage using storage-agnostic methods
            entities_to_delete = set()
            entities_to_update = {}  # entity_name -> new_source_id
            relationships_to_delete = set()
            relationships_to_update = {}  # (src, tgt) -> new_source_id

            # Process entities - use storage-agnostic methods
            all_labels = await self.chunk_entity_relation_graph.get_all_labels()
            for node_label in all_labels:
                node_data = await self.chunk_entity_relation_graph.get_node(node_label)
                if node_data and "source_id" in node_data:
                    # Split source_id using GRAPH_FIELD_SEP
                    sources = set(node_data["source_id"].split(GRAPH_FIELD_SEP))
                    sources.difference_update(chunk_ids)
                    if not sources:
                        entities_to_delete.add(node_label)
                        logger.debug(
                            f"Entity {node_label} marked for deletion - no remaining sources"
                        )
                    else:
                        new_source_id = GRAPH_FIELD_SEP.join(sources)
                        entities_to_update[node_label] = new_source_id
                        logger.debug(
                            f"Entity {node_label} will be updated with new source_id: {new_source_id}"
                        )

            # Process relationships
            for node_label in all_labels:
                node_edges = await self.chunk_entity_relation_graph.get_node_edges(
                    node_label
                )
                if node_edges:
                    for src, tgt in node_edges:
                        edge_data = await self.chunk_entity_relation_graph.get_edge(
                            src, tgt
                        )
                        if edge_data and "source_id" in edge_data:
                            # Split source_id using GRAPH_FIELD_SEP
                            sources = set(edge_data["source_id"].split(GRAPH_FIELD_SEP))
                            sources.difference_update(chunk_ids)
                            if not sources:
                                relationships_to_delete.add((src, tgt))
                                logger.debug(
                                    f"Relationship {src}-{tgt} marked for deletion - no remaining sources"
                                )
                            else:
                                new_source_id = GRAPH_FIELD_SEP.join(sources)
                                relationships_to_update[(src, tgt)] = new_source_id
                                logger.debug(
                                    f"Relationship {src}-{tgt} will be updated with new source_id: {new_source_id}"
                                )

            # Delete entities
            if entities_to_delete:
                for entity in entities_to_delete:
                    await self.entities_vdb.delete_entity(entity)
                    logger.debug(f"Deleted entity {entity} from vector DB")
                await self.chunk_entity_relation_graph.remove_nodes(
                    list(entities_to_delete)
                )
                logger.debug(f"Deleted {len(entities_to_delete)} entities from graph")

            # Update entities
            for entity, new_source_id in entities_to_update.items():
                node_data = await self.chunk_entity_relation_graph.get_node(entity)
                if node_data:
                    node_data["source_id"] = new_source_id
                    await self.chunk_entity_relation_graph.upsert_node(
                        entity, node_data
                    )
                    logger.debug(
                        f"Updated entity {entity} with new source_id: {new_source_id}"
                    )

            # Delete relationships
            if relationships_to_delete:
                for src, tgt in relationships_to_delete:
                    rel_id_0 = compute_mdhash_id(src + tgt, prefix="rel-")
                    rel_id_1 = compute_mdhash_id(tgt + src, prefix="rel-")
                    await self.relationships_vdb.delete([rel_id_0, rel_id_1])
                    logger.debug(f"Deleted relationship {src}-{tgt} from vector DB")
                await self.chunk_entity_relation_graph.remove_edges(
                    list(relationships_to_delete)
                )
                logger.debug(
                    f"Deleted {len(relationships_to_delete)} relationships from graph"
                )

            # Update relationships
            for (src, tgt), new_source_id in relationships_to_update.items():
                edge_data = await self.chunk_entity_relation_graph.get_edge(src, tgt)
                if edge_data:
                    edge_data["source_id"] = new_source_id
                    await self.chunk_entity_relation_graph.upsert_edge(
                        src, tgt, edge_data
                    )
                    logger.debug(
                        f"Updated relationship {src}-{tgt} with new source_id: {new_source_id}"
                    )

            # 6. Delete original document and status
            await self.full_docs.delete([doc_id])
            await self.doc_status.delete([doc_id])

            # 7. Ensure all indexes are updated
            await self._insert_done()

            logger.info(
                f"Successfully deleted document {doc_id} and related data. "
                f"Deleted {len(entities_to_delete)} entities and {len(relationships_to_delete)} relationships. "
                f"Updated {len(entities_to_update)} entities and {len(relationships_to_update)} relationships."
            )

            async def process_data(data_type, vdb, chunk_id):
                # Check data (entities or relationships)
                storage = await vdb.client_storage
                data_with_chunk = [
                    dp
                    for dp in storage["data"]
                    if chunk_id in (dp.get("source_id") or "").split(GRAPH_FIELD_SEP)
                ]

                data_for_vdb = {}
                if data_with_chunk:
                    logger.warning(
                        f"found {len(data_with_chunk)} {data_type} still referencing chunk {chunk_id}"
                    )

                    for item in data_with_chunk:
                        old_sources = item["source_id"].split(GRAPH_FIELD_SEP)
                        new_sources = [src for src in old_sources if src != chunk_id]

                        if not new_sources:
                            logger.info(
                                f"{data_type} {item.get('entity_name', 'N/A')} is deleted because source_id is not exists"
                            )
                            await vdb.delete_entity(item)
                        else:
                            item["source_id"] = GRAPH_FIELD_SEP.join(new_sources)
                            item_id = item["__id__"]
                            data_for_vdb[item_id] = item.copy()
                            if data_type == "entities":
                                data_for_vdb[item_id]["content"] = data_for_vdb[
                                    item_id
                                ].get("content") or (
                                    item.get("entity_name", "")
                                    + (item.get("description") or "")
                                )
                            else:  # relationships
                                data_for_vdb[item_id]["content"] = data_for_vdb[
                                    item_id
                                ].get("content") or (
                                    (item.get("keywords") or "")
                                    + (item.get("src_id") or "")
                                    + (item.get("tgt_id") or "")
                                    + (item.get("description") or "")
                                )

                    if data_for_vdb:
                        await vdb.upsert(data_for_vdb)
                        logger.info(f"Successfully updated {data_type} in vector DB")

            # Add verification step
            async def verify_deletion():
                # Verify if the document has been deleted
                if await self.full_docs.get_by_id(doc_id):
                    logger.warning(f"Document {doc_id} still exists in full_docs")

                # Verify if chunks have been deleted
                all_remaining_chunks = await self.text_chunks.get_all()
                remaining_related_chunks = {
                    chunk_id: chunk_data
                    for chunk_id, chunk_data in all_remaining_chunks.items()
                    if isinstance(chunk_data, dict)
                    and chunk_data.get("full_doc_id") == doc_id
                }

                if remaining_related_chunks:
                    logger.warning(
                        f"Found {len(remaining_related_chunks)} remaining chunks"
                    )

                # Verify entities and relationships
                for chunk_id in chunk_ids:
                    await process_data("entities", self.entities_vdb, chunk_id)
                    await process_data(
                        "relationships", self.relationships_vdb, chunk_id
                    )

            await verify_deletion()

        except Exception as e:
            logger.error(f"Error while deleting document {doc_id}: {e}")


    async def adelete_by_entity(self, entity_name: str) -> None:
        """Asynchronously delete an entity and all its relationships.

        Args:
            entity_name: Name of the entity to delete
        """
        from .utils_graph import adelete_by_entity
        return await adelete_by_entity(
            self.chunk_entity_relation_graph,
            self.entities_vdb,
            self.relationships_vdb,
            entity_name
        )

    def delete_by_entity(self, entity_name: str) -> None:
        loop = always_get_an_event_loop()
        return loop.run_until_complete(self.adelete_by_entity(entity_name))

    async def adelete_by_relation(self, source_entity: str, target_entity: str) -> None:
        """Asynchronously delete a relation between two entities.

        Args:
            source_entity: Name of the source entity
            target_entity: Name of the target entity
        """
        from .utils_graph import adelete_by_relation
        return await adelete_by_relation(
            self.chunk_entity_relation_graph,
            self.relationships_vdb,
            source_entity,
            target_entity
        )

    def delete_by_relation(self, source_entity: str, target_entity: str) -> None:
        loop = always_get_an_event_loop()
        return loop.run_until_complete(self.adelete_by_relation(source_entity, target_entity))

    async def get_processing_status(self) -> dict[str, int]:
        """Get current document processing status counts

        Returns:
            Dict with counts for each status
        """
        return await self.doc_status.get_status_counts()

    async def get_entity_info(
        self, entity_name: str, include_vector_data: bool = False
    ) -> dict[str, str | None | dict[str, str]]:
        """Get detailed information of an entity"""
        from .utils_graph import get_entity_info
        return await get_entity_info(
            self.chunk_entity_relation_graph,
            self.entities_vdb,
            entity_name,
            include_vector_data
        )

    async def get_relation_info(
        self, src_entity: str, tgt_entity: str, include_vector_data: bool = False
    ) -> dict[str, str | None | dict[str, str]]:
        """Get detailed information of a relationship"""
        from .utils_graph import get_relation_info
        return await get_relation_info(
            self.chunk_entity_relation_graph,
            self.relationships_vdb,
            src_entity,
            tgt_entity,
            include_vector_data
        )

    async def aedit_entity(
        self, entity_name: str, updated_data: dict[str, str], allow_rename: bool = True
    ) -> dict[str, Any]:
        """Asynchronously edit entity information.

        Updates entity information in the knowledge graph and re-embeds the entity in the vector database.

        Args:
            entity_name: Name of the entity to edit
            updated_data: Dictionary containing updated attributes, e.g. {"description": "new description", "entity_type": "new type"}
            allow_rename: Whether to allow entity renaming, defaults to True

        Returns:
            Dictionary containing updated entity information
        """
        from .utils_graph import aedit_entity
        return await aedit_entity(
            self.chunk_entity_relation_graph,
            self.entities_vdb,
            self.relationships_vdb,
            entity_name,
            updated_data,
            allow_rename
        )

    def edit_entity(
        self, entity_name: str, updated_data: dict[str, str], allow_rename: bool = True
    ) -> dict[str, Any]:
        loop = always_get_an_event_loop()
        return loop.run_until_complete(
            self.aedit_entity(entity_name, updated_data, allow_rename)
        )

    async def aedit_relation(
        self, source_entity: str, target_entity: str, updated_data: dict[str, Any]
    ) -> dict[str, Any]:
        """Asynchronously edit relation information.

        Updates relation (edge) information in the knowledge graph and re-embeds the relation in the vector database.

        Args:
            source_entity: Name of the source entity
            target_entity: Name of the target entity
            updated_data: Dictionary containing updated attributes, e.g. {"description": "new description", "keywords": "new keywords"}

        Returns:
            Dictionary containing updated relation information
        """
        from .utils_graph import aedit_relation
        return await aedit_relation(
            self.chunk_entity_relation_graph,
            self.entities_vdb,
            self.relationships_vdb,
            source_entity,
            target_entity,
            updated_data
        )

    def edit_relation(
        self, source_entity: str, target_entity: str, updated_data: dict[str, Any]
    ) -> dict[str, Any]:
        loop = always_get_an_event_loop()
        return loop.run_until_complete(
            self.aedit_relation(source_entity, target_entity, updated_data)
        )

    async def acreate_entity(
        self, entity_name: str, entity_data: dict[str, Any]
    ) -> dict[str, Any]:
        """Asynchronously create a new entity.

        Creates a new entity in the knowledge graph and adds it to the vector database.

        Args:
            entity_name: Name of the new entity
            entity_data: Dictionary containing entity attributes, e.g. {"description": "description", "entity_type": "type"}

        Returns:
            Dictionary containing created entity information
        """
        from .utils_graph import acreate_entity
        return await acreate_entity(
            self.chunk_entity_relation_graph,
            self.entities_vdb,
            self.relationships_vdb,
            entity_name,
            entity_data
        )

    def create_entity(
        self, entity_name: str, entity_data: dict[str, Any]
    ) -> dict[str, Any]:
        loop = always_get_an_event_loop()
        return loop.run_until_complete(self.acreate_entity(entity_name, entity_data))

    async def acreate_relation(
        self, source_entity: str, target_entity: str, relation_data: dict[str, Any]
    ) -> dict[str, Any]:
        """Asynchronously create a new relation between entities.

        Creates a new relation (edge) in the knowledge graph and adds it to the vector database.

        Args:
            source_entity: Name of the source entity
            target_entity: Name of the target entity
            relation_data: Dictionary containing relation attributes, e.g. {"description": "description", "keywords": "keywords"}

        Returns:
            Dictionary containing created relation information
        """
        from .utils_graph import acreate_relation
        return await acreate_relation(
            self.chunk_entity_relation_graph,
            self.entities_vdb,
            self.relationships_vdb,
            source_entity,
            target_entity,
            relation_data
        )

    def create_relation(
        self, source_entity: str, target_entity: str, relation_data: dict[str, Any]
    ) -> dict[str, Any]:
        loop = always_get_an_event_loop()
        return loop.run_until_complete(
            self.acreate_relation(source_entity, target_entity, relation_data)
        )

    async def amerge_entities(
        self,
        source_entities: list[str],
        target_entity: str,
        merge_strategy: dict[str, str] = None,
        target_entity_data: dict[str, Any] = None,
    ) -> dict[str, Any]:
        """Asynchronously merge multiple entities into one entity.

        Merges multiple source entities into a target entity, handling all relationships,
        and updating both the knowledge graph and vector database.

        Args:
            source_entities: List of source entity names to merge
            target_entity: Name of the target entity after merging
            merge_strategy: Merge strategy configuration, e.g. {"description": "concatenate", "entity_type": "keep_first"}
                Supported strategies:
                - "concatenate": Concatenate all values (for text fields)
                - "keep_first": Keep the first non-empty value
                - "keep_last": Keep the last non-empty value
                - "join_unique": Join all unique values (for fields separated by delimiter)
            target_entity_data: Dictionary of specific values to set for the target entity,
                overriding any merged values, e.g. {"description": "custom description", "entity_type": "PERSON"}

        Returns:
            Dictionary containing the merged entity information
        """
        from .utils_graph import amerge_entities
        return await amerge_entities(
            self.chunk_entity_relation_graph,
            self.entities_vdb,
            self.relationships_vdb,
            source_entities,
            target_entity,
            merge_strategy,
            target_entity_data
        )

    def merge_entities(
        self,
        source_entities: list[str],
        target_entity: str,
        merge_strategy: dict[str, str] = None,
        target_entity_data: dict[str, Any] = None,
    ) -> dict[str, Any]:
        loop = always_get_an_event_loop()
        return loop.run_until_complete(
            self.amerge_entities(
                source_entities, target_entity, merge_strategy, target_entity_data
            )
        )

    async def aexport_data(
        self,
        output_path: str,
        file_format: Literal["csv", "excel", "md", "txt"] = "csv",
        include_vector_data: bool = False,
    ) -> None:
        """
        Asynchronously exports all entities, relations, and relationships to various formats.
        Args:
            output_path: The path to the output file (including extension).
            file_format: Output format - "csv", "excel", "md", "txt".
                - csv: Comma-separated values file
                - excel: Microsoft Excel file with multiple sheets
                - md: Markdown tables
                - txt: Plain text formatted output
                - table: Print formatted tables to console
            include_vector_data: Whether to include data from the vector database.
        """
        from .utils import aexport_data as utils_aexport_data
        
        await utils_aexport_data(
            self.chunk_entity_relation_graph,
            self.entities_vdb,
            self.relationships_vdb,
            output_path,
            file_format,
            include_vector_data
        )

    def export_data(
        self,
        output_path: str,
        file_format: Literal["csv", "excel", "md", "txt"] = "csv",
        include_vector_data: bool = False,
    ) -> None:
        """
        Synchronously exports all entities, relations, and relationships to various formats.
        Args:
            output_path: The path to the output file (including extension).
            file_format: Output format - "csv", "excel", "md", "txt".
                - csv: Comma-separated values file
                - excel: Microsoft Excel file with multiple sheets
                - md: Markdown tables
                - txt: Plain text formatted output
                - table: Print formatted tables to console
            include_vector_data: Whether to include data from the vector database.
        """
        try:
            loop = asyncio.get_event_loop()
        except RuntimeError:
            loop = asyncio.new_event_loop()
            asyncio.set_event_loop(loop)

        loop.run_until_complete(
            self.aexport_data(output_path, file_format, include_vector_data)
        )
-												cleaning the mess

											
										
										
											2025-02-14 22:50:49 +01:00
+								from __future__ import annotations
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								import asyncio
-												Inject oracle db to LightRag storage class when needed

											
										
										
											2025-02-11 03:54:54 +08:00
+								import configparser
-												add support of providing ids for documents insert

											
										
										
											2025-02-20 00:26:35 +01:00
+								import os
-												Deprecate log_level and log_file_path in LightRAG.

- Remove log_level from API initialization
- Add warnings for deprecated logging params

											
										
										
											2025-03-04 01:07:34 +08:00
+								import warnings
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								from dataclasses import asdict, dataclass, field
 								from datetime import datetime
 								from functools import partial
-												Fixed lint and Added new imports at the top of the file

											
										
										
											2025-03-12 00:04:23 +05:30
+								from typing import Any, AsyncIterator, Callable, Iterator, cast, final, Literal
-												removed lock

											
										
										
											2025-02-20 12:54:52 +01:00
-												cleanup

											
										
										
											2025-02-20 13:44:17 +01:00
+								from lightrag.kg import (
 								    STORAGES,
 								    verify_storage_implementation,
 								)
-												cleanup storages

											
										
										
											2025-02-20 13:21:41 +01:00
-												Add graph_db_lock to esure consistency across multiple processes for node and edge edition jobs

											
										
										
											2025-04-14 00:07:31 +08:00
+								from lightrag.kg.shared_storage import (
 								    get_namespace_data,
 								    get_pipeline_status_lock,
 								)
-												fixed bugs

											
										
										
											2025-02-09 19:21:49 +01:00
+								from .base import (
 								    BaseGraphStorage,
 								    BaseKVStorage,
 								    BaseVectorStorage,
 								    DocProcessingStatus,
 								    DocStatus,
 								    DocStatusStorage,
 								    QueryParam,
 								    StorageNameSpace,
-												refactor database connection management and improve storage lifecycle handling

update

											
										
										
											2025-02-19 03:46:18 +08:00
+								    StoragesStatus,
-												fixed bugs

											
										
										
											2025-02-09 19:21:49 +01:00
+								)
 								from .namespace import NameSpace, make_namespace
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								from .operate import (
 								    chunking_by_token_size,
-												cleaned import

											
										
										
											2025-02-09 11:24:08 +01:00
+								    extract_entities,
 								    kg_query,
 								    mix_kg_vector_query,
 								    naive_query,
-												clean lightrag.py

											
										
										
											2025-03-11 15:43:04 +08:00
+								    query_with_keywords,
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								)
-												Add summary language setting by env

											
										
										
											2025-03-04 12:45:35 +08:00
+								from .prompt import GRAPH_FIELD_SEP, PROMPTS
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								from .utils import (
 								    EmbeddingFunc,
-												cleanup code

											
										
										
											2025-02-20 13:18:17 +01:00
+								    always_get_an_event_loop,
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								    compute_mdhash_id,
 								    convert_response_to_json,
-												add support of providing ids for documents insert

											
										
										
											2025-02-20 00:26:35 +01:00
+								    encode_string_by_tiktoken,
-												cleanup code

											
										
										
											2025-02-20 13:18:17 +01:00
+								    lazy_external_import,
-												fixed bugs

											
										
										
											2025-02-09 19:21:49 +01:00
+								    limit_async_func_call,
-												clean lightrag.py

											
										
										
											2025-03-11 15:43:04 +08:00
+								    get_content_summary,
 								    clean_text,
 								    check_storage_env_vars,
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								    logger,
 								)
-												Revert "removed get_knowledge_graph"

											
										
										
											2025-02-20 14:29:36 +01:00
+								from .types import KnowledgeGraph
-												Add automatic comment handling in .env files

											
										
										
											2025-02-22 13:25:12 +08:00
+								from dotenv import load_dotenv
-												standardize .env loading behavior across modules

											
										
										
											2025-03-29 03:48:38 +08:00
+								# use the .env that is inside the current folder
 								# allows to use different .env file for each lightrag instance
 								# the OS environment variables take precedence over the .env file
 								load_dotenv(dotenv_path=".env", override=False)
-												cleaned import

											
										
										
											2025-02-09 11:24:08 +01:00
-												cleanup kg

											
										
										
											2025-02-20 13:39:46 +01:00
+								# TODO: TO REMOVE @Yannick
-												Inject oracle db to LightRag storage class when needed

											
										
										
											2025-02-11 03:54:54 +08:00
+								config = configparser.ConfigParser()
 								config.read("config.ini", "utf-8")
-												added docs and fields

											
										
										
											2025-02-20 13:09:33 +01:00
-												added final

											
										
										
											2025-02-20 13:05:35 +01:00
+								@final
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								@dataclass
 								class LightRAG:
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """LightRAG: Simple and Fast Retrieval-Augmented Generation."""
-												cleanup

											
										
										
											2025-02-20 13:13:38 +01:00
+								    # Directory
 								    # ---
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								    working_dir: str = field(
-												added field

											
										
										
											2025-02-20 13:05:59 +01:00
+								        default=f"./lightrag_cache_{datetime.now().strftime('%Y-%m-%d-%H:%M:%S')}"
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								    )
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Directory where cache and temporary files are stored."""
-												cleanup

											
										
										
											2025-02-20 13:13:38 +01:00
+								    # Storage
 								    # ---
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
-												fix pre commit

											
										
										
											2024-11-12 13:32:40 +08:00
+								    kv_storage: str = field(default="JsonKVStorage")
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Storage backend for key-value data."""
-												Oracle Database support

Add oracle 23ai database as the KV/vector/graph storage

											
										
										
											2024-11-08 14:58:41 +08:00
+								    vector_storage: str = field(default="NanoVectorDBStorage")
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Storage backend for vector embeddings."""
-												Oracle Database support

Add oracle 23ai database as the KV/vector/graph storage

											
										
										
											2024-11-08 14:58:41 +08:00
+								    graph_storage: str = field(default="NetworkXStorage")
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Storage backend for knowledge graphs."""
-												securing for production with env vars for creds

											
										
										
											2024-11-01 11:01:50 -04:00
-												refactor: improve database initialization by centralizing db instance injection

- Move db configs to separate methods
- Remove db field defaults in storage classes
- Add _initialize_database_if_needed method
- Inject db instances during initialization
- Clean up storage implementation code

											
										
										
											2025-02-12 22:25:34 +08:00
+								    doc_status_storage: str = field(default="JsonDocStatusStorage")
 								    """Storage type for tracking document processing statuses."""
-												Deprecate log_level and log_file_path in LightRAG.

- Remove log_level from API initialization
- Add warnings for deprecated logging params

											
										
										
											2025-03-04 01:07:34 +08:00
+								    # Logging (Deprecated, use setup_logger in utils.py instead)
-												cleanup

											
										
										
											2025-02-20 13:13:38 +01:00
+								    # ---
-												Deprecate and remove logging parameters in LightRAG.

- Set log_level and log_file_path to None by default
- Issue warnings if deprecated parameters are used
- Maintain backward compatibility with warnings

											
										
										
											2025-03-04 01:28:08 +08:00
+								    log_level: int | None = field(default=None)
 								    log_file_path: str | None = field(default=None)
-												securing for production with env vars for creds

											
										
										
											2024-11-01 11:01:50 -04:00
-												cleanup

											
										
										
											2025-02-20 13:13:38 +01:00
+								    # Entity extraction
 								    # ---
 								    entity_extract_max_gleaning: int = field(default=1)
 								    """Maximum number of entity extraction attempts for ambiguous content."""
-												Add env FORCE_LLM_SUMMARY_ON_MERGE

											
										
										
											2025-04-10 17:29:07 +08:00
+								    summary_to_max_tokens: int = field(default=int(os.getenv("MAX_TOKEN_SUMMARY", 500)))
 								    force_llm_summary_on_merge: int = field(
 								        default=int(os.getenv("FORCE_LLM_SUMMARY_ON_MERGE", 6))
-												cleanup

											
										
										
											2025-02-20 13:13:38 +01:00
+								    )
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    # Text chunking
-												cleanup

											
										
										
											2025-02-20 13:13:38 +01:00
+								    # ---
-												added field

											
										
										
											2025-02-20 13:05:59 +01:00
+								    chunk_token_size: int = field(default=int(os.getenv("CHUNK_SIZE", 1200)))
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Maximum number of tokens per text chunk when splitting documents."""
-												added docs and fields

											
										
										
											2025-02-20 13:09:33 +01:00
+								    chunk_overlap_token_size: int = field(
 								        default=int(os.getenv("CHUNK_OVERLAP_SIZE", 100))
 								    )
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Number of overlapping tokens between consecutive text chunks to preserve context."""
-												added field

											
										
										
											2025-02-20 13:05:59 +01:00
+								    tiktoken_model_name: str = field(default="gpt-4o-mini")
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Model name used for tokenization when chunking text."""
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Maximum number of tokens used for summarizing extracted entities."""
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
-												cleanup

											
										
										
											2025-02-20 13:13:38 +01:00
+								    chunking_func: Callable[
 								        [
 								            str,
 								            str | None,
 								            bool,
 								            int,
 								            int,
 								            str,
 								        ],
 								        list[dict[str, Any]],
 								    ] = field(default_factory=lambda: chunking_by_token_size)
 								    """
 								    Custom chunking function for splitting text into chunks before processing.
 								    The function should take the following parameters:
 								        - `content`: The text to be split into chunks.
 								        - `split_by_character`: The character to split the text on. If None, the text is split into chunks of `chunk_token_size` tokens.
 								        - `split_by_character_only`: If True, the text is split only on the specified character.
 								        - `chunk_token_size`: The maximum number of tokens per chunk.
 								        - `chunk_overlap_token_size`: The number of overlapping tokens between consecutive chunks.
 								        - `tiktoken_model_name`: The name of the tiktoken model to use for tokenization.
 								    The function should return a list of dictionaries, where each dictionary contains the following keys:
 								        - `tokens`: The number of tokens in the chunk.
 								        - `content`: The text content of the chunk.
 								    Defaults to `chunking_by_token_size` if not specified.
 								    """
 								    # Embedding
 								    # ---
-												added fields

											
										
										
											2025-02-20 13:06:16 +01:00
+								    embedding_func: EmbeddingFunc | None = field(default=None)
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Function for computing text embeddings. Must be set before use."""
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
-												add embedding_bathc_num and embedding_func_max_async to env

											
										
										
											2025-03-21 13:47:53 +08:00
+								    embedding_batch_num: int = field(default=int(os.getenv("EMBEDDING_BATCH_NUM", 32)))
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Batch size for embedding computations."""
-												Fix linting

											
										
										
											2025-03-21 21:51:52 +08:00
+								    embedding_func_max_async: int = field(
 								        default=int(os.getenv("EMBEDDING_FUNC_MAX_ASYNC", 16))
 								    )
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Maximum number of concurrent embedding function calls."""
-												cleanup

											
										
										
											2025-02-20 13:13:38 +01:00
+								    embedding_cache_config: dict[str, Any] = field(
-												cleanup extraction

											
										
										
											2025-02-20 14:17:26 +01:00
+								        default_factory=lambda: {
-												cleanup

											
										
										
											2025-02-20 13:13:38 +01:00
+								            "enabled": False,
 								            "similarity_threshold": 0.95,
 								            "use_llm_check": False,
 								        }
 								    )
 								    """Configuration for embedding cache.
 								    - enabled: If True, enables caching to avoid redundant computations.
 								    - similarity_threshold: Minimum similarity score to use cached embeddings.
 								    - use_llm_check: If True, validates cached embeddings using an LLM.
 								    """
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    # LLM Configuration
-												cleanup

											
										
										
											2025-02-20 13:13:38 +01:00
+								    # ---
-												added fields

											
										
										
											2025-02-20 13:06:16 +01:00
+								    llm_model_func: Callable[..., object] | None = field(default=None)
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Function for interacting with the large language model (LLM). Must be set before use."""
-												added fields

											
										
										
											2025-02-20 13:06:16 +01:00
+								    llm_model_name: str = field(default="gpt-4o-mini")
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Name of the LLM model used for generating responses."""
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
-												added fields

											
										
										
											2025-02-20 13:06:16 +01:00
+								    llm_model_max_token_size: int = field(default=int(os.getenv("MAX_TOKENS", 32768)))
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Maximum number of tokens allowed per LLM response."""
-												Adjust concurrency limits  more LLM friendly settings for new comers

- Lowered max async LLM processes to 4
- Enabled LLM cache for entity extraction
- Reduced max parallel insert to 2

											
										
										
											2025-03-16 23:56:34 +08:00
+								    llm_model_max_async: int = field(default=int(os.getenv("MAX_ASYNC", 4)))
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Maximum number of concurrent LLM calls."""
 								    llm_model_kwargs: dict[str, Any] = field(default_factory=dict)
 								    """Additional keyword arguments passed to the LLM model function."""
 								    # Storage
-												cleanup

											
										
										
											2025-02-20 13:13:38 +01:00
+								    # ---
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    vector_db_storage_cls_kwargs: dict[str, Any] = field(default_factory=dict)
 								    """Additional parameters for vector database storage."""
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
-												Remove namespace_prefix from PostgreSQL, maintain consistency with other storage implementation

											
										
										
											2025-03-31 02:59:44 +08:00
+								    # TODO：deprecated, remove in the future, use WORKSPACE instead
-												 add namespace prefix to storage namespaces

											
										
										
											2025-02-07 23:04:29 +08:00
+								    namespace_prefix: str = field(default="")
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Prefix for namespacing stored data across different environments."""
-												fix pre commit

											
										
										
											2024-11-12 13:32:40 +08:00
-												added fields

											
										
										
											2025-02-20 13:06:34 +01:00
+								    enable_llm_cache: bool = field(default=True)
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """Enables caching for LLM responses to avoid redundant computations."""
-												added fields

											
										
										
											2025-02-20 13:06:34 +01:00
+								    enable_llm_cache_for_entity_extract: bool = field(default=True)
-												improved docs

											
										
										
											2025-02-09 00:23:55 +01:00
+								    """If True, enables caching for entity extraction steps to reduce LLM costs."""
 								    # Extensions
-												cleanup

											
										
										
											2025-02-20 13:13:38 +01:00
+								    # ---
-												Adjust concurrency limits  more LLM friendly settings for new comers

- Lowered max async LLM processes to 4
- Enabled LLM cache for entity extraction
- Reduced max parallel insert to 2

											
										
										
											2025-03-16 23:56:34 +08:00
+								    max_parallel_insert: int = field(default=int(os.getenv("MAX_PARALLEL_INSERT", 2)))
-												added max paralle insert

											
										
										
											2025-02-20 12:57:25 +01:00
+								    """Maximum number of parallel insert operations."""
-												added docs and fields

											
										
										
											2025-02-20 13:09:33 +01:00
-												Fix linting

											
										
										
											2025-03-04 14:02:14 +08:00
+								    addon_params: dict[str, Any] = field(
 								        default_factory=lambda: {
 								            "language": os.getenv("SUMMARY_LANGUAGE", PROMPTS["DEFAULT_LANGUAGE"])
 								        }
 								    )
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
-												rename is_managed_by_server to auto_manage_storages_states

											
										
										
											2025-02-19 05:27:38 +08:00
+								    # Storages Management
-												cleanup

											
										
										
											2025-02-20 13:13:38 +01:00
+								    # ---
-												added fields

											
										
										
											2025-02-20 13:06:34 +01:00
+								    auto_manage_storages_states: bool = field(default=True)
-												rename is_managed_by_server to auto_manage_storages_states

											
										
										
											2025-02-19 05:27:38 +08:00
+								    """If True, lightrag will automatically calls initialize_storages and finalize_storages at the appropriate times."""
-												refactor database connection management and improve storage lifecycle handling

update

											
										
										
											2025-02-19 03:46:18 +08:00
-												cleanup

											
										
										
											2025-02-20 13:13:38 +01:00
+								    # Storages Management
 								    # ---
-												added docs and fields

											
										
										
											2025-02-20 13:09:33 +01:00
+								    convert_response_to_json_func: Callable[[str], dict[str, Any]] = field(
 								        default_factory=lambda: convert_response_to_json
-												cleaned code

											
										
										
											2025-02-09 13:18:47 +01:00
+								    )
-												added docs and fields

											
										
										
											2025-02-20 13:09:33 +01:00
+								    """
 								    Custom function for converting LLM responses to JSON format.
 								    The default function is :func:`.utils.convert_response_to_json`.
 								    """
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
-												cleanup

											
										
										
											2025-02-20 13:44:17 +01:00
+								    cosine_better_than_threshold: float = field(
 								        default=float(os.getenv("COSINE_THRESHOLD", 0.2))
 								    )
-												cleanup storage state

											
										
										
											2025-02-20 13:30:30 +01:00
+								    _storages_status: StoragesStatus = field(default=StoragesStatus.NOT_CREATED)
-												chore: added pre-commit-hooks and ruff formatting for commit-hooks

											
										
										
											2024-10-19 09:43:17 +05:30
+								    def __post_init__(self):
-												Fix linting

											
										
										
											2025-02-27 19:05:51 +08:00
+								        from lightrag.kg.shared_storage import (
 								            initialize_share_data,
 								        )
-												Fix linting

											
										
										
											2025-02-28 21:35:04 +08:00
-												Deprecate log_level and log_file_path in LightRAG.

- Remove log_level from API initialization
- Add warnings for deprecated logging params

											
										
										
											2025-03-04 01:07:34 +08:00
+								        # Handle deprecated parameters
-												Deprecate and remove logging parameters in LightRAG.

- Set log_level and log_file_path to None by default
- Issue warnings if deprecated parameters are used
- Maintain backward compatibility with warnings

											
										
										
											2025-03-04 01:28:08 +08:00
+								        if self.log_level is not None:
-												Deprecate log_level and log_file_path in LightRAG.

- Remove log_level from API initialization
- Add warnings for deprecated logging params

											
										
										
											2025-03-04 01:07:34 +08:00
+								            warnings.warn(
 								                "WARNING: log_level parameter is deprecated, use setup_logger in utils.py instead",
 								                UserWarning,
 								                stacklevel=2,
 								            )
-												Deprecate and remove logging parameters in LightRAG.

- Set log_level and log_file_path to None by default
- Issue warnings if deprecated parameters are used
- Maintain backward compatibility with warnings

											
										
										
											2025-03-04 01:28:08 +08:00
+								        if self.log_file_path is not None:
-												Deprecate log_level and log_file_path in LightRAG.

- Remove log_level from API initialization
- Add warnings for deprecated logging params

											
										
										
											2025-03-04 01:07:34 +08:00
+								            warnings.warn(
 								                "WARNING: log_file_path parameter is deprecated, use setup_logger in utils.py instead",
 								                UserWarning,
 								                stacklevel=2,
 								            )
-												Fix linting

											
										
										
											2025-03-04 01:28:39 +08:00
-												Deprecate and remove logging parameters in LightRAG.

- Set log_level and log_file_path to None by default
- Issue warnings if deprecated parameters are used
- Maintain backward compatibility with warnings

											
										
										
											2025-03-04 01:28:08 +08:00
+								        # Remove these attributes to prevent their use
 								        if hasattr(self, "log_level"):
 								            delattr(self, "log_level")
 								        if hasattr(self, "log_file_path"):
-												Deprecate log_level and log_file_path in LightRAG.

- Remove log_level from API initialization
- Add warnings for deprecated logging params

											
										
										
											2025-03-04 01:07:34 +08:00
+								            delattr(self, "log_file_path")
-												Fix multiprocess dict creation logic, add process safety locks for namespace creation.

											
										
										
											2025-02-27 19:03:53 +08:00
+								        initialize_share_data()
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
+								        if not os.path.exists(self.working_dir):
 								            logger.info(f"Creating working directory {self.working_dir}")
 								            os.makedirs(self.working_dir)
-												chore: added pre-commit-hooks and ruff formatting for commit-hooks

											
										
										
											2024-10-19 09:43:17 +05:30
-												feat optimize storage configuration and environment variables

* add storage type compatibility validation table
* add enviroment variables check for storage
* modify storage init to get setting from confing.ini and env

											
										
										
											2025-02-11 00:55:52 +08:00
+								        # Verify storage implementation compatibility and environment variables
 								        storage_configs = [
 								            ("KV_STORAGE", self.kv_storage),
 								            ("VECTOR_STORAGE", self.vector_storage),
 								            ("GRAPH_STORAGE", self.graph_storage),
 								            ("DOC_STATUS_STORAGE", self.doc_status_storage),
 								        ]
 								        for storage_type, storage_name in storage_configs:
 								            # Verify storage implementation compatibility
-												cleanup kg

											
										
										
											2025-02-20 13:39:46 +01:00
+								            verify_storage_implementation(storage_type, storage_name)
-												feat optimize storage configuration and environment variables

* add storage type compatibility validation table
* add enviroment variables check for storage
* modify storage init to get setting from confing.ini and env

											
										
										
											2025-02-11 00:55:52 +08:00
+								            # Check environment variables
-												clean lightrag.py

											
										
										
											2025-03-11 15:43:04 +08:00
+								            check_storage_env_vars(storage_name)
-												feat optimize storage configuration and environment variables

* add storage type compatibility validation table
* add enviroment variables check for storage
* modify storage init to get setting from confing.ini and env

											
										
										
											2025-02-11 00:55:52 +08:00
-												refactor: make cosine similarity threshold a required config parameter

• Remove default threshold from env var
• Add validation for missing threshold
• Move default to lightrag.py config init
• Update all vector DB implementations
• Improve threshold validation consistency

											
										
										
											2025-02-13 03:25:48 +08:00
+								        # Ensure vector_db_storage_cls_kwargs has required fields
 								        self.vector_db_storage_cls_kwargs = {
-												cleanup

											
										
										
											2025-02-20 13:44:17 +01:00
+								            "cosine_better_than_threshold": self.cosine_better_than_threshold,
-												Fix linting

											
										
										
											2025-02-13 04:12:00 +08:00
+								            **self.vector_db_storage_cls_kwargs,
-												refactor: make cosine similarity threshold a required config parameter

• Remove default threshold from env var
• Add validation for missing threshold
• Move default to lightrag.py config init
• Update all vector DB implementations
• Improve threshold validation consistency

											
										
										
											2025-02-13 03:25:48 +08:00
+								        }
-												refactor database connection management and improve storage lifecycle handling

update

											
										
										
											2025-02-19 03:46:18 +08:00
+								        # Show config
-												support pipeline mode

											
										
										
											2025-01-16 12:58:15 +08:00
+								        global_config = asdict(self)
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
+								        _print_config = ",\n  ".join([f"{k} = {v}" for k, v in global_config.items()])
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								        logger.debug(f"LightRAG init with param:\n  {_print_config}\n")
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
+								        # Init LLM
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        self.embedding_func = limit_async_func_call(self.embedding_func_max_async)(  # type: ignore
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
+								            self.embedding_func
 								        )
-												set kg by start param, defaults to networkx

											
										
										
											2024-11-01 08:47:52 -04:00
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
+								        # Initialize all storages
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        self.key_string_value_json_storage_cls: type[BaseKVStorage] = (
-												support pipeline mode

											
										
										
											2025-01-16 12:58:15 +08:00
+								            self._get_storage_class(self.kv_storage)
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        )  # type: ignore
 								        self.vector_db_storage_cls: type[BaseVectorStorage] = self._get_storage_class(
-												fix pre commit

											
										
										
											2024-11-12 13:32:40 +08:00
+								            self.vector_storage
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        )  # type: ignore
 								        self.graph_storage_cls: type[BaseGraphStorage] = self._get_storage_class(
-												fix pre commit

											
										
										
											2024-11-12 13:32:40 +08:00
+								            self.graph_storage
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        )  # type: ignore
 								        self.key_string_value_json_storage_cls = partial(  # type: ignore
-												support pipeline mode

											
										
										
											2025-01-16 12:58:15 +08:00
+								            self.key_string_value_json_storage_cls, global_config=global_config
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
+								        )
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        self.vector_db_storage_cls = partial(  # type: ignore
-												support pipeline mode

											
										
										
											2025-01-16 12:58:15 +08:00
+								            self.vector_db_storage_cls, global_config=global_config
-												fix pre commit

											
										
										
											2024-11-12 13:32:40 +08:00
+								        )
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        self.graph_storage_cls = partial(  # type: ignore
-												support pipeline mode

											
										
										
											2025-01-16 12:58:15 +08:00
+								            self.graph_storage_cls, global_config=global_config
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
+								        )
-												Fix doc_status error

											
										
										
											2025-02-11 10:17:51 +08:00
+								        # Initialize document status storage
 								        self.doc_status_storage_cls = self._get_storage_class(self.doc_status_storage)
-												Inject Postgres to LightRag storage class when needed

											
										
										
											2025-02-11 03:55:15 +08:00
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        self.llm_response_cache: BaseKVStorage = self.key_string_value_json_storage_cls(  # type: ignore
-												Fix doc_status error

											
										
										
											2025-02-11 10:17:51 +08:00
+								            namespace=make_namespace(
 								                self.namespace_prefix, NameSpace.KV_STORE_LLM_RESPONSE_CACHE
 								            ),
-												Fix linting

											
										
										
											2025-03-10 02:07:19 +08:00
+								            global_config=asdict(
 								                self
 								            ),  # Add global_config to ensure cache works properly
-												Fix doc_status error

											
										
										
											2025-02-11 10:17:51 +08:00
+								            embedding_func=self.embedding_func,
 								        )
-												Add huggingface model support

											
										
										
											2024-10-15 19:40:08 +08:00
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        self.full_docs: BaseKVStorage = self.key_string_value_json_storage_cls(  # type: ignore
-												use namespace as neo4j database name

format

fix

											
										
										
											2025-02-08 16:06:07 +08:00
+								            namespace=make_namespace(
 								                self.namespace_prefix, NameSpace.KV_STORE_FULL_DOCS
 								            ),
-												fix pre commit

											
										
										
											2024-11-12 13:32:40 +08:00
+								            embedding_func=self.embedding_func,
-												Oracle Database support

Add oracle 23ai database as the KV/vector/graph storage

											
										
										
											2024-11-08 14:58:41 +08:00
+								        )
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        self.text_chunks: BaseKVStorage = self.key_string_value_json_storage_cls(  # type: ignore
-												use namespace as neo4j database name

format

fix

											
										
										
											2025-02-08 16:06:07 +08:00
+								            namespace=make_namespace(
 								                self.namespace_prefix, NameSpace.KV_STORE_TEXT_CHUNKS
 								            ),
-												fix pre commit

											
										
										
											2024-11-12 13:32:40 +08:00
+								            embedding_func=self.embedding_func,
-												Oracle Database support

Add oracle 23ai database as the KV/vector/graph storage

											
										
										
											2024-11-08 14:58:41 +08:00
+								        )
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        self.chunk_entity_relation_graph: BaseGraphStorage = self.graph_storage_cls(  # type: ignore
-												use namespace as neo4j database name

format

fix

											
										
										
											2025-02-08 16:06:07 +08:00
+								            namespace=make_namespace(
 								                self.namespace_prefix, NameSpace.GRAPH_STORE_CHUNK_ENTITY_RELATION
 								            ),
-												fix neo4jstorage bug

											
										
										
											2024-12-03 16:04:58 +08:00
+								            embedding_func=self.embedding_func,
-												Oracle Database support

Add oracle 23ai database as the KV/vector/graph storage

											
										
										
											2024-11-08 14:58:41 +08:00
+								        )
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        self.entities_vdb: BaseVectorStorage = self.vector_db_storage_cls(  # type: ignore
-												use namespace as neo4j database name

format

fix

											
										
										
											2025-02-08 16:06:07 +08:00
+								            namespace=make_namespace(
 								                self.namespace_prefix, NameSpace.VECTOR_STORE_ENTITIES
 								            ),
-												chore: added pre-commit-hooks and ruff formatting for commit-hooks

											
										
										
											2024-10-19 09:43:17 +05:30
+								            embedding_func=self.embedding_func,
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								            meta_fields={"entity_name", "source_id", "content", "file_path"},
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								        )
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        self.relationships_vdb: BaseVectorStorage = self.vector_db_storage_cls(  # type: ignore
-												use namespace as neo4j database name

format

fix

											
										
										
											2025-02-08 16:06:07 +08:00
+								            namespace=make_namespace(
 								                self.namespace_prefix, NameSpace.VECTOR_STORE_RELATIONSHIPS
 								            ),
-												chore: added pre-commit-hooks and ruff formatting for commit-hooks

											
										
										
											2024-10-19 09:43:17 +05:30
+								            embedding_func=self.embedding_func,
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								            meta_fields={"src_id", "tgt_id", "source_id", "content", "file_path"},
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								        )
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        self.chunks_vdb: BaseVectorStorage = self.vector_db_storage_cls(  # type: ignore
-												use namespace as neo4j database name

format

fix

											
										
										
											2025-02-08 16:06:07 +08:00
+								            namespace=make_namespace(
 								                self.namespace_prefix, NameSpace.VECTOR_STORE_CHUNKS
 								            ),
-												chore: added pre-commit-hooks and ruff formatting for commit-hooks

											
										
										
											2024-10-19 09:43:17 +05:30
+								            embedding_func=self.embedding_func,
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								            meta_fields={"full_doc_id", "content", "file_path"},
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								        )
-												chore: added pre-commit-hooks and ruff formatting for commit-hooks

											
										
										
											2024-10-19 09:43:17 +05:30
-												refactor: improve database initialization by centralizing db instance injection

- Move db configs to separate methods
- Remove db field defaults in storage classes
- Add _initialize_database_if_needed method
- Inject db instances during initialization
- Clean up storage implementation code

											
										
										
											2025-02-12 22:25:34 +08:00
+								        # Initialize document status storage
 								        self.doc_status: DocStatusStorage = self.doc_status_storage_cls(
 								            namespace=make_namespace(self.namespace_prefix, NameSpace.DOC_STATUS),
 								            global_config=global_config,
 								            embedding_func=None,
 								        )
-												Unify llm_response_cache and hashing_kv, prevent creating an independent hashing_kv.

											
										
										
											2025-03-09 22:15:26 +08:00
+								        # Directly use llm_response_cache, don't create a new object
 								        hashing_kv = self.llm_response_cache
-												code clean

											
										
										
											2025-02-14 23:33:59 +01:00
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								        self.llm_model_func = limit_async_func_call(self.llm_model_max_async)(
-												Fix lint issue

											
										
										
											2024-10-28 17:05:38 +02:00
+								            partial(
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								                self.llm_model_func,  # type: ignore
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
+								                hashing_kv=hashing_kv,
-												Fix lint issue

											
										
										
											2024-10-28 17:05:38 +02:00
+								                **self.llm_model_kwargs,
 								            )
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								        )
-												fix event loop conflict

											
										
										
											2024-11-06 11:18:14 -05:00
-												cleanup storage state

											
										
										
											2025-02-20 13:30:30 +01:00
+								        self._storages_status = StoragesStatus.CREATED
-												refactor database connection management and improve storage lifecycle handling

update

											
										
										
											2025-02-19 03:46:18 +08:00
-												rename is_managed_by_server to auto_manage_storages_states

											
										
										
											2025-02-19 05:27:38 +08:00
+								        if self.auto_manage_storages_states:
-												fix this event loop is already running

											
										
										
											2025-02-25 04:16:22 +07:00
+								            self._run_async_safely(self.initialize_storages, "Storage Initialization")
-												refactor database connection management and improve storage lifecycle handling

update

											
										
										
											2025-02-19 03:46:18 +08:00
 								    def __del__(self):
-												rename is_managed_by_server to auto_manage_storages_states

											
										
										
											2025-02-19 05:27:38 +08:00
+								        if self.auto_manage_storages_states:
-												fix this event loop is already running

											
										
										
											2025-02-25 04:16:22 +07:00
+								            self._run_async_safely(self.finalize_storages, "Storage Finalization")
 								    def _run_async_safely(self, async_func, action_name=""):
 								        """Safely execute an async function, avoiding event loop conflicts."""
 								        try:
-												refactor database connection management and improve storage lifecycle handling

update

											
										
										
											2025-02-19 03:46:18 +08:00
+								            loop = always_get_an_event_loop()
-												fix this event loop is already running

											
										
										
											2025-02-25 04:16:22 +07:00
+								            if loop.is_running():
 								                task = loop.create_task(async_func())
 								                task.add_done_callback(
-												remove character ticks

											
										
										
											2025-02-25 04:18:52 +07:00
+								                    lambda t: logger.info(f"{action_name} completed!")
-												fix this event loop is already running

											
										
										
											2025-02-25 04:16:22 +07:00
+								                )
 								            else:
 								                loop.run_until_complete(async_func())
 								        except RuntimeError:
 								            logger.warning(
 								                f"No running event loop, creating a new loop for {action_name}."
 								            )
 								            loop = asyncio.new_event_loop()
 								            loop.run_until_complete(async_func())
 								            loop.close()
-												refactor database connection management and improve storage lifecycle handling

update

											
										
										
											2025-02-19 03:46:18 +08:00
 								    async def initialize_storages(self):
 								        """Asynchronously initialize the storages"""
-												cleanup storage state

											
										
										
											2025-02-20 13:30:30 +01:00
+								        if self._storages_status == StoragesStatus.CREATED:
-												refactor database connection management and improve storage lifecycle handling

update

											
										
										
											2025-02-19 03:46:18 +08:00
+								            tasks = []
 								            for storage in (
 								                self.full_docs,
 								                self.text_chunks,
 								                self.entities_vdb,
 								                self.relationships_vdb,
 								                self.chunks_vdb,
 								                self.chunk_entity_relation_graph,
 								                self.llm_response_cache,
 								                self.doc_status,
 								            ):
 								                if storage:
 								                    tasks.append(storage.initialize())
 								            await asyncio.gather(*tasks)
-												cleanup storage state

											
										
										
											2025-02-20 13:30:30 +01:00
+								            self._storages_status = StoragesStatus.INITIALIZED
-												refactor database connection management and improve storage lifecycle handling

update

											
										
										
											2025-02-19 03:46:18 +08:00
+								            logger.debug("Initialized Storages")
 								    async def finalize_storages(self):
 								        """Asynchronously finalize the storages"""
-												cleanup storage state

											
										
										
											2025-02-20 13:30:30 +01:00
+								        if self._storages_status == StoragesStatus.INITIALIZED:
-												refactor database connection management and improve storage lifecycle handling

update

											
										
										
											2025-02-19 03:46:18 +08:00
+								            tasks = []
 								            for storage in (
 								                self.full_docs,
 								                self.text_chunks,
 								                self.entities_vdb,
 								                self.relationships_vdb,
 								                self.chunks_vdb,
 								                self.chunk_entity_relation_graph,
 								                self.llm_response_cache,
 								                self.doc_status,
 								            ):
 								                if storage:
 								                    tasks.append(storage.finalize())
 								            await asyncio.gather(*tasks)
-												cleanup storage state

											
										
										
											2025-02-20 13:30:30 +01:00
+								            self._storages_status = StoragesStatus.FINALIZED
-												improve MongoDB client management and storage init

											
										
										
											2025-02-19 04:30:52 +08:00
+								            logger.debug("Finalized Storages")
-												refactor database connection management and improve storage lifecycle handling

update

											
										
										
											2025-02-19 03:46:18 +08:00
-												Revert "Cleanup of code"

											
										
										
											2025-02-20 15:09:43 +01:00
+								    async def get_graph_labels(self):
 								        text = await self.chunk_entity_relation_graph.get_all_labels()
 								        return text
-												Revert "removed get_knowledge_graph"

											
										
										
											2025-02-20 14:29:36 +01:00
+								    async def get_knowledge_graph(
-												Added minimum degree filter for graph queries

- Introduced min_degree parameter in graph query
- Updated UI to include minimum degree setting
- Modified API to handle min_degree parameter
- Updated graph query logic in LightRAG

											
										
										
											2025-03-05 11:37:55 +08:00
+								        self,
 								        node_label: str,
-												Set default max_depth to 3 for knowledge graph retrieval

											
										
										
											2025-03-07 07:34:29 +08:00
+								        max_depth: int = 3,
-												Update graph retrival api(abandon pydantic model)

											
										
										
											2025-04-02 18:32:03 +08:00
+								        max_nodes: int = 1000,
-												Revert "removed get_knowledge_graph"

											
										
										
											2025-02-20 14:29:36 +01:00
+								    ) -> KnowledgeGraph:
-												Added search mode and min degree filtering for NetworkX

- Implemented exact and inclusive search modes
- Added min degree filtering for nodes
- Updated API to parse label for search options

											
										
										
											2025-03-04 16:08:05 +08:00
+								        """Get knowledge graph for a given label
 								        Args:
 								            node_label (str): Label to get knowledge graph for
 								            max_depth (int): Maximum depth of graph
-												Update graph retrival api(abandon pydantic model)

											
										
										
											2025-04-02 18:32:03 +08:00
+								            max_nodes (int, optional): Maximum number of nodes to return. Defaults to 1000.
-												Added search mode and min degree filtering for NetworkX

- Implemented exact and inclusive search modes
- Added min degree filtering for nodes
- Updated API to parse label for search options

											
										
										
											2025-03-04 16:08:05 +08:00
 								        Returns:
 								            KnowledgeGraph: Knowledge graph containing nodes and edges
 								        """
-												Refactor code and update environment type definitions.

- Consolidate type definitions in vite-env.d.ts
- Update TypeScript include paths

											
										
										
											2025-03-07 08:17:25 +08:00
-												Fix linting

											
										
										
											2025-04-02 18:36:05 +08:00
+								        return await self.chunk_entity_relation_graph.get_knowledge_graph(
 								            node_label, max_depth, max_nodes
 								        )
-												Revert "removed get_knowledge_graph"

											
										
										
											2025-02-20 14:29:36 +01:00
-												cleaning the mess

											
										
										
											2025-02-14 22:50:49 +01:00
+								    def _get_storage_class(self, storage_name: str) -> Callable[..., Any]:
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
+								        import_path = STORAGES[storage_name]
 								        storage_class = lazy_external_import(import_path, storage_name)
 								        return storage_class
-												support pipeline mode

											
										
										
											2025-01-16 12:58:15 +08:00
-												增加仅字符分割参数，如果开启，仅采用字符分割，不开启，在分割完以后如果chunk过大，会继续根据token size分割，更新测试文件

											
										
										
											2025-01-09 11:55:49 +08:00
+								    def insert(
-												cleaned code

											
										
										
											2025-02-09 13:18:47 +01:00
+								        self,
-												cleaning the mess

											
										
										
											2025-02-14 22:50:49 +01:00
+								        input: str | list[str],
-												added docs

											
										
										
											2025-02-09 11:29:05 +01:00
+								        split_by_character: str | None = None,
 								        split_by_character_only: bool = False,
-												add support for the single document and custom chunks method

											
										
										
											2025-02-26 14:41:10 +08:00
+								        ids: str | list[str] | None = None,
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								        file_paths: str | list[str] | None = None,
-												cleaned typing

											
										
										
											2025-02-18 21:16:52 +01:00
+								    ) -> None:
-												added docs

											
										
										
											2025-02-09 11:29:05 +01:00
+								        """Sync Insert documents with checkpoint support
 								        Args:
-												cleaning the mess

											
										
										
											2025-02-14 22:50:49 +01:00
+								            input: Single document string or list of document strings
-												added docs

											
										
										
											2025-02-09 11:29:05 +01:00
+								            split_by_character: if split_by_character is not None, split the string by character, if chunk longer than
-												Update code comments in ainsert method

Update code comments in ainsert method. The original comment was cut off in the middle, not a complete sentence, and cannot be read
											
										
										
											2025-03-14 10:59:24 +08:00
+								            chunk_token_size, it will be split again by token size.
-												added docs

											
										
										
											2025-02-09 11:29:05 +01:00
+								            split_by_character_only: if split_by_character_only is True, split the string by character only, when
 								            split_by_character is None, this parameter is ignored.
-												add support for the single document and custom chunks method

											
										
										
											2025-02-26 14:41:10 +08:00
+								            ids: single string of the document ID or list of unique document IDs, if not provided, MD5 hash IDs will be generated
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								            file_paths: single string of the file path or list of file paths, used for citation
-												cleaned code

											
										
										
											2025-02-09 13:18:47 +01:00
+								        """
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								        loop = always_get_an_event_loop()
-												cleaned typing

											
										
										
											2025-02-18 21:16:52 +01:00
+								        loop.run_until_complete(
-												fix lint

											
										
										
											2025-03-17 23:36:00 +08:00
+								            self.ainsert(
 								                input, split_by_character, split_by_character_only, ids, file_paths
 								            )
-												chunk split retry

											
										
										
											2025-01-07 16:26:12 +08:00
+								        )
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
-												增加仅字符分割参数，如果开启，仅采用字符分割，不开启，在分割完以后如果chunk过大，会继续根据token size分割，更新测试文件

											
										
										
											2025-01-09 11:55:49 +08:00
+								    async def ainsert(
-												cleaned import

											
										
										
											2025-02-09 11:24:08 +01:00
+								        self,
-												cleaning the mess

											
										
										
											2025-02-14 22:50:49 +01:00
+								        input: str | list[str],
-												cleaned import

											
										
										
											2025-02-09 11:24:08 +01:00
+								        split_by_character: str | None = None,
 								        split_by_character_only: bool = False,
-												add support for the single document and custom chunks method

											
										
										
											2025-02-26 14:41:10 +08:00
+								        ids: str | list[str] | None = None,
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								        file_paths: str | list[str] | None = None,
-												cleaned typing

											
										
										
											2025-02-18 21:16:52 +01:00
+								    ) -> None:
-												added docs

											
										
										
											2025-02-09 11:29:05 +01:00
+								        """Async Insert documents with checkpoint support
-												feat(lightrag): Add document status tracking and checkpoint support
功能(lightrag): 添加文档状态跟踪和断点续传支持

- Add DocStatus enum and DocProcessingStatus class for document processing state management
- 添加 DocStatus 枚举和 DocProcessingStatus 类用于文档处理状态管理

- Implement JsonDocStatusStorage for persistent status storage
- 实现 JsonDocStatusStorage 用于持久化状态存储

- Add document-level deduplication in batch processing
- 在批处理中添加文档级别的去重功能

- Add checkpoint support in ainsert method for resumable document processing
- 在 ainsert 方法中添加断点续传支持，实现可恢复的文档处理

- Add status query methods for monitoring processing progress
- 添加状态查询方法用于监控处理进度

- Update LightRAG initialization to support document status tracking
- 更新 LightRAG 初始化以支持文档状态跟踪

											
										
										
											2024-12-28 00:11:25 +08:00
 								        Args:
-												cleaning the mess

											
										
										
											2025-02-14 22:50:49 +01:00
+								            input: Single document string or list of document strings
-												增加仅字符分割参数，如果开启，仅采用字符分割，不开启，在分割完以后如果chunk过大，会继续根据token size分割，更新测试文件

											
										
										
											2025-01-09 11:55:49 +08:00
+								            split_by_character: if split_by_character is not None, split the string by character, if chunk longer than
-												Update code comments in ainsert method

Update code comments in ainsert method. The original comment was cut off in the middle, not a complete sentence, and cannot be read
											
										
										
											2025-03-14 10:59:24 +08:00
+								            chunk_token_size, it will be split again by token size.
-												增加仅字符分割参数，如果开启，仅采用字符分割，不开启，在分割完以后如果chunk过大，会继续根据token size分割，更新测试文件

											
										
										
											2025-01-09 11:55:49 +08:00
+								            split_by_character_only: if split_by_character_only is True, split the string by character only, when
 								            split_by_character is None, this parameter is ignored.
-												add support of providing ids for documents insert

											
										
										
											2025-02-20 00:26:35 +01:00
+								            ids: list of unique document IDs, if not provided, MD5 hash IDs will be generated
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								            file_paths: list of file paths corresponding to each document, used for citation
-												feat(lightrag): Add document status tracking and checkpoint support
功能(lightrag): 添加文档状态跟踪和断点续传支持

- Add DocStatus enum and DocProcessingStatus class for document processing state management
- 添加 DocStatus 枚举和 DocProcessingStatus 类用于文档处理状态管理

- Implement JsonDocStatusStorage for persistent status storage
- 实现 JsonDocStatusStorage 用于持久化状态存储

- Add document-level deduplication in batch processing
- 在批处理中添加文档级别的去重功能

- Add checkpoint support in ainsert method for resumable document processing
- 在 ainsert 方法中添加断点续传支持，实现可恢复的文档处理

- Add status query methods for monitoring processing progress
- 添加状态查询方法用于监控处理进度

- Update LightRAG initialization to support document status tracking
- 更新 LightRAG 初始化以支持文档状态跟踪

											
										
										
											2024-12-28 00:11:25 +08:00
+								        """
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								        await self.apipeline_enqueue_documents(input, ids, file_paths)
-												improved get status

											
										
										
											2025-02-09 15:24:52 +01:00
+								        await self.apipeline_process_enqueue_documents(
 								            split_by_character, split_by_character_only
 								        )
-												feat(lightrag): Add document status tracking and checkpoint support
功能(lightrag): 添加文档状态跟踪和断点续传支持

- Add DocStatus enum and DocProcessingStatus class for document processing state management
- 添加 DocStatus 枚举和 DocProcessingStatus 类用于文档处理状态管理

- Implement JsonDocStatusStorage for persistent status storage
- 实现 JsonDocStatusStorage 用于持久化状态存储

- Add document-level deduplication in batch processing
- 在批处理中添加文档级别的去重功能

- Add checkpoint support in ainsert method for resumable document processing
- 在 ainsert 方法中添加断点续传支持，实现可恢复的文档处理

- Add status query methods for monitoring processing progress
- 添加状态查询方法用于监控处理进度

- Update LightRAG initialization to support document status tracking
- 更新 LightRAG 初始化以支持文档状态跟踪

											
										
										
											2024-12-28 00:11:25 +08:00
-												Refactor pipeline status updates and entity extraction.

- Let all parrallel jobs using one pipe_status objects
- Improved thread safety with pipeline_status_lock
- Only pipeline jobs can add message to pipe_status
- Marked insert_custom_chunks as deprecated

											
										
										
											2025-03-10 16:48:59 +08:00
+								    # TODO: deprecated, use insert instead
-												fixed lint

											
										
										
											2025-02-26 12:11:28 +01:00
+								    def insert_custom_chunks(
 								        self,
 								        full_text: str,
 								        text_chunks: list[str],
 								        doc_id: str | list[str] | None = None,
 								    ) -> None:
-												Implement custom chunking feature

											
										
										
											2025-01-07 20:57:39 +05:30
+								        loop = always_get_an_event_loop()
-												fixed lint

											
										
										
											2025-02-26 12:11:28 +01:00
+								        loop.run_until_complete(
 								            self.ainsert_custom_chunks(full_text, text_chunks, doc_id)
 								        )
-												Implement custom chunking feature

											
										
										
											2025-01-07 20:57:39 +05:30
-												Refactor pipeline status updates and entity extraction.

- Let all parrallel jobs using one pipe_status objects
- Improved thread safety with pipeline_status_lock
- Only pipeline jobs can add message to pipe_status
- Marked insert_custom_chunks as deprecated

											
										
										
											2025-03-10 16:48:59 +08:00
+								    # TODO: deprecated, use ainsert instead
-												cleaned typing

											
										
										
											2025-02-18 21:16:52 +01:00
+								    async def ainsert_custom_chunks(
-												add support for the single document and custom chunks method

											
										
										
											2025-02-26 14:41:10 +08:00
+								        self, full_text: str, text_chunks: list[str], doc_id: str | None = None
-												cleaned typing

											
										
										
											2025-02-18 21:16:52 +01:00
+								    ) -> None:
-												Implement custom chunking feature

											
										
										
											2025-01-07 20:57:39 +05:30
+								        update_storage = False
 								        try:
-												fix: handle null bytes (0x00) in text processing

- Fix PostgreSQL encoding error by properly handling null bytes (0x00) in text processing.
- The clean_text function now removes null bytes from all input text during the indexing phase.

											
										
										
											2025-02-21 13:18:26 +08:00
+								            # Clean input texts
-												clean lightrag.py

											
										
										
											2025-03-11 15:43:04 +08:00
+								            full_text = clean_text(full_text)
 								            text_chunks = [clean_text(chunk) for chunk in text_chunks]
-												fix: handle null bytes (0x00) in text processing

- Fix PostgreSQL encoding error by properly handling null bytes (0x00) in text processing.
- The clean_text function now removes null bytes from all input text during the indexing phase.

											
										
										
											2025-02-21 13:18:26 +08:00
 								            # Process cleaned texts
-												add support for the single document and custom chunks method

											
										
										
											2025-02-26 14:41:10 +08:00
+								            if doc_id is None:
 								                doc_key = compute_mdhash_id(full_text, prefix="doc-")
 								            else:
 								                doc_key = doc_id
-												fix: handle null bytes (0x00) in text processing

- Fix PostgreSQL encoding error by properly handling null bytes (0x00) in text processing.
- The clean_text function now removes null bytes from all input text during the indexing phase.

											
										
										
											2025-02-21 13:18:26 +08:00
+								            new_docs = {doc_key: {"content": full_text}}
-												Implement custom chunking feature

											
										
										
											2025-01-07 20:57:39 +05:30
-												fix insert_custom_chunks skipping every new doc with "This document is already in the storage."

											
										
										
											2025-02-20 23:08:36 +01:00
+								            _add_doc_keys = await self.full_docs.filter_keys({doc_key})
-												Implement custom chunking feature

											
										
										
											2025-01-07 20:57:39 +05:30
+								            new_docs = {k: v for k, v in new_docs.items() if k in _add_doc_keys}
 								            if not len(new_docs):
 								                logger.warning("This document is already in the storage.")
 								                return
 								            update_storage = True
-												cleaned code

											
										
										
											2025-02-19 22:07:25 +01:00
+								            logger.info(f"Inserting {len(new_docs)} docs")
-												Implement custom chunking feature

											
										
										
											2025-01-07 20:57:39 +05:30
-												cleaned set

											
										
										
											2025-02-09 19:56:12 +01:00
+								            inserting_chunks: dict[str, Any] = {}
-												Implement custom chunking feature

											
										
										
											2025-01-07 20:57:39 +05:30
+								            for chunk_text in text_chunks:
-												fix: handle null bytes (0x00) in text processing

- Fix PostgreSQL encoding error by properly handling null bytes (0x00) in text processing.
- The clean_text function now removes null bytes from all input text during the indexing phase.

											
										
										
											2025-02-21 13:18:26 +08:00
+								                chunk_key = compute_mdhash_id(chunk_text, prefix="chunk-")
-												Fix trailing whitespace and formatting issues in lightrag.py

											
										
										
											2025-01-09 00:39:22 +05:30
-												Implement custom chunking feature

											
										
										
											2025-01-07 20:57:39 +05:30
+								                inserting_chunks[chunk_key] = {
-												fix: handle null bytes (0x00) in text processing

- Fix PostgreSQL encoding error by properly handling null bytes (0x00) in text processing.
- The clean_text function now removes null bytes from all input text during the indexing phase.

											
										
										
											2025-02-21 13:18:26 +08:00
+								                    "content": chunk_text,
-												Implement custom chunking feature

											
										
										
											2025-01-07 20:57:39 +05:30
+								                    "full_doc_id": doc_key,
 								                }
-												cleaned set

											
										
										
											2025-02-09 19:56:12 +01:00
+								            doc_ids = set(inserting_chunks.keys())
 								            add_chunk_keys = await self.text_chunks.filter_keys(doc_ids)
-												Implement custom chunking feature

											
										
										
											2025-01-07 20:57:39 +05:30
+								            inserting_chunks = {
-												cleaned set

											
										
										
											2025-02-09 19:56:12 +01:00
+								                k: v for k, v in inserting_chunks.items() if k in add_chunk_keys
-												Implement custom chunking feature

											
										
										
											2025-01-07 20:57:39 +05:30
+								            }
 								            if not len(inserting_chunks):
 								                logger.warning("All chunks are already in the storage.")
 								                return
-												improving ainsert_custom_chunks paralelism

											
										
										
											2025-02-09 21:42:04 +01:00
+								            tasks = [
 								                self.chunks_vdb.upsert(inserting_chunks),
 								                self._process_entity_relation_graph(inserting_chunks),
 								                self.full_docs.upsert(new_docs),
 								                self.text_chunks.upsert(inserting_chunks),
 								            ]
 								            await asyncio.gather(*tasks)
-												Implement custom chunking feature

											
										
										
											2025-01-07 20:57:39 +05:30
 								        finally:
 								            if update_storage:
 								                await self._insert_done()
-												add support of providing ids for documents insert

											
										
										
											2025-02-20 00:26:35 +01:00
+								    async def apipeline_enqueue_documents(
-												fix lint

											
										
										
											2025-03-17 23:36:00 +08:00
+								        self,
 								        input: str | list[str],
 								        ids: list[str] | None = None,
 								        file_paths: str | list[str] | None = None,
-												add support of providing ids for documents insert

											
										
										
											2025-02-20 00:26:35 +01:00
+								    ) -> None:
-												cleaned docs

											
										
										
											2025-02-09 14:39:32 +01:00
+								        """
 								        Pipeline for Processing Documents
-												improved get status

											
										
										
											2025-02-09 15:24:52 +01:00
-												add support of providing ids for documents insert

											
										
										
											2025-02-20 00:26:35 +01:00
+. Validate ids if provided or generate MD5 hash IDs
 . Remove duplicate contents
 . Generate document initial status
 . Filter out already processed documents
 . Enqueue document in status
-												fix lint

											
										
										
											2025-03-17 23:36:00 +08:00
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								        Args:
 								            input: Single document string or list of document strings
 								            ids: list of unique document IDs, if not provided, MD5 hash IDs will be generated
 								            file_paths: list of file paths corresponding to each document, used for citation
-												improved get status

											
										
										
											2025-02-09 15:24:52 +01:00
+								        """
-												cleaning the mess

											
										
										
											2025-02-14 22:50:49 +01:00
+								        if isinstance(input, str):
 								            input = [input]
-												add support for the single document and custom chunks method

											
										
										
											2025-02-26 14:41:10 +08:00
+								        if isinstance(ids, str):
 								            ids = [ids]
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								        if isinstance(file_paths, str):
 								            file_paths = [file_paths]
-												fix lint

											
										
										
											2025-03-17 23:36:00 +08:00
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								        # If file_paths is provided, ensure it matches the number of documents
 								        if file_paths is not None:
 								            if isinstance(file_paths, str):
 								                file_paths = [file_paths]
 								            if len(file_paths) != len(input):
-												fix lint

											
										
										
											2025-03-17 23:36:00 +08:00
+								                raise ValueError(
 								                    "Number of file paths must match the number of documents"
 								                )
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								        else:
 								            # If no file paths provided, use placeholder
 								            file_paths = ["unknown_source"] * len(input)
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
-												add support of providing ids for documents insert

											
										
										
											2025-02-20 00:26:35 +01:00
+								        # 1. Validate ids if provided or generate MD5 hash IDs
 								        if ids is not None:
 								            # Check if the number of IDs matches the number of documents
 								            if len(ids) != len(input):
 								                raise ValueError("Number of IDs must match the number of documents")
 								            # Check if IDs are unique
 								            if len(ids) != len(set(ids)):
 								                raise ValueError("IDs must be unique")
 								            # Generate contents dict of IDs provided by user and documents
-												fix lint

											
										
										
											2025-03-17 23:36:00 +08:00
+								            contents = {
 								                id_: {"content": doc, "file_path": path}
 								                for id_, doc, path in zip(ids, input, file_paths)
 								            }
-												add support of providing ids for documents insert

											
										
										
											2025-02-20 00:26:35 +01:00
+								        else:
-												fix: make ids parameter optional and optimize input text cleaning

- Add default None value for ids parameter
- Move text cleaning into else branch
- Only clean text when auto-generating ids
- Preserve original text with custom ids
- Improve code readability

											
										
										
											2025-02-23 15:46:47 +08:00
+								            # Clean input text and remove duplicates
-												fix lint

											
										
										
											2025-03-17 23:36:00 +08:00
+								            cleaned_input = [
 								                (clean_text(doc), path) for doc, path in zip(input, file_paths)
 								            ]
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								            unique_content_with_paths = {}
-												fix lint

											
										
										
											2025-03-17 23:36:00 +08:00
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								            # Keep track of unique content and their paths
 								            for content, path in cleaned_input:
 								                if content not in unique_content_with_paths:
 								                    unique_content_with_paths[content] = path
-												fix lint

											
										
										
											2025-03-17 23:36:00 +08:00
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								            # Generate contents dict of MD5 hash IDs and documents with paths
-												fix lint

											
										
										
											2025-03-17 23:36:00 +08:00
+								            contents = {
 								                compute_mdhash_id(content, prefix="doc-"): {
 								                    "content": content,
 								                    "file_path": path,
 								                }
 								                for content, path in unique_content_with_paths.items()
 								            }
-												add support of providing ids for documents insert

											
										
										
											2025-02-20 00:26:35 +01:00
 								        # 2. Remove duplicate contents
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								        unique_contents = {}
 								        for id_, content_data in contents.items():
 								            content = content_data["content"]
 								            file_path = content_data["file_path"]
 								            if content not in unique_contents:
 								                unique_contents[content] = (id_, file_path)
-												fix lint

											
										
										
											2025-03-17 23:36:00 +08:00
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								        # Reconstruct contents with unique content
-												fix lint

											
										
										
											2025-03-17 23:36:00 +08:00
+								        contents = {
 								            id_: {"content": content, "file_path": file_path}
 								            for content, (id_, file_path) in unique_contents.items()
 								        }
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
-												add support of providing ids for documents insert

											
										
										
											2025-02-20 00:26:35 +01:00
+								        # 3. Generate document initial status
-												cleaned insert by using pipe

											
										
										
											2025-02-09 11:10:46 +01:00
+								        new_docs: dict[str, Any] = {
-												add support of providing ids for documents insert

											
										
										
											2025-02-20 00:26:35 +01:00
+								            id_: {
-												fixed str enum

											
										
										
											2025-02-17 18:26:07 +01:00
+								                "status": DocStatus.PENDING,
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								                "content": content_data["content"],
 								                "content_summary": get_content_summary(content_data["content"]),
 								                "content_length": len(content_data["content"]),
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
+								                "created_at": datetime.now().isoformat(),
-												cleaned insert by using pipe

											
										
										
											2025-02-09 11:10:46 +01:00
+								                "updated_at": datetime.now().isoformat(),
-												fix lint

											
										
										
											2025-03-17 23:36:00 +08:00
+								                "file_path": content_data[
 								                    "file_path"
 								                ],  # Store file path in document status
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
+								            }
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								            for id_, content_data in contents.items()
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
+								        }
-												add support of providing ids for documents insert

											
										
										
											2025-02-20 00:26:35 +01:00
+								        # 4. Filter out already processed documents
-												cleaned code

											
										
										
											2025-02-09 14:55:52 +01:00
+								        # Get docs ids
-												fixed typo

											
										
										
											2025-02-09 19:24:41 +01:00
+								        all_new_doc_ids = set(new_docs.keys())
 								        # Exclude IDs of documents that are already in progress
-												improved parallele

											
										
										
											2025-02-09 21:17:09 +01:00
+								        unique_new_doc_ids = await self.doc_status.filter_keys(all_new_doc_ids)
-												Improved file handling and validation for document processing

• Enhanced UTF-8 validation for text files
• Added content validation checks
• Better handling of binary data
• Added logging for ignored document IDs
• Improved document ID filtering

											
										
										
											2025-03-02 23:57:57 +08:00
 								        # Log ignored document IDs
 								        ignored_ids = [
 								            doc_id for doc_id in unique_new_doc_ids if doc_id not in new_docs
 								        ]
 								        if ignored_ids:
 								            logger.warning(
 								                f"Ignoring {len(ignored_ids)} document IDs not found in new_docs"
 								            )
 								            for doc_id in ignored_ids:
 								                logger.warning(f"Ignored document ID: {doc_id}")
-												fixed typo

											
										
										
											2025-02-09 19:24:41 +01:00
+								        # Filter new_docs to only include documents with unique IDs
-												Improved file handling and validation for document processing

• Enhanced UTF-8 validation for text files
• Added content validation checks
• Better handling of binary data
• Added logging for ignored document IDs
• Improved document ID filtering

											
										
										
											2025-03-02 23:57:57 +08:00
+								        new_docs = {
 								            doc_id: new_docs[doc_id]
 								            for doc_id in unique_new_doc_ids
 								            if doc_id in new_docs
 								        }
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
 								        if not new_docs:
-												Fix cache bugs

											
										
										
											2025-02-11 13:28:18 +08:00
+								            logger.info("No new unique documents were found.")
-												cleaned insert by using pipe

											
										
										
											2025-02-09 11:10:46 +01:00
+								            return
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
-												add support of providing ids for documents insert

											
										
										
											2025-02-20 00:26:35 +01:00
+								        # 5. Store status document
-												cleaned code

											
										
										
											2025-02-09 13:18:47 +01:00
+								        await self.doc_status.upsert(new_docs)
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
+								        logger.info(f"Stored {len(new_docs)} new unique documents")
-												support pipeline mode

											
										
										
											2025-01-16 12:58:15 +08:00
-												updated naming

											
										
										
											2025-02-09 14:32:48 +01:00
+								    async def apipeline_process_enqueue_documents(
-												cleaned import

											
										
										
											2025-02-09 11:24:08 +01:00
+								        self,
 								        split_by_character: str | None = None,
 								        split_by_character_only: bool = False,
 								    ) -> None:
-												added docs

											
										
										
											2025-02-09 11:30:54 +01:00
+								        """
-												updated naming

											
										
										
											2025-02-09 14:32:48 +01:00
+								        Process pending documents by splitting them into chunks, processing
-												cleaned docs

											
										
										
											2025-02-09 14:36:49 +01:00
+								        each chunk for entity and relation extraction, and updating the
-												updated naming

											
										
										
											2025-02-09 14:32:48 +01:00
+								        document status.
-												cleaned docs

											
										
										
											2025-02-09 14:36:49 +01:00
-												Fix cache bugs

											
										
										
											2025-02-11 13:28:18 +08:00
+. Get all pending, failed, and abnormally terminated processing documents.
-												updated naming

											
										
										
											2025-02-09 14:32:48 +01:00
+. Split document content into chunks
 . Process each chunk for entity and relation extraction
 . Update the document status
-												cleaned docs

											
										
										
											2025-02-09 14:36:49 +01:00
+								        """
-												Add pipeline status control for concurrent document indexing processes

• Add shared pipeline status namespace
• Implement concurrent process control
• Add request queuing for pending jobs

											
										
										
											2025-02-28 11:52:42 +08:00
+								        # Get pipeline status shared data and lock
-												refactor: migrate synchronous locks to async locks for improved concurrency

• Add UnifiedLock wrapper class
• Convert with blocks to async with

											
										
										
											2025-03-01 02:22:35 +08:00
+								        pipeline_status = await get_namespace_data("pipeline_status")
-												Refactor shared storage locks to separate pipeline, storage and internal locks for deadlock preventing

											
										
										
											2025-03-01 10:48:55 +08:00
+								        pipeline_status_lock = get_pipeline_status_lock()
-												Fix linting

											
										
										
											2025-02-28 21:35:04 +08:00
-												Add pipeline status control for concurrent document indexing processes

• Add shared pipeline status namespace
• Implement concurrent process control
• Add request queuing for pending jobs

											
										
										
											2025-02-28 11:52:42 +08:00
+								        # Check if another process is already processing the queue
-												Refactor shared storage locks to separate pipeline, storage and internal locks for deadlock preventing

											
										
										
											2025-03-01 10:48:55 +08:00
+								        async with pipeline_status_lock:
-												Fix linting

											
										
										
											2025-02-28 21:35:04 +08:00
+								            # Ensure only one worker is processing documents
-												Add pipeline status control for concurrent document indexing processes

• Add shared pipeline status namespace
• Implement concurrent process control
• Add request queuing for pending jobs

											
										
										
											2025-02-28 11:52:42 +08:00
+								            if not pipeline_status.get("busy", False):
-												Optimize document processing pipeline with better status tracking & batch handling

• Add upfront doc processing check
• Optimize pipeline status updates

											
										
										
											2025-03-02 11:09:32 +08:00
+								                processing_docs, failed_docs, pending_docs = await asyncio.gather(
 								                    self.doc_status.get_docs_by_status(DocStatus.PROCESSING),
 								                    self.doc_status.get_docs_by_status(DocStatus.FAILED),
 								                    self.doc_status.get_docs_by_status(DocStatus.PENDING),
 								                )
-												cleaned docs

											
										
										
											2025-02-09 14:36:49 +01:00
-												Optimize document processing pipeline with better status tracking & batch handling

• Add upfront doc processing check
• Optimize pipeline status updates

											
										
										
											2025-03-02 11:09:32 +08:00
+								                to_process_docs: dict[str, DocProcessingStatus] = {}
 								                to_process_docs.update(processing_docs)
 								                to_process_docs.update(failed_docs)
 								                to_process_docs.update(pending_docs)
 								                if not to_process_docs:
 								                    logger.info("No documents to process")
 								                    return
-												Fix linting

											
										
										
											2025-02-28 21:35:04 +08:00
+								                pipeline_status.update(
 								                    {
 								                        "busy": True,
-												fix: optimize job name handling in document processing pipeline

- Move job name setting to before batch processing
- Fix document and batch counter accumulation

											
										
										
											2025-03-26 16:58:31 +08:00
+								                        "job_name": "Default Job",
-												Fix linting

											
										
										
											2025-02-28 21:35:04 +08:00
+								                        "job_start": datetime.now().isoformat(),
 								                        "docs": 0,
 								                        "batchs": 0,
 								                        "cur_batch": 0,
 								                        "request_pending": False,  # Clear any previous request
 								                        "latest_message": "",
-												multi batches

											
										
										
											2025-02-19 23:53:25 +01:00
+								                    }
-												Fix linting

											
										
										
											2025-02-28 21:35:04 +08:00
+								                )
-												Fix history_messages clearing in LightRAG pipeline status initialization

											
										
										
											2025-03-02 04:43:41 +08:00
+								                # Cleaning history_messages without breaking it as a shared list object
-												Optimize document processing pipeline with better status tracking & batch handling

• Add upfront doc processing check
• Optimize pipeline status updates

											
										
										
											2025-03-02 11:09:32 +08:00
+								                del pipeline_status["history_messages"][:]
-												Add pipeline status control for concurrent document indexing processes

• Add shared pipeline status namespace
• Implement concurrent process control
• Add request queuing for pending jobs

											
										
										
											2025-02-28 11:52:42 +08:00
+								            else:
 								                # Another process is busy, just set request flag and return
 								                pipeline_status["request_pending"] = True
-												Fix linting

											
										
										
											2025-02-28 21:35:04 +08:00
+								                logger.info(
 								                    "Another process is already processing the document queue. Request queued."
 								                )
-												Optimize document processing pipeline with better status tracking & batch handling

• Add upfront doc processing check
• Optimize pipeline status updates

											
										
										
											2025-03-02 11:09:32 +08:00
+								                return
-												Fix linting

											
										
										
											2025-02-28 21:35:04 +08:00
-												Add pipeline status control for concurrent document indexing processes

• Add shared pipeline status namespace
• Implement concurrent process control
• Add request queuing for pending jobs

											
										
										
											2025-02-28 11:52:42 +08:00
+								        try:
 								            # Process documents until no more documents or requests
 								            while True:
 								                if not to_process_docs:
-												feat: add history_messages to track pipeline processing progress

• Add shared history_messages list
• Track pipeline progress with messages

											
										
										
											2025-02-28 13:53:40 +08:00
+								                    log_message = "All documents have been processed or are duplicates"
 								                    logger.info(log_message)
 								                    pipeline_status["latest_message"] = log_message
 								                    pipeline_status["history_messages"].append(log_message)
-												Add pipeline status control for concurrent document indexing processes

• Add shared pipeline status namespace
• Implement concurrent process control
• Add request queuing for pending jobs

											
										
										
											2025-02-28 11:52:42 +08:00
+								                    break
 								                # 2. split docs into chunks, insert chunks, update doc status
 								                docs_batches = [
 								                    list(to_process_docs.items())[i : i + self.max_parallel_insert]
 								                    for i in range(0, len(to_process_docs), self.max_parallel_insert)
 								                ]
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                log_message = f"Processing {len(to_process_docs)} document(s) in {len(docs_batches)} batches"
-												feat: add history_messages to track pipeline processing progress

• Add shared history_messages list
• Track pipeline progress with messages

											
										
										
											2025-02-28 13:53:40 +08:00
+								                logger.info(log_message)
-												Optimize document processing pipeline with better status tracking & batch handling

• Add upfront doc processing check
• Optimize pipeline status updates

											
										
										
											2025-03-02 11:09:32 +08:00
 								                # Update pipeline status with current batch information
-												fix: optimize job name handling in document processing pipeline

- Move job name setting to before batch processing
- Fix document and batch counter accumulation

											
										
										
											2025-03-26 16:58:31 +08:00
+								                pipeline_status["docs"] = len(to_process_docs)
 								                pipeline_status["batchs"] = len(docs_batches)
-												feat: add history_messages to track pipeline processing progress

• Add shared history_messages list
• Track pipeline progress with messages

											
										
										
											2025-02-28 13:53:40 +08:00
+								                pipeline_status["latest_message"] = log_message
 								                pipeline_status["history_messages"].append(log_message)
-												Add pipeline status control for concurrent document indexing processes

• Add shared pipeline status namespace
• Implement concurrent process control
• Add request queuing for pending jobs

											
										
										
											2025-02-28 11:52:42 +08:00
-												fix: optimize job name handling in document processing pipeline

- Move job name setting to before batch processing
- Fix document and batch counter accumulation

											
										
										
											2025-03-26 16:58:31 +08:00
+								                # Get first document's file path and total count for job name
 								                first_doc_id, first_doc = next(iter(to_process_docs.items()))
 								                first_doc_path = first_doc.file_path
-												Fix linting

											
										
										
											2025-03-26 17:30:06 +08:00
+								                path_prefix = first_doc_path[:20] + (
 								                    "..." if len(first_doc_path) > 20 else ""
 								                )
-												fix: optimize job name handling in document processing pipeline

- Move job name setting to before batch processing
- Fix document and batch counter accumulation

											
										
										
											2025-03-26 16:58:31 +08:00
+								                total_files = len(to_process_docs)
 								                job_name = f"{path_prefix}[{total_files} files]"
 								                pipeline_status["job_name"] = job_name
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                async def process_document(
-												Fix linting

											
										
										
											2025-03-17 04:11:25 +08:00
+								                    doc_id: str,
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                    status_doc: DocProcessingStatus,
 								                    split_by_character: str | None,
 								                    split_by_character_only: bool,
 								                    pipeline_status: dict,
-												Fix linting

											
										
										
											2025-03-17 04:11:25 +08:00
+								                    pipeline_status_lock: asyncio.Lock,
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                ) -> None:
 								                    """Process single document"""
 								                    try:
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								                        # Get file path from status document
 								                        file_path = getattr(status_doc, "file_path", "unknown_source")
-												fix lint

											
										
										
											2025-03-17 23:36:00 +08:00
-												Optimize pipeline status message

											
										
										
											2025-04-10 21:19:26 +08:00
+								                        async with pipeline_status_lock:
 								                            log_message = f"Processing file: {file_path}"
-												Update log message

											
										
										
											2025-04-12 21:34:50 +08:00
+								                            logger.info(log_message)
-												Optimize pipeline status message

											
										
										
											2025-04-10 21:19:26 +08:00
+								                            pipeline_status["history_messages"].append(log_message)
 								                            log_message = f"Processing d-id: {doc_id}"
-												Update log message

											
										
										
											2025-04-12 21:34:50 +08:00
+								                            logger.info(log_message)
-												Optimize pipeline status message

											
										
										
											2025-04-10 21:19:26 +08:00
+								                            pipeline_status["latest_message"] = log_message
 								                            pipeline_status["history_messages"].append(log_message)
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                        # Generate chunks from document
 								                        chunks: dict[str, Any] = {
 								                            compute_mdhash_id(dp["content"], prefix="chunk-"): {
 								                                **dp,
 								                                "full_doc_id": doc_id,
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								                                "file_path": file_path,  # Add file path to each chunk
-												multi batches

											
										
										
											2025-02-19 23:53:25 +01:00
+								                            }
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                            for dp in self.chunking_func(
 								                                status_doc.content,
 								                                split_by_character,
 								                                split_by_character_only,
 								                                self.chunk_overlap_token_size,
 								                                self.chunk_token_size,
 								                                self.tiktoken_model_name,
-												No need the await entity_relation_task first

											
										
										
											2025-03-04 15:30:52 +08:00
+								                            )
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                        }
-												fix lint

											
										
										
											2025-03-17 23:36:00 +08:00
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                        # Process document (text chunks and full docs) in parallel
 								                        # Create tasks with references for potential cancellation
 								                        doc_status_task = asyncio.create_task(
 								                            self.doc_status.upsert(
 								                                {
 								                                    doc_id: {
 								                                        "status": DocStatus.PROCESSING,
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								                                        "chunks_count": len(chunks),
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                                        "content": status_doc.content,
 								                                        "content_summary": status_doc.content_summary,
 								                                        "content_length": status_doc.content_length,
 								                                        "created_at": status_doc.created_at,
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								                                        "updated_at": datetime.now().isoformat(),
 								                                        "file_path": file_path,
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                                    }
 								                                }
-												No need the await entity_relation_task first

											
										
										
											2025-03-04 15:30:52 +08:00
+								                            )
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                        )
 								                        chunks_vdb_task = asyncio.create_task(
 								                            self.chunks_vdb.upsert(chunks)
 								                        )
 								                        entity_relation_task = asyncio.create_task(
 								                            self._process_entity_relation_graph(
 								                                chunks, pipeline_status, pipeline_status_lock
-												Improved task handling and error management in LightRAG

- Created tasks with references for cancellation
- Prioritized entity relation task execution
- Implemented task cancellation on failure
- Added error logging to pipeline_status

											
										
										
											2025-03-04 14:59:50 +08:00
+								                            )
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                        )
 								                        full_docs_task = asyncio.create_task(
 								                            self.full_docs.upsert(
 								                                {doc_id: {"content": status_doc.content}}
-												No need the await entity_relation_task first

											
										
										
											2025-03-04 15:30:52 +08:00
+								                            )
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                        )
 								                        text_chunks_task = asyncio.create_task(
 								                            self.text_chunks.upsert(chunks)
 								                        )
 								                        tasks = [
 								                            doc_status_task,
 								                            chunks_vdb_task,
 								                            entity_relation_task,
 								                            full_docs_task,
 								                            text_chunks_task,
 								                        ]
 								                        await asyncio.gather(*tasks)
 								                        await self.doc_status.upsert(
 								                            {
 								                                doc_id: {
 								                                    "status": DocStatus.PROCESSED,
 								                                    "chunks_count": len(chunks),
 								                                    "content": status_doc.content,
 								                                    "content_summary": status_doc.content_summary,
 								                                    "content_length": status_doc.content_length,
 								                                    "created_at": status_doc.created_at,
 								                                    "updated_at": datetime.now().isoformat(),
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								                                    "file_path": file_path,
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                                }
 								                            }
 								                        )
 								                    except Exception as e:
 								                        # Log error and update pipeline status
-												Fix linting

											
										
										
											2025-03-17 04:11:25 +08:00
+								                        error_msg = f"Failed to process document {doc_id}: {str(e)}"
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                        logger.error(error_msg)
 								                        async with pipeline_status_lock:
 								                            pipeline_status["latest_message"] = error_msg
 								                            pipeline_status["history_messages"].append(error_msg)
 								                            # Cancel other tasks as they are no longer meaningful
 								                            for task in [
-												No need the await entity_relation_task first

											
										
										
											2025-03-04 15:30:52 +08:00
+								                                chunks_vdb_task,
 								                                entity_relation_task,
 								                                full_docs_task,
 								                                text_chunks_task,
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                            ]:
 								                                if not task.done():
 								                                    task.cancel()
 								                        # Update document status to failed
 								                        await self.doc_status.upsert(
 								                            {
 								                                doc_id: {
 								                                    "status": DocStatus.FAILED,
 								                                    "error": str(e),
 								                                    "content": status_doc.content,
 								                                    "content_summary": status_doc.content_summary,
 								                                    "content_length": status_doc.content_length,
 								                                    "created_at": status_doc.created_at,
 								                                    "updated_at": datetime.now().isoformat(),
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								                                    "file_path": file_path,
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                                }
 								                            }
-												multi batches

											
										
										
											2025-02-19 23:53:25 +01:00
+								                        )
-												Add pipeline status control for concurrent document indexing processes

• Add shared pipeline status namespace
• Implement concurrent process control
• Add request queuing for pending jobs

											
										
										
											2025-02-28 11:52:42 +08:00
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                # 3. iterate over batches
-												Fix linting

											
										
										
											2025-03-17 04:11:25 +08:00
+								                total_batches = len(docs_batches)
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                for batch_idx, docs_batch in enumerate(docs_batches):
-												Fix linting

											
										
										
											2025-03-17 04:11:25 +08:00
+								                    current_batch = batch_idx + 1
 								                    log_message = (
 								                        f"Start processing batch {current_batch} of {total_batches}."
 								                    )
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                    logger.info(log_message)
 								                    pipeline_status["cur_batch"] = current_batch
 								                    pipeline_status["latest_message"] = log_message
 								                    pipeline_status["history_messages"].append(log_message)
-												Fix linting

											
										
										
											2025-03-17 04:11:25 +08:00
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                    doc_tasks = []
 								                    for doc_id, status_doc in docs_batch:
 								                        doc_tasks.append(
 								                            process_document(
-												Fix linting

											
										
										
											2025-03-17 04:11:25 +08:00
+								                                doc_id,
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                                status_doc,
 								                                split_by_character,
 								                                split_by_character_only,
 								                                pipeline_status,
-												Fix linting

											
										
										
											2025-03-17 04:11:25 +08:00
+								                                pipeline_status_lock,
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                            )
 								                        )
-												Fix linting

											
										
										
											2025-03-17 04:11:25 +08:00
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                    # Process documents in one batch parallelly
 								                    await asyncio.gather(*doc_tasks)
 								                    await self._insert_done()
-												Fix linting

											
										
										
											2025-03-17 04:11:25 +08:00
-												Fix pipeline bactch process problem

- Process batch one by one
- Process documents in parallel within each batch

											
										
										
											2025-03-17 04:00:38 +08:00
+								                    log_message = f"Completed batch {current_batch} of {total_batches}."
 								                    logger.info(log_message)
 								                    pipeline_status["latest_message"] = log_message
 								                    pipeline_status["history_messages"].append(log_message)
-												Fix linting

											
										
										
											2025-02-28 21:35:04 +08:00
-												Add pipeline status control for concurrent document indexing processes

• Add shared pipeline status namespace
• Implement concurrent process control
• Add request queuing for pending jobs

											
										
										
											2025-02-28 11:52:42 +08:00
+								                # Check if there's a pending request to process more documents (with lock)
 								                has_pending_request = False
-												Refactor shared storage locks to separate pipeline, storage and internal locks for deadlock preventing

											
										
										
											2025-03-01 10:48:55 +08:00
+								                async with pipeline_status_lock:
-												Add pipeline status control for concurrent document indexing processes

• Add shared pipeline status namespace
• Implement concurrent process control
• Add request queuing for pending jobs

											
										
										
											2025-02-28 11:52:42 +08:00
+								                    has_pending_request = pipeline_status.get("request_pending", False)
 								                    if has_pending_request:
 								                        # Clear the request flag before checking for more documents
 								                        pipeline_status["request_pending"] = False
-												Fix linting

											
										
										
											2025-02-28 21:35:04 +08:00
-												Add pipeline status control for concurrent document indexing processes

• Add shared pipeline status namespace
• Implement concurrent process control
• Add request queuing for pending jobs

											
										
										
											2025-02-28 11:52:42 +08:00
+								                if not has_pending_request:
 								                    break
-												Fix linting

											
										
										
											2025-02-28 21:35:04 +08:00
-												feat: add history_messages to track pipeline processing progress

• Add shared history_messages list
• Track pipeline progress with messages

											
										
										
											2025-02-28 13:53:40 +08:00
+								                log_message = "Processing additional documents due to pending request"
 								                logger.info(log_message)
 								                pipeline_status["latest_message"] = log_message
 								                pipeline_status["history_messages"].append(log_message)
-												Fix linting

											
										
										
											2025-02-28 21:35:04 +08:00
-												No need the await entity_relation_task first

											
										
										
											2025-03-04 15:30:52 +08:00
+								                # Check for pending documents again
-												Optimize document processing pipeline with better status tracking & batch handling

• Add upfront doc processing check
• Optimize pipeline status updates

											
										
										
											2025-03-02 11:09:32 +08:00
+								                processing_docs, failed_docs, pending_docs = await asyncio.gather(
 								                    self.doc_status.get_docs_by_status(DocStatus.PROCESSING),
 								                    self.doc_status.get_docs_by_status(DocStatus.FAILED),
 								                    self.doc_status.get_docs_by_status(DocStatus.PENDING),
 								                )
-												multi batches

											
										
										
											2025-02-19 23:53:25 +01:00
-												Optimize document processing pipeline with better status tracking & batch handling

• Add upfront doc processing check
• Optimize pipeline status updates

											
										
										
											2025-03-02 11:09:32 +08:00
+								                to_process_docs = {}
 								                to_process_docs.update(processing_docs)
 								                to_process_docs.update(failed_docs)
 								                to_process_docs.update(pending_docs)
-												multi batches

											
										
										
											2025-02-19 23:53:25 +01:00
-												Add pipeline status control for concurrent document indexing processes

• Add shared pipeline status namespace
• Implement concurrent process control
• Add request queuing for pending jobs

											
										
										
											2025-02-28 11:52:42 +08:00
+								        finally:
-												feat: add history_messages to track pipeline processing progress

• Add shared history_messages list
• Track pipeline progress with messages

											
										
										
											2025-02-28 13:53:40 +08:00
+								            log_message = "Document processing pipeline completed"
 								            logger.info(log_message)
-												refactor: migrate synchronous locks to async locks for improved concurrency

• Add UnifiedLock wrapper class
• Convert with blocks to async with

											
										
										
											2025-03-01 02:22:35 +08:00
+								            # Always reset busy status when done or if an exception occurs (with lock)
-												Refactor shared storage locks to separate pipeline, storage and internal locks for deadlock preventing

											
										
										
											2025-03-01 10:48:55 +08:00
+								            async with pipeline_status_lock:
-												refactor: migrate synchronous locks to async locks for improved concurrency

• Add UnifiedLock wrapper class
• Convert with blocks to async with

											
										
										
											2025-03-01 02:22:35 +08:00
+								                pipeline_status["busy"] = False
 								                pipeline_status["latest_message"] = log_message
 								                pipeline_status["history_messages"].append(log_message)
-												support pipeline mode

											
										
										
											2025-01-16 12:52:37 +08:00
-												Fix linting

											
										
										
											2025-03-10 17:30:40 +08:00
+								    async def _process_entity_relation_graph(
 								        self, chunk: dict[str, Any], pipeline_status=None, pipeline_status_lock=None
 								    ) -> None:
-												cleaned code

											
										
										
											2025-02-09 13:18:47 +01:00
+								        try:
-												cleanup extraction

											
										
										
											2025-02-20 14:17:26 +01:00
+								            await extract_entities(
-												cleaned code

											
										
										
											2025-02-09 13:18:47 +01:00
+								                chunk,
 								                knowledge_graph_inst=self.chunk_entity_relation_graph,
 								                entity_vdb=self.entities_vdb,
 								                relationships_vdb=self.relationships_vdb,
 								                global_config=asdict(self),
-												Refactor pipeline status updates and entity extraction.

- Let all parrallel jobs using one pipe_status objects
- Improved thread safety with pipeline_status_lock
- Only pipeline jobs can add message to pipe_status
- Marked insert_custom_chunks as deprecated

											
										
										
											2025-03-10 16:48:59 +08:00
+								                pipeline_status=pipeline_status,
 								                pipeline_status_lock=pipeline_status_lock,
 								                llm_response_cache=self.llm_response_cache,
-												cleaned code

											
										
										
											2025-02-09 13:18:47 +01:00
+								            )
 								        except Exception as e:
 								            logger.error("Failed to extract entities and relationships")
 								            raise e
-												Fix linting

											
										
										
											2025-03-10 17:30:40 +08:00
+								    async def _insert_done(
 								        self, pipeline_status=None, pipeline_status_lock=None
 								    ) -> None:
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        tasks = [
 								            cast(StorageNameSpace, storage_inst).index_done_callback()
 								            for storage_inst in [  # type: ignore
 								                self.full_docs,
 								                self.text_chunks,
 								                self.llm_response_cache,
 								                self.entities_vdb,
 								                self.relationships_vdb,
 								                self.chunks_vdb,
 								                self.chunk_entity_relation_graph,
 								            ]
 								            if storage_inst is not None
 								        ]
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								        await asyncio.gather(*tasks)
-												Fix linting

											
										
										
											2025-02-28 21:35:04 +08:00
-												Update log message for in-memory DB persistence

											
										
										
											2025-03-17 04:25:23 +08:00
+								        log_message = "In memory DB persist to disk"
-												feat: add history_messages to track pipeline processing progress

• Add shared history_messages list
• Track pipeline progress with messages

											
										
										
											2025-02-28 13:53:40 +08:00
+								        logger.info(log_message)
-												Fix linting

											
										
										
											2025-02-28 21:35:04 +08:00
-												Refactor pipeline status updates and entity extraction.

- Let all parrallel jobs using one pipe_status objects
- Improved thread safety with pipeline_status_lock
- Only pipeline jobs can add message to pipe_status
- Marked insert_custom_chunks as deprecated

											
										
										
											2025-03-10 16:48:59 +08:00
+								        if pipeline_status is not None and pipeline_status_lock is not None:
 								            async with pipeline_status_lock:
 								                pipeline_status["latest_message"] = log_message
 								                pipeline_status["history_messages"].append(log_message)
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
-												fix linting

											
										
										
											2025-03-03 14:54:28 +08:00
+								    def insert_custom_kg(
 								        self, custom_kg: dict[str, Any], full_doc_id: str = None
 								    ) -> None:
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
+								        loop = always_get_an_event_loop()
-												adding full_doc_id to insert

											
										
										
											2025-03-01 13:26:02 +01:00
+								        loop.run_until_complete(self.ainsert_custom_kg(custom_kg, full_doc_id))
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
-												fix linting

											
										
										
											2025-03-03 14:54:28 +08:00
+								    async def ainsert_custom_kg(
-												fix lint

											
										
										
											2025-03-17 23:36:00 +08:00
+								        self,
 								        custom_kg: dict[str, Any],
 								        full_doc_id: str = None,
 								        file_path: str = "custom_kg",
-												fix linting

											
										
										
											2025-03-03 14:54:28 +08:00
+								    ) -> None:
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
+								        update_storage = False
 								        try:
-												update insert custom kg

											
										
										
											2024-12-04 19:44:04 +08:00
+								            # Insert chunks into vector storage
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								            all_chunks_data: dict[str, dict[str, str]] = {}
 								            chunk_to_source_map: dict[str, str] = {}
-												Add a feature that allows modifying nodes and relationships.

											
										
										
											2025-03-03 21:09:45 +08:00
+								            for chunk_data in custom_kg.get("chunks", []):
-												clean lightrag.py

											
										
										
											2025-03-11 15:43:04 +08:00
+								                chunk_content = clean_text(chunk_data["content"])
-												update insert custom kg

											
										
										
											2024-12-04 19:44:04 +08:00
+								                source_id = chunk_data["source_id"]
-												Applied lint

											
										
										
											2025-02-19 10:28:25 +01:00
+								                tokens = len(
 								                    encode_string_by_tiktoken(
 								                        chunk_content, model_name=self.tiktoken_model_name
 								                    )
 								                )
 								                chunk_order_index = (
 
 								                    if "chunk_order_index" not in chunk_data.keys()
 								                    else chunk_data["chunk_order_index"]
 								                )
-												Fixed formatting

											
										
										
											2025-02-17 15:25:50 +01:00
+								                chunk_id = compute_mdhash_id(chunk_content, prefix="chunk-")
-												update insert custom kg

											
										
										
											2024-12-04 19:44:04 +08:00
-												Fixed broken ainsert_custom_kg()

											
										
										
											2025-02-17 15:12:35 +01:00
+								                chunk_entry = {
-												Fixed formatting

											
										
										
											2025-02-17 15:25:50 +01:00
+								                    "content": chunk_content,
-												Fixed broken ainsert_custom_kg()

											
										
										
											2025-02-17 15:12:35 +01:00
+								                    "source_id": source_id,
-												Renamed chunk_order_index and improve token calculation

											
										
										
											2025-02-19 07:15:30 +01:00
+								                    "tokens": tokens,
 								                    "chunk_order_index": chunk_order_index,
-												fix linting

											
										
										
											2025-03-03 14:54:28 +08:00
+								                    "full_doc_id": full_doc_id
 								                    if full_doc_id is not None
 								                    else source_id,
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								                    "file_path": file_path,  # Add file path
-												Fixed formatting

											
										
										
											2025-02-17 15:25:50 +01:00
+								                    "status": DocStatus.PROCESSED,
-												Fixed broken ainsert_custom_kg()

											
										
										
											2025-02-17 15:12:35 +01:00
+								                }
-												update insert custom kg

											
										
										
											2024-12-04 19:44:04 +08:00
+								                all_chunks_data[chunk_id] = chunk_entry
 								                chunk_to_source_map[source_id] = chunk_id
 								                update_storage = True
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								            if all_chunks_data:
-												Add a feature that allows modifying nodes and relationships.

											
										
										
											2025-03-03 21:09:45 +08:00
+								                await asyncio.gather(
 								                    self.chunks_vdb.upsert(all_chunks_data),
 								                    self.text_chunks.upsert(all_chunks_data),
 								                )
-												update insert custom kg

											
										
										
											2024-12-04 19:44:04 +08:00
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
+								            # Insert entities into knowledge graph
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								            all_entities_data: list[dict[str, str]] = []
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
+								            for entity_data in custom_kg.get("entities", []):
-												remove " and upper()

											
										
										
											2025-03-02 14:23:06 +08:00
+								                entity_name = entity_data["entity_name"]
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
+								                entity_type = entity_data.get("entity_type", "UNKNOWN")
 								                description = entity_data.get("description", "No description provided")
-												update insert custom kg

											
										
										
											2024-12-04 19:44:04 +08:00
+								                source_chunk_id = entity_data.get("source_id", "UNKNOWN")
 								                source_id = chunk_to_source_map.get(source_chunk_id, "UNKNOWN")
 								                # Log if source_id is UNKNOWN
 								                if source_id == "UNKNOWN":
 								                    logger.warning(
 								                        f"Entity '{entity_name}' has an UNKNOWN source_id. Please check the source mapping."
 								                    )
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
 								                # Prepare node data
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								                node_data: dict[str, str] = {
-												Update lightrag.py

											
										
										
											2025-03-13 16:52:48 +08:00
+								                    "entity_id": entity_name,
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
+								                    "entity_type": entity_type,
 								                    "description": description,
 								                    "source_id": source_id,
 								                }
 								                # Insert node data into the knowledge graph
 								                await self.chunk_entity_relation_graph.upsert_node(
 								                    entity_name, node_data=node_data
 								                )
 								                node_data["entity_name"] = entity_name
 								                all_entities_data.append(node_data)
 								                update_storage = True
 								            # Insert relationships into knowledge graph
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								            all_relationships_data: list[dict[str, str]] = []
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
+								            for relationship_data in custom_kg.get("relationships", []):
-												remove " and upper()

											
										
										
											2025-03-02 14:23:06 +08:00
+								                src_id = relationship_data["src_id"]
 								                tgt_id = relationship_data["tgt_id"]
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
+								                description = relationship_data["description"]
 								                keywords = relationship_data["keywords"]
 								                weight = relationship_data.get("weight", 1.0)
-												update insert custom kg

											
										
										
											2024-12-04 19:44:04 +08:00
+								                source_chunk_id = relationship_data.get("source_id", "UNKNOWN")
 								                source_id = chunk_to_source_map.get(source_chunk_id, "UNKNOWN")
 								                # Log if source_id is UNKNOWN
 								                if source_id == "UNKNOWN":
 								                    logger.warning(
 								                        f"Relationship from '{src_id}' to '{tgt_id}' has an UNKNOWN source_id. Please check the source mapping."
 								                    )
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
 								                # Check if nodes exist in the knowledge graph
 								                for need_insert_id in [src_id, tgt_id]:
 								                    if not (
-												chunk split retry

											
										
										
											2025-01-07 16:26:12 +08:00
+								                        await self.chunk_entity_relation_graph.has_node(need_insert_id)
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
+								                    ):
 								                        await self.chunk_entity_relation_graph.upsert_node(
 								                            need_insert_id,
 								                            node_data={
-												Update lightrag.py

											
										
										
											2025-03-13 16:52:48 +08:00
+								                                "entity_id": need_insert_id,
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
+								                                "source_id": source_id,
 								                                "description": "UNKNOWN",
 								                                "entity_type": "UNKNOWN",
 								                            },
 								                        )
 								                # Insert edge into the knowledge graph
 								                await self.chunk_entity_relation_graph.upsert_edge(
 								                    src_id,
 								                    tgt_id,
 								                    edge_data={
 								                        "weight": weight,
 								                        "description": description,
 								                        "keywords": keywords,
 								                        "source_id": source_id,
 								                    },
 								                )
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								                edge_data: dict[str, str] = {
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
+								                    "src_id": src_id,
 								                    "tgt_id": tgt_id,
 								                    "description": description,
 								                    "keywords": keywords,
-												Add a feature that allows modifying nodes and relationships.

											
										
										
											2025-03-03 21:09:45 +08:00
+								                    "source_id": source_id,
 								                    "weight": weight,
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
+								                }
 								                all_relationships_data.append(edge_data)
 								                update_storage = True
-												Add a feature that allows modifying nodes and relationships.

											
										
										
											2025-03-03 21:09:45 +08:00
+								            # Insert entities into vector storage with consistent format
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								            data_for_vdb = {
-												code clean

											
										
										
											2025-02-14 23:33:59 +01:00
+								                compute_mdhash_id(dp["entity_name"], prefix="ent-"): {
-												Add a feature that allows modifying nodes and relationships.

											
										
										
											2025-03-03 21:09:45 +08:00
+								                    "content": dp["entity_name"] + "\n" + dp["description"],
-												code clean

											
										
										
											2025-02-14 23:33:59 +01:00
+								                    "entity_name": dp["entity_name"],
-												Add a feature that allows modifying nodes and relationships.

											
										
										
											2025-03-03 21:09:45 +08:00
+								                    "source_id": dp["source_id"],
 								                    "description": dp["description"],
 								                    "entity_type": dp["entity_type"],
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								                    "file_path": file_path,  # Add file path
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
+								                }
-												code clean

											
										
										
											2025-02-14 23:33:59 +01:00
+								                for dp in all_entities_data
 								            }
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								            await self.entities_vdb.upsert(data_for_vdb)
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
-												Add a feature that allows modifying nodes and relationships.

											
										
										
											2025-03-03 21:09:45 +08:00
+								            # Insert relationships into vector storage with consistent format
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								            data_for_vdb = {
-												code clean

											
										
										
											2025-02-14 23:33:59 +01:00
+								                compute_mdhash_id(dp["src_id"] + dp["tgt_id"], prefix="rel-"): {
 								                    "src_id": dp["src_id"],
 								                    "tgt_id": dp["tgt_id"],
-												Add a feature that allows modifying nodes and relationships.

											
										
										
											2025-03-03 21:09:45 +08:00
+								                    "source_id": dp["source_id"],
 								                    "content": f"{dp['keywords']}\t{dp['src_id']}\n{dp['tgt_id']}\n{dp['description']}",
 								                    "keywords": dp["keywords"],
 								                    "description": dp["description"],
 								                    "weight": dp["weight"],
-												add citation

											
										
										
											2025-03-17 23:32:35 +08:00
+								                    "file_path": file_path,  # Add file path
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
+								                }
-												code clean

											
										
										
											2025-02-14 23:33:59 +01:00
+								                for dp in all_relationships_data
 								            }
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								            await self.relationships_vdb.upsert(data_for_vdb)
-												code clean

											
										
										
											2025-02-14 23:33:59 +01:00
-												Add a feature that allows modifying nodes and relationships.

											
										
										
											2025-03-03 21:09:45 +08:00
+								        except Exception as e:
 								            logger.error(f"Error in ainsert_custom_kg: {e}")
 								            raise
-												Add custom KG insertion

											
										
										
											2024-11-25 18:06:19 +08:00
+								        finally:
 								            if update_storage:
 								                await self._insert_done()
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								    def query(
-												Added system prompt support in all modes

											
										
										
											2025-02-17 16:45:00 +05:30
+								        self,
 								        query: str,
 								        param: QueryParam = QueryParam(),
 								        system_prompt: str | None = None,
-												added type and cleaned code

											
										
										
											2025-02-14 23:42:52 +01:00
+								    ) -> str | Iterator[str]:
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        """
 								        Perform a sync query.
 								        Args:
 								            query (str): The query to be executed.
 								            param (QueryParam): Configuration parameters for query execution.
 								            prompt (Optional[str]): Custom prompts for fine-tuned control over the system's behavior. Defaults to None, which uses PROMPTS["rag_response"].
 								        Returns:
 								            str: The result of the query execution.
-												code clean

											
										
										
											2025-02-14 23:33:59 +01:00
+								        """
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								        loop = always_get_an_event_loop()
-												cleaned code

											
										
										
											2025-02-14 23:52:05 +01:00
-												Added system prompt support in all modes

											
										
										
											2025-02-17 16:45:00 +05:30
+								        return loop.run_until_complete(self.aquery(query, param, system_prompt))  # type: ignore
-												chore: added pre-commit-hooks and ruff formatting for commit-hooks

											
										
										
											2024-10-19 09:43:17 +05:30
-												Query with your custom prompts

											
										
										
											2025-01-27 10:32:22 +05:30
+								    async def aquery(
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        self,
 								        query: str,
 								        param: QueryParam = QueryParam(),
-												Added system prompt support in all modes

											
										
										
											2025-02-17 16:45:00 +05:30
+								        system_prompt: str | None = None,
-												added type and cleaned code

											
										
										
											2025-02-14 23:42:52 +01:00
+								    ) -> str | AsyncIterator[str]:
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								        """
 								        Perform a async query.
 								        Args:
 								            query (str): The query to be executed.
 								            param (QueryParam): Configuration parameters for query execution.
-												specify LLM for query

											
										
										
											2025-03-23 21:33:49 +05:30
+								                If param.model_func is provided, it will be used instead of the global model.
-												cleaning the message and project no needed

											
										
										
											2025-02-14 23:31:27 +01:00
+								            prompt (Optional[str]): Custom prompts for fine-tuned control over the system's behavior. Defaults to None, which uses PROMPTS["rag_response"].
 								        Returns:
 								            str: The result of the query execution.
 								        """
-												specify LLM for query

											
										
										
											2025-03-23 21:33:49 +05:30
+								        # If a custom model is provided in param, temporarily update global config
 								        global_config = asdict(self)
-												linting errors

											
										
										
											2025-03-25 15:20:09 +05:30
-												Optimization logic

											
										
										
											2024-11-25 13:29:55 +08:00
+								        if param.mode in ["local", "global", "hybrid"]:
 								            response = await kg_query(
-												Improved cashing check

											
										
										
											2025-03-03 13:53:45 +05:30
+								                query.strip(),
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								                self.chunk_entity_relation_graph,
 								                self.entities_vdb,
 								                self.relationships_vdb,
 								                self.text_chunks,
 								                param,
-												specify LLM for query

											
										
										
											2025-03-23 21:33:49 +05:30
+								                global_config,
-												Unify llm_response_cache and hashing_kv, prevent creating an independent hashing_kv.

											
										
										
											2025-03-09 22:15:26 +08:00
+								                hashing_kv=self.llm_response_cache,  # Directly use llm_response_cache
-												Added system prompt support in all modes

											
										
										
											2025-02-17 16:45:00 +05:30
+								                system_prompt=system_prompt,
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								            )
 								        elif param.mode == "naive":
 								            response = await naive_query(
-												Improved cashing check

											
										
										
											2025-03-03 13:53:45 +05:30
+								                query.strip(),
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								                self.chunks_vdb,
 								                self.text_chunks,
 								                param,
-												specify LLM for query

											
										
										
											2025-03-23 21:33:49 +05:30
+								                global_config,
-												Unify llm_response_cache and hashing_kv, prevent creating an independent hashing_kv.

											
										
										
											2025-03-09 22:15:26 +08:00
+								                hashing_kv=self.llm_response_cache,  # Directly use llm_response_cache
-												Added system prompt support in all modes

											
										
										
											2025-02-17 16:45:00 +05:30
+								                system_prompt=system_prompt,
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								            )
-												feat(lightrag): Implement mix search mode combining knowledge graph and vector retrieval

- Add 'mix' mode to QueryParam for hybrid search functionality
- Implement mix_kg_vector_query to combine knowledge graph and vector search results
- Update LightRAG class to handle 'mix' mode queries
- Enhance README with examples and explanations for the new mix search mode
- Introduce new prompt structure for generating responses based on combined search results

											
										
										
											2024-12-28 11:56:28 +08:00
+								        elif param.mode == "mix":
 								            response = await mix_kg_vector_query(
-												Improved cashing check

											
										
										
											2025-03-03 13:53:45 +05:30
+								                query.strip(),
-												feat(lightrag): Implement mix search mode combining knowledge graph and vector retrieval

- Add 'mix' mode to QueryParam for hybrid search functionality
- Implement mix_kg_vector_query to combine knowledge graph and vector search results
- Update LightRAG class to handle 'mix' mode queries
- Enhance README with examples and explanations for the new mix search mode
- Introduce new prompt structure for generating responses based on combined search results

											
										
										
											2024-12-28 11:56:28 +08:00
+								                self.chunk_entity_relation_graph,
 								                self.entities_vdb,
 								                self.relationships_vdb,
 								                self.chunks_vdb,
 								                self.text_chunks,
 								                param,
-												specify LLM for query

											
										
										
											2025-03-23 21:33:49 +05:30
+								                global_config,
-												Unify llm_response_cache and hashing_kv, prevent creating an independent hashing_kv.

											
										
										
											2025-03-09 22:15:26 +08:00
+								                hashing_kv=self.llm_response_cache,  # Directly use llm_response_cache
-												Added system prompt support in all modes

											
										
										
											2025-02-17 16:45:00 +05:30
+								                system_prompt=system_prompt,
-												feat(lightrag): Implement mix search mode combining knowledge graph and vector retrieval

- Add 'mix' mode to QueryParam for hybrid search functionality
- Implement mix_kg_vector_query to combine knowledge graph and vector search results
- Update LightRAG class to handle 'mix' mode queries
- Enhance README with examples and explanations for the new mix search mode
- Introduce new prompt structure for generating responses based on combined search results

											
										
										
											2024-12-28 11:56:28 +08:00
+								            )
-												feat: Add query mode 'bypass' to bypass knowledge retrieval and directly use LLM

											
										
										
											2025-04-10 23:17:33 +08:00
+								        elif param.mode == "bypass":
 								            # Bypass mode: directly use LLM without knowledge retrieval
 								            use_llm_func = param.model_func or global_config["llm_model_func"]
-												docs(locales): Update multilingual files to include descriptions of bypass mode

											
										
										
											2025-04-11 11:12:01 +08:00
+								            param.stream = True if param.stream is None else param.stream
-												feat: Add query mode 'bypass' to bypass knowledge retrieval and directly use LLM

											
										
										
											2025-04-10 23:17:33 +08:00
+								            response = await use_llm_func(
 								                query.strip(),
 								                system_prompt=system_prompt,
 								                history_messages=param.conversation_history,
-												docs(locales): Update multilingual files to include descriptions of bypass mode

											
										
										
											2025-04-11 11:12:01 +08:00
+								                stream=param.stream,
-												feat: Add query mode 'bypass' to bypass knowledge retrieval and directly use LLM

											
										
										
											2025-04-10 23:17:33 +08:00
+								            )
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								        else:
 								            raise ValueError(f"Unknown mode {param.mode}")
 								        await self._query_done()
-												Add custom function with separate keyword extraction for user's query and a separate prompt

											
										
										
											2025-01-14 22:10:47 +05:30
+								        return response
 								    def query_with_separate_keyword_extraction(
-												cleaned code

											
										
										
											2025-02-14 23:52:05 +01:00
+								        self, query: str, prompt: str, param: QueryParam = QueryParam()
-												Add custom function with separate keyword extraction for user's query and a separate prompt

											
										
										
											2025-01-14 22:10:47 +05:30
+								    ):
 								        """
-												clean lightrag.py

											
										
										
											2025-03-11 15:43:04 +08:00
+								        Query with separate keyword extraction step.
-												fix linting

											
										
										
											2025-03-11 15:44:01 +08:00
-												clean lightrag.py

											
										
										
											2025-03-11 15:43:04 +08:00
+								        This method extracts keywords from the query first, then uses them for the query.
-												fix linting

											
										
										
											2025-03-11 15:44:01 +08:00
-												clean lightrag.py

											
										
										
											2025-03-11 15:43:04 +08:00
+								        Args:
 								            query: User query
 								            prompt: Additional prompt for the query
 								            param: Query parameters
-												fix linting

											
										
										
											2025-03-11 15:44:01 +08:00
-												clean lightrag.py

											
										
										
											2025-03-11 15:43:04 +08:00
+								        Returns:
 								            Query response
-												Add custom function with separate keyword extraction for user's query and a separate prompt

											
										
										
											2025-01-14 22:10:47 +05:30
+								        """
 								        loop = always_get_an_event_loop()
-												Fix linting errors

											
										
										
											2025-01-14 22:23:14 +05:30
+								        return loop.run_until_complete(
 								            self.aquery_with_separate_keyword_extraction(query, prompt, param)
 								        )
-												Add custom function with separate keyword extraction for user's query and a separate prompt

											
										
										
											2025-01-14 22:10:47 +05:30
+								    async def aquery_with_separate_keyword_extraction(
-												cleaned code

											
										
										
											2025-02-14 23:52:05 +01:00
+								        self, query: str, prompt: str, param: QueryParam = QueryParam()
-												cleaned code

											
										
										
											2025-02-15 00:01:21 +01:00
+								    ) -> str | AsyncIterator[str]:
-												Add custom function with separate keyword extraction for user's query and a separate prompt

											
										
										
											2025-01-14 22:10:47 +05:30
+								        """
-												clean lightrag.py

											
										
										
											2025-03-11 15:43:04 +08:00
+								        Async version of query_with_separate_keyword_extraction.
-												fix linting

											
										
										
											2025-03-11 15:44:01 +08:00
-												clean lightrag.py

											
										
										
											2025-03-11 15:43:04 +08:00
+								        Args:
 								            query: User query
 								            prompt: Additional prompt for the query
 								            param: Query parameters
-												fix linting

											
										
										
											2025-03-11 15:44:01 +08:00
-												clean lightrag.py

											
										
										
											2025-03-11 15:43:04 +08:00
+								        Returns:
 								            Query response or async iterator
-												Add custom function with separate keyword extraction for user's query and a separate prompt

											
										
										
											2025-01-14 22:10:47 +05:30
+								        """
-												clean lightrag.py

											
										
										
											2025-03-11 15:43:04 +08:00
+								        response = await query_with_keywords(
 								            query=query,
 								            prompt=prompt,
-												Add custom function with separate keyword extraction for user's query and a separate prompt

											
										
										
											2025-01-14 22:10:47 +05:30
+								            param=param,
-												clean lightrag.py

											
										
										
											2025-03-11 15:43:04 +08:00
+								            knowledge_graph_inst=self.chunk_entity_relation_graph,
 								            entities_vdb=self.entities_vdb,
 								            relationships_vdb=self.relationships_vdb,
 								            chunks_vdb=self.chunks_vdb,
 								            text_chunks_db=self.text_chunks,
-												Add custom function with separate keyword extraction for user's query and a separate prompt

											
										
										
											2025-01-14 22:10:47 +05:30
+								            global_config=asdict(self),
-												clean lightrag.py

											
										
										
											2025-03-11 15:43:04 +08:00
+								            hashing_kv=self.llm_response_cache,
-												Add custom function with separate keyword extraction for user's query and a separate prompt

											
										
										
											2025-01-14 22:10:47 +05:30
+								        )
-												fix linting

											
										
										
											2025-03-11 15:44:01 +08:00
-												Add custom function with separate keyword extraction for user's query and a separate prompt

											
										
										
											2025-01-14 22:10:47 +05:30
+								        await self._query_done()
-												update

											
										
										
											2024-10-10 15:02:30 +08:00
+								        return response
 								    async def _query_done(self):
-												cleaned code

											
										
										
											2025-02-15 00:01:21 +01:00
+								        await self.llm_response_cache.index_done_callback()
-												Add delete method

											
										
										
											2024-11-11 17:48:40 +08:00
-												feat(lightrag): Add document status tracking and checkpoint support
功能(lightrag): 添加文档状态跟踪和断点续传支持

- Add DocStatus enum and DocProcessingStatus class for document processing state management
- 添加 DocStatus 枚举和 DocProcessingStatus 类用于文档处理状态管理

- Implement JsonDocStatusStorage for persistent status storage
- 实现 JsonDocStatusStorage 用于持久化状态存储

- Add document-level deduplication in batch processing
- 在批处理中添加文档级别的去重功能

- Add checkpoint support in ainsert method for resumable document processing
- 在 ainsert 方法中添加断点续传支持，实现可恢复的文档处理

- Add status query methods for monitoring processing progress
- 添加状态查询方法用于监控处理进度

- Update LightRAG initialization to support document status tracking
- 更新 LightRAG 初始化以支持文档状态跟踪

											
										
										
											2024-12-28 00:11:25 +08:00
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								    async def aclear_cache(self, modes: list[str] | None = None) -> None:
 								        """Clear cache data from the LLM response cache storage.
-												fix linting

											
										
										
											2025-03-04 15:53:20 +08:00
-												Implement the missing methods.

											
										
										
											2025-03-04 15:50:53 +08:00
+								        Args:
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								            modes (list[str] | None): Modes of cache to clear. Options: ["default", "naive", "local", "global", "hybrid", "mix"].
 								                             "default" represents extraction cache.
 								                             If None, clears all cache.
-												Implement the missing methods.

											
										
										
											2025-03-04 15:50:53 +08:00
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								        Example:
 								            # Clear all cache
 								            await rag.aclear_cache()
-												fix linting

											
										
										
											2025-03-04 15:53:20 +08:00
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								            # Clear local mode cache
 								            await rag.aclear_cache(modes=["local"])
 								            # Clear extraction cache
 								            await rag.aclear_cache(modes=["default"])
-												Implement the missing methods.

											
										
										
											2025-03-04 15:50:53 +08:00
+								        """
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								        if not self.llm_response_cache:
 								            logger.warning("No cache storage configured")
 								            return
-												fix linting

											
										
										
											2025-03-04 15:53:20 +08:00
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								        valid_modes = ["default", "naive", "local", "global", "hybrid", "mix"]
-												fix linting

											
										
										
											2025-03-04 15:53:20 +08:00
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								        # Validate input
 								        if modes and not all(mode in valid_modes for mode in modes):
 								            raise ValueError(f"Invalid mode. Valid modes are: {valid_modes}")
-												fix linting

											
										
										
											2025-03-04 15:53:20 +08:00
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								        try:
 								            # Reset the cache storage for specified mode
 								            if modes:
 								                success = await self.llm_response_cache.drop_cache_by_modes(modes)
 								                if success:
 								                    logger.info(f"Cleared cache for modes: {modes}")
 								                else:
 								                    logger.warning(f"Failed to clear cache for modes: {modes}")
 								            else:
 								                # Clear all modes
 								                success = await self.llm_response_cache.drop_cache_by_modes(valid_modes)
 								                if success:
 								                    logger.info("Cleared all cache")
 								                else:
 								                    logger.warning("Failed to clear all cache")
-												fix linting

											
										
										
											2025-03-04 15:53:20 +08:00
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								            await self.llm_response_cache.index_done_callback()
-												Implement the missing methods.

											
										
										
											2025-03-04 15:50:53 +08:00
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								        except Exception as e:
 								            logger.error(f"Error while clearing cache: {e}")
 								    def clear_cache(self, modes: list[str] | None = None) -> None:
 								        """Synchronous version of aclear_cache."""
 								        return always_get_an_event_loop().run_until_complete(self.aclear_cache(modes))
-												feat(lightrag): Add document status tracking and checkpoint support
功能(lightrag): 添加文档状态跟踪和断点续传支持

- Add DocStatus enum and DocProcessingStatus class for document processing state management
- 添加 DocStatus 枚举和 DocProcessingStatus 类用于文档处理状态管理

- Implement JsonDocStatusStorage for persistent status storage
- 实现 JsonDocStatusStorage 用于持久化状态存储

- Add document-level deduplication in batch processing
- 在批处理中添加文档级别的去重功能

- Add checkpoint support in ainsert method for resumable document processing
- 在 ainsert 方法中添加断点续传支持，实现可恢复的文档处理

- Add status query methods for monitoring processing progress
- 添加状态查询方法用于监控处理进度

- Update LightRAG initialization to support document status tracking
- 更新 LightRAG 初始化以支持文档状态跟踪

											
										
										
											2024-12-28 00:11:25 +08:00
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
-												 implement endpoint to retrieve document statuses

											
										
										
											2025-02-17 01:03:05 +08:00
+								    async def get_docs_by_status(
 								        self, status: DocStatus
 								    ) -> dict[str, DocProcessingStatus]:
 								        """Get documents by status
 								        Returns:
 								            Dict with document id is keys and document status is values
 								        """
 								        return await self.doc_status.get_docs_by_status(status)
-												Add graph_db_lock to esure consistency across multiple processes for node and edge edition jobs

											
										
										
											2025-04-14 00:07:31 +08:00
+								    # TODO: Deprecated (Deleting documents can cause hallucinations in RAG.)
 								    # Document delete is not working properly for most of the storage implementations.
-												cleaned code

											
										
										
											2025-02-15 00:10:37 +01:00
+								    async def adelete_by_doc_id(self, doc_id: str) -> None:
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								        """Delete a document and all its related data
 								        Args:
 								            doc_id: Document ID to delete
 								        """
 								        try:
 								            # 1. Get the document status and related data
-												Fixed get_by_id type error in PSQL impl

											
										
										
											2025-03-17 17:32:54 -07:00
+								            if not await self.doc_status.get_by_id(doc_id):
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								                logger.warning(f"Document {doc_id} not found")
 								                return
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								            logger.debug(f"Starting deletion for document {doc_id}")
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
-												fix delete_by_doc_id

											
										
										
											2025-03-04 13:22:33 +08:00
+								            # 2. Get all chunks related to this document
 								            # Find all chunks where full_doc_id equals the current doc_id
 								            all_chunks = await self.text_chunks.get_all()
 								            related_chunks = {
 								                chunk_id: chunk_data
 								                for chunk_id, chunk_data in all_chunks.items()
 								                if isinstance(chunk_data, dict)
 								                and chunk_data.get("full_doc_id") == doc_id
 								            }
-												feat: fix delete by document id

											
										
										
											2025-02-27 23:34:57 +07:00
-												fix delete_by_doc_id

											
										
										
											2025-03-04 13:22:33 +08:00
+								            if not related_chunks:
 								                logger.warning(f"No chunks found for document {doc_id}")
-												cleaned code

											
										
										
											2025-02-15 00:10:37 +01:00
+								                return
-												fix delete_by_doc_id

											
										
										
											2025-03-04 13:22:33 +08:00
+								            # Get all related chunk IDs
 								            chunk_ids = set(related_chunks.keys())
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								            logger.debug(f"Found {len(chunk_ids)} chunks to delete")
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
-												feat: remove check_storage_env_vars and add TODOs

- Remove unused check_storage_env_vars method
- Add TODO to check if has_edge works on reverse relation
- Add TODO about entities_vdb.client_storage local storage limitation

											
										
										
											2025-03-30 10:25:49 +08:00
+								            # TODO: self.entities_vdb.client_storage only works for local storage, need to fix this
-												remove check_storage_env_vars from lightrag.py

											
										
										
											2025-03-30 15:25:04 +08:00
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								            # 3. Before deleting, check the related entities and relationships for these chunks
 								            for chunk_id in chunk_ids:
 								                # Check entities
-												fix delete_by_doc_id

											
										
										
											2025-03-03 19:17:34 +08:00
+								                entities_storage = await self.entities_vdb.client_storage
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								                entities = [
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
+								                    dp
-												fix delete_by_doc_id

											
										
										
											2025-03-03 19:17:34 +08:00
+								                    for dp in entities_storage["data"]
-												feat: fix delete by document id

											
										
										
											2025-02-27 23:34:57 +07:00
+								                    if chunk_id in dp.get("source_id")
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								                ]
 								                logger.debug(f"Chunk {chunk_id} has {len(entities)} related entities")
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								                # Check relationships
-												fix delete_by_doc_id

											
										
										
											2025-03-03 19:17:34 +08:00
+								                relationships_storage = await self.relationships_vdb.client_storage
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								                relations = [
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
+								                    dp
-												fix delete_by_doc_id

											
										
										
											2025-03-03 19:17:34 +08:00
+								                    for dp in relationships_storage["data"]
-												feat: fix delete by document id

											
										
										
											2025-02-27 23:34:57 +07:00
+								                    if chunk_id in dp.get("source_id")
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								                ]
 								                logger.debug(f"Chunk {chunk_id} has {len(relations)} related relations")
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								            # Continue with the original deletion process...
 								            # 4. Delete chunks from vector database
 								            if chunk_ids:
 								                await self.chunks_vdb.delete(chunk_ids)
 								                await self.text_chunks.delete(chunk_ids)
 								            # 5. Find and process entities and relationships that have these chunks as source
-												Update delete_by_doc_id

											
										
										
											2025-03-04 16:36:58 +08:00
+								            # Get all nodes and edges from the graph storage using storage-agnostic methods
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								            entities_to_delete = set()
 								            entities_to_update = {}  # entity_name -> new_source_id
 								            relationships_to_delete = set()
 								            relationships_to_update = {}  # (src, tgt) -> new_source_id
-												Update delete_by_doc_id

											
										
										
											2025-03-04 16:36:58 +08:00
+								            # Process entities - use storage-agnostic methods
 								            all_labels = await self.chunk_entity_relation_graph.get_all_labels()
 								            for node_label in all_labels:
 								                node_data = await self.chunk_entity_relation_graph.get_node(node_label)
 								                if node_data and "source_id" in node_data:
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								                    # Split source_id using GRAPH_FIELD_SEP
-												Update delete_by_doc_id

											
										
										
											2025-03-04 16:36:58 +08:00
+								                    sources = set(node_data["source_id"].split(GRAPH_FIELD_SEP))
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								                    sources.difference_update(chunk_ids)
 								                    if not sources:
-												Update delete_by_doc_id

											
										
										
											2025-03-04 16:36:58 +08:00
+								                        entities_to_delete.add(node_label)
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
+								                        logger.debug(
-												Update delete_by_doc_id

											
										
										
											2025-03-04 16:36:58 +08:00
+								                            f"Entity {node_label} marked for deletion - no remaining sources"
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
+								                        )
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								                    else:
 								                        new_source_id = GRAPH_FIELD_SEP.join(sources)
-												Update delete_by_doc_id

											
										
										
											2025-03-04 16:36:58 +08:00
+								                        entities_to_update[node_label] = new_source_id
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
+								                        logger.debug(
-												Update delete_by_doc_id

											
										
										
											2025-03-04 16:36:58 +08:00
+								                            f"Entity {node_label} will be updated with new source_id: {new_source_id}"
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
+								                        )
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
 								            # Process relationships
-												Update delete_by_doc_id

											
										
										
											2025-03-04 16:36:58 +08:00
+								            for node_label in all_labels:
 								                node_edges = await self.chunk_entity_relation_graph.get_node_edges(
 								                    node_label
 								                )
 								                if node_edges:
 								                    for src, tgt in node_edges:
 								                        edge_data = await self.chunk_entity_relation_graph.get_edge(
 								                            src, tgt
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
+								                        )
-												Update delete_by_doc_id

											
										
										
											2025-03-04 16:36:58 +08:00
+								                        if edge_data and "source_id" in edge_data:
 								                            # Split source_id using GRAPH_FIELD_SEP
 								                            sources = set(edge_data["source_id"].split(GRAPH_FIELD_SEP))
 								                            sources.difference_update(chunk_ids)
 								                            if not sources:
 								                                relationships_to_delete.add((src, tgt))
 								                                logger.debug(
 								                                    f"Relationship {src}-{tgt} marked for deletion - no remaining sources"
 								                                )
 								                            else:
 								                                new_source_id = GRAPH_FIELD_SEP.join(sources)
 								                                relationships_to_update[(src, tgt)] = new_source_id
 								                                logger.debug(
 								                                    f"Relationship {src}-{tgt} will be updated with new source_id: {new_source_id}"
 								                                )
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
 								            # Delete entities
 								            if entities_to_delete:
 								                for entity in entities_to_delete:
 								                    await self.entities_vdb.delete_entity(entity)
 								                    logger.debug(f"Deleted entity {entity} from vector DB")
-												fix delete_by_doc_id

											
										
										
											2025-03-03 19:17:34 +08:00
+								                await self.chunk_entity_relation_graph.remove_nodes(
 								                    list(entities_to_delete)
 								                )
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								                logger.debug(f"Deleted {len(entities_to_delete)} entities from graph")
 								            # Update entities
 								            for entity, new_source_id in entities_to_update.items():
-												Update delete_by_doc_id

											
										
										
											2025-03-04 16:36:58 +08:00
+								                node_data = await self.chunk_entity_relation_graph.get_node(entity)
 								                if node_data:
 								                    node_data["source_id"] = new_source_id
 								                    await self.chunk_entity_relation_graph.upsert_node(
 								                        entity, node_data
 								                    )
 								                    logger.debug(
 								                        f"Updated entity {entity} with new source_id: {new_source_id}"
 								                    )
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
 								            # Delete relationships
 								            if relationships_to_delete:
 								                for src, tgt in relationships_to_delete:
 								                    rel_id_0 = compute_mdhash_id(src + tgt, prefix="rel-")
 								                    rel_id_1 = compute_mdhash_id(tgt + src, prefix="rel-")
 								                    await self.relationships_vdb.delete([rel_id_0, rel_id_1])
 								                    logger.debug(f"Deleted relationship {src}-{tgt} from vector DB")
-												fix delete_by_doc_id

											
										
										
											2025-03-03 19:17:34 +08:00
+								                await self.chunk_entity_relation_graph.remove_edges(
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
+								                    list(relationships_to_delete)
 								                )
 								                logger.debug(
 								                    f"Deleted {len(relationships_to_delete)} relationships from graph"
 								                )
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
 								            # Update relationships
 								            for (src, tgt), new_source_id in relationships_to_update.items():
-												Update delete_by_doc_id

											
										
										
											2025-03-04 16:36:58 +08:00
+								                edge_data = await self.chunk_entity_relation_graph.get_edge(src, tgt)
 								                if edge_data:
 								                    edge_data["source_id"] = new_source_id
 								                    await self.chunk_entity_relation_graph.upsert_edge(
 								                        src, tgt, edge_data
 								                    )
 								                    logger.debug(
 								                        f"Updated relationship {src}-{tgt} with new source_id: {new_source_id}"
 								                    )
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
 								            # 6. Delete original document and status
 								            await self.full_docs.delete([doc_id])
 								            await self.doc_status.delete([doc_id])
 								            # 7. Ensure all indexes are updated
 								            await self._insert_done()
 								            logger.info(
 								                f"Successfully deleted document {doc_id} and related data. "
 								                f"Deleted {len(entities_to_delete)} entities and {len(relationships_to_delete)} relationships. "
 								                f"Updated {len(entities_to_update)} entities and {len(relationships_to_update)} relationships."
 								            )
-												feat: fix delete by document id

											
										
										
											2025-02-27 23:34:57 +07:00
+								            async def process_data(data_type, vdb, chunk_id):
 								                # Check data (entities or relationships)
-												fix delete_by_doc_id

											
										
										
											2025-03-03 19:17:34 +08:00
+								                storage = await vdb.client_storage
-												feat: fix delete by document id

											
										
										
											2025-02-27 23:34:57 +07:00
+								                data_with_chunk = [
 								                    dp
-												fix delete_by_doc_id

											
										
										
											2025-03-03 19:17:34 +08:00
+								                    for dp in storage["data"]
-												feat: fix delete by document id

											
										
										
											2025-02-27 23:34:57 +07:00
+								                    if chunk_id in (dp.get("source_id") or "").split(GRAPH_FIELD_SEP)
 								                ]
 								                data_for_vdb = {}
 								                if data_with_chunk:
 								                    logger.warning(
 								                        f"found {len(data_with_chunk)} {data_type} still referencing chunk {chunk_id}"
 								                    )
 								                    for item in data_with_chunk:
 								                        old_sources = item["source_id"].split(GRAPH_FIELD_SEP)
 								                        new_sources = [src for src in old_sources if src != chunk_id]
 								                        if not new_sources:
 								                            logger.info(
 								                                f"{data_type} {item.get('entity_name', 'N/A')} is deleted because source_id is not exists"
 								                            )
 								                            await vdb.delete_entity(item)
 								                        else:
 								                            item["source_id"] = GRAPH_FIELD_SEP.join(new_sources)
 								                            item_id = item["__id__"]
 								                            data_for_vdb[item_id] = item.copy()
 								                            if data_type == "entities":
 								                                data_for_vdb[item_id]["content"] = data_for_vdb[
 								                                    item_id
 								                                ].get("content") or (
 								                                    item.get("entity_name", "")
 								                                    + (item.get("description") or "")
 								                                )
 								                            else:  # relationships
 								                                data_for_vdb[item_id]["content"] = data_for_vdb[
 								                                    item_id
 								                                ].get("content") or (
 								                                    (item.get("keywords") or "")
 								                                    + (item.get("src_id") or "")
 								                                    + (item.get("tgt_id") or "")
 								                                    + (item.get("description") or "")
 								                                )
 								                    if data_for_vdb:
 								                        await vdb.upsert(data_for_vdb)
 								                        logger.info(f"Successfully updated {data_type} in vector DB")
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								            # Add verification step
 								            async def verify_deletion():
 								                # Verify if the document has been deleted
 								                if await self.full_docs.get_by_id(doc_id):
-												feat: fix delete by document id

											
										
										
											2025-02-27 23:34:57 +07:00
+								                    logger.warning(f"Document {doc_id} still exists in full_docs")
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								                # Verify if chunks have been deleted
-												fix delete_by_doc_id

											
										
										
											2025-03-04 13:22:33 +08:00
+								                all_remaining_chunks = await self.text_chunks.get_all()
 								                remaining_related_chunks = {
 								                    chunk_id: chunk_data
 								                    for chunk_id, chunk_data in all_remaining_chunks.items()
 								                    if isinstance(chunk_data, dict)
 								                    and chunk_data.get("full_doc_id") == doc_id
 								                }
 								                if remaining_related_chunks:
 								                    logger.warning(
 								                        f"Found {len(remaining_related_chunks)} remaining chunks"
 								                    )
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
+								                # Verify entities and relationships
 								                for chunk_id in chunk_ids:
-												feat: fix delete by document id

											
										
										
											2025-02-27 23:34:57 +07:00
+								                    await process_data("entities", self.entities_vdb, chunk_id)
 								                    await process_data(
 								                        "relationships", self.relationships_vdb, chunk_id
 								                    )
-												add delete by doc id

											
										
										
											2024-12-31 17:15:57 +08:00
 								            await verify_deletion()
 								        except Exception as e:
 								            logger.error(f"Error while deleting document {doc_id}: {e}")
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								    async def adelete_by_entity(self, entity_name: str) -> None:
 								        """Asynchronously delete an entity and all its relationships.
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								        Args:
 								            entity_name: Name of the entity to delete
 								        """
 								        from .utils_graph import adelete_by_entity
 								        return await adelete_by_entity(
 								            self.chunk_entity_relation_graph,
 								            self.entities_vdb,
 								            self.relationships_vdb,
 								            entity_name
-												fix linting errors

											
										
										
											2024-12-31 17:32:04 +08:00
+								        )
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								    def delete_by_entity(self, entity_name: str) -> None:
 								        loop = always_get_an_event_loop()
 								        return loop.run_until_complete(self.adelete_by_entity(entity_name))
-												cleanup code

											
										
										
											2025-02-20 13:18:17 +01:00
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								    async def adelete_by_relation(self, source_entity: str, target_entity: str) -> None:
 								        """Asynchronously delete a relation between two entities.
-												Add clear_cache

											
										
										
											2025-03-01 18:30:58 +08:00
 								        Args:
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								            source_entity: Name of the source entity
 								            target_entity: Name of the target entity
-												Add clear_cache

											
										
										
											2025-03-01 18:30:58 +08:00
+								        """
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								        from .utils_graph import adelete_by_relation
 								        return await adelete_by_relation(
 								            self.chunk_entity_relation_graph,
 								            self.relationships_vdb,
 								            source_entity,
 								            target_entity
 								        )
-												Add clear_cache

											
										
										
											2025-03-01 18:30:58 +08:00
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								    def delete_by_relation(self, source_entity: str, target_entity: str) -> None:
 								        loop = always_get_an_event_loop()
 								        return loop.run_until_complete(self.adelete_by_relation(source_entity, target_entity))
-												Add clear_cache

											
										
										
											2025-03-01 18:30:58 +08:00
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								    async def get_processing_status(self) -> dict[str, int]:
 								        """Get current document processing status counts
-												Add clear_cache

											
										
										
											2025-03-01 18:30:58 +08:00
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								        Returns:
 								            Dict with counts for each status
 								        """
 								        return await self.doc_status.get_status_counts()
-												Add clear_cache

											
										
										
											2025-03-01 18:30:58 +08:00
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								    async def get_entity_info(
 								        self, entity_name: str, include_vector_data: bool = False
 								    ) -> dict[str, str | None | dict[str, str]]:
 								        """Get detailed information of an entity"""
 								        from .utils_graph import get_entity_info
 								        return await get_entity_info(
 								            self.chunk_entity_relation_graph,
 								            self.entities_vdb,
 								            entity_name,
 								            include_vector_data
 								        )
-												Add clear_cache

											
										
										
											2025-03-01 18:30:58 +08:00
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								    async def get_relation_info(
 								        self, src_entity: str, tgt_entity: str, include_vector_data: bool = False
 								    ) -> dict[str, str | None | dict[str, str]]:
 								        """Get detailed information of a relationship"""
 								        from .utils_graph import get_relation_info
 								        return await get_relation_info(
 								            self.chunk_entity_relation_graph,
 								            self.relationships_vdb,
 								            src_entity,
 								            tgt_entity,
 								            include_vector_data
 								        )
-												Add a feature that allows modifying nodes and relationships.

											
										
										
											2025-03-03 21:09:45 +08:00
 								    async def aedit_entity(
 								        self, entity_name: str, updated_data: dict[str, str], allow_rename: bool = True
 								    ) -> dict[str, Any]:
 								        """Asynchronously edit entity information.
 								        Updates entity information in the knowledge graph and re-embeds the entity in the vector database.
 								        Args:
 								            entity_name: Name of the entity to edit
 								            updated_data: Dictionary containing updated attributes, e.g. {"description": "new description", "entity_type": "new type"}
 								            allow_rename: Whether to allow entity renaming, defaults to True
 								        Returns:
 								            Dictionary containing updated entity information
 								        """
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								        from .utils_graph import aedit_entity
 								        return await aedit_entity(
 								            self.chunk_entity_relation_graph,
 								            self.entities_vdb,
 								            self.relationships_vdb,
 								            entity_name,
 								            updated_data,
 								            allow_rename
 								        )
-												Add a feature that allows modifying nodes and relationships.

											
										
										
											2025-03-03 21:09:45 +08:00
 								    def edit_entity(
 								        self, entity_name: str, updated_data: dict[str, str], allow_rename: bool = True
 								    ) -> dict[str, Any]:
 								        loop = always_get_an_event_loop()
 								        return loop.run_until_complete(
 								            self.aedit_entity(entity_name, updated_data, allow_rename)
 								        )
 								    async def aedit_relation(
 								        self, source_entity: str, target_entity: str, updated_data: dict[str, Any]
 								    ) -> dict[str, Any]:
 								        """Asynchronously edit relation information.
 								        Updates relation (edge) information in the knowledge graph and re-embeds the relation in the vector database.
 								        Args:
 								            source_entity: Name of the source entity
 								            target_entity: Name of the target entity
 								            updated_data: Dictionary containing updated attributes, e.g. {"description": "new description", "keywords": "new keywords"}
 								        Returns:
 								            Dictionary containing updated relation information
 								        """
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								        from .utils_graph import aedit_relation
 								        return await aedit_relation(
 								            self.chunk_entity_relation_graph,
 								            self.entities_vdb,
 								            self.relationships_vdb,
 								            source_entity,
 								            target_entity,
 								            updated_data
 								        )
-												Add a feature that allows modifying nodes and relationships.

											
										
										
											2025-03-03 21:09:45 +08:00
 								    def edit_relation(
 								        self, source_entity: str, target_entity: str, updated_data: dict[str, Any]
 								    ) -> dict[str, Any]:
 								        loop = always_get_an_event_loop()
 								        return loop.run_until_complete(
 								            self.aedit_relation(source_entity, target_entity, updated_data)
 								        )
 								    async def acreate_entity(
 								        self, entity_name: str, entity_data: dict[str, Any]
 								    ) -> dict[str, Any]:
 								        """Asynchronously create a new entity.
 								        Creates a new entity in the knowledge graph and adds it to the vector database.
 								        Args:
 								            entity_name: Name of the new entity
 								            entity_data: Dictionary containing entity attributes, e.g. {"description": "description", "entity_type": "type"}
 								        Returns:
 								            Dictionary containing created entity information
 								        """
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								        from .utils_graph import acreate_entity
 								        return await acreate_entity(
 								            self.chunk_entity_relation_graph,
 								            self.entities_vdb,
 								            self.relationships_vdb,
 								            entity_name,
 								            entity_data
 								        )
-												Add a feature that allows modifying nodes and relationships.

											
										
										
											2025-03-03 21:09:45 +08:00
 								    def create_entity(
 								        self, entity_name: str, entity_data: dict[str, Any]
 								    ) -> dict[str, Any]:
 								        loop = always_get_an_event_loop()
 								        return loop.run_until_complete(self.acreate_entity(entity_name, entity_data))
 								    async def acreate_relation(
 								        self, source_entity: str, target_entity: str, relation_data: dict[str, Any]
 								    ) -> dict[str, Any]:
 								        """Asynchronously create a new relation between entities.
 								        Creates a new relation (edge) in the knowledge graph and adds it to the vector database.
 								        Args:
 								            source_entity: Name of the source entity
 								            target_entity: Name of the target entity
 								            relation_data: Dictionary containing relation attributes, e.g. {"description": "description", "keywords": "keywords"}
 								        Returns:
 								            Dictionary containing created relation information
 								        """
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								        from .utils_graph import acreate_relation
 								        return await acreate_relation(
 								            self.chunk_entity_relation_graph,
 								            self.entities_vdb,
 								            self.relationships_vdb,
 								            source_entity,
 								            target_entity,
 								            relation_data
 								        )
-												Add a feature that allows modifying nodes and relationships.

											
										
										
											2025-03-03 21:09:45 +08:00
 								    def create_relation(
 								        self, source_entity: str, target_entity: str, relation_data: dict[str, Any]
 								    ) -> dict[str, Any]:
 								        loop = always_get_an_event_loop()
 								        return loop.run_until_complete(
 								            self.acreate_relation(source_entity, target_entity, relation_data)
 								        )
-												Add merge entities

											
										
										
											2025-03-06 00:53:23 +08:00
 								    async def amerge_entities(
 								        self,
 								        source_entities: list[str],
 								        target_entity: str,
 								        merge_strategy: dict[str, str] = None,
 								        target_entity_data: dict[str, Any] = None,
 								    ) -> dict[str, Any]:
 								        """Asynchronously merge multiple entities into one entity.
 								        Merges multiple source entities into a target entity, handling all relationships,
 								        and updating both the knowledge graph and vector database.
 								        Args:
 								            source_entities: List of source entity names to merge
 								            target_entity: Name of the target entity after merging
 								            merge_strategy: Merge strategy configuration, e.g. {"description": "concatenate", "entity_type": "keep_first"}
 								                Supported strategies:
 								                - "concatenate": Concatenate all values (for text fields)
 								                - "keep_first": Keep the first non-empty value
 								                - "keep_last": Keep the last non-empty value
 								                - "join_unique": Join all unique values (for fields separated by delimiter)
 								            target_entity_data: Dictionary of specific values to set for the target entity,
 								                overriding any merged values, e.g. {"description": "custom description", "entity_type": "PERSON"}
 								        Returns:
 								            Dictionary containing the merged entity information
 								        """
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								        from .utils_graph import amerge_entities
 								        return await amerge_entities(
 								            self.chunk_entity_relation_graph,
 								            self.entities_vdb,
 								            self.relationships_vdb,
 								            source_entities,
 								            target_entity,
 								            merge_strategy,
 								            target_entity_data
 								        )
-												Add merge entities

											
										
										
											2025-03-06 00:53:23 +08:00
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								    def merge_entities(
 								        self,
 								        source_entities: list[str],
 								        target_entity: str,
 								        merge_strategy: dict[str, str] = None,
 								        target_entity_data: dict[str, Any] = None,
 								    ) -> dict[str, Any]:
 								        loop = always_get_an_event_loop()
 								        return loop.run_until_complete(
 								            self.amerge_entities(
 								                source_entities, target_entity, merge_strategy, target_entity_data
 								            )
 								        )
-												Add merge entities

											
										
										
											2025-03-06 00:53:23 +08:00
-												Fixed lint and Added new imports at the top of the file

											
										
										
											2025-03-12 00:04:23 +05:30
+								    async def aexport_data(
 								        self,
 								        output_path: str,
 								        file_format: Literal["csv", "excel", "md", "txt"] = "csv",
 								        include_vector_data: bool = False,
 								    ) -> None:
 								        """
 								        Asynchronously exports all entities, relations, and relationships to various formats.
 								        Args:
 								            output_path: The path to the output file (including extension).
 								            file_format: Output format - "csv", "excel", "md", "txt".
 								                - csv: Comma-separated values file
 								                - excel: Microsoft Excel file with multiple sheets
 								                - md: Markdown tables
 								                - txt: Plain text formatted output
 								                - table: Print formatted tables to console
 								            include_vector_data: Whether to include data from the vector database.
 								        """
-												Move graph edit function implemention to a utils_graph.py to educe the size of lightray.py

											
										
										
											2025-04-14 03:06:23 +08:00
+								        from .utils import aexport_data as utils_aexport_data
 								        await utils_aexport_data(
 								            self.chunk_entity_relation_graph,
 								            self.entities_vdb,
 								            self.relationships_vdb,
 								            output_path,
 								            file_format,
 								            include_vector_data
 								        )
-												Fixed lint and Added new imports at the top of the file

											
										
										
											2025-03-12 00:04:23 +05:30
 								    def export_data(
 								        self,
 								        output_path: str,
 								        file_format: Literal["csv", "excel", "md", "txt"] = "csv",
 								        include_vector_data: bool = False,
 								    ) -> None:
 								        """
 								        Synchronously exports all entities, relations, and relationships to various formats.
 								        Args:
 								            output_path: The path to the output file (including extension).
 								            file_format: Output format - "csv", "excel", "md", "txt".
 								                - csv: Comma-separated values file
 								                - excel: Microsoft Excel file with multiple sheets
 								                - md: Markdown tables
 								                - txt: Plain text formatted output
 								                - table: Print formatted tables to console
 								            include_vector_data: Whether to include data from the vector database.
 								        """
 								        try:
 								            loop = asyncio.get_event_loop()
 								        except RuntimeError:
 								            loop = asyncio.new_event_loop()
 								            asyncio.set_event_loop(loop)
 								        loop.run_until_complete(
 								            self.aexport_data(output_path, file_format, include_vector_data)
 								        )