2025-01-21 20:52:28 +08:00
|
|
|
#
|
|
|
|
|
# Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
|
|
|
|
|
#
|
|
|
|
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
|
|
|
# you may not use this file except in compliance with the License.
|
|
|
|
|
# You may obtain a copy of the License at
|
|
|
|
|
#
|
|
|
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
|
|
|
#
|
|
|
|
|
# Unless required by applicable law or agreed to in writing, software
|
|
|
|
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
|
|
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
|
|
|
# See the License for the specific language governing permissions and
|
|
|
|
|
# limitations under the License.
|
|
|
|
|
#
|
|
|
|
|
|
2024-11-14 17:13:48 +08:00
|
|
|
import logging
|
2024-09-09 09:41:14 +08:00
|
|
|
import os
|
|
|
|
|
import time
|
2025-11-02 12:24:08 +08:00
|
|
|
from common.decorator import singleton
|
2024-09-09 09:41:14 +08:00
|
|
|
from azure.identity import ClientSecretCredential, AzureAuthorityHosts
|
|
|
|
|
from azure.storage.filedatalake import FileSystemClient
|
2025-11-06 09:36:38 +08:00
|
|
|
from common import settings
|
2024-09-09 09:41:14 +08:00
|
|
|
|
2026-04-03 12:51:26 +08:00
|
|
|
_CLOUD_AUTHORITY_MAP = {
|
|
|
|
|
"public": AzureAuthorityHosts.AZURE_PUBLIC_CLOUD,
|
|
|
|
|
"china": AzureAuthorityHosts.AZURE_CHINA,
|
|
|
|
|
"government": AzureAuthorityHosts.AZURE_GOVERNMENT,
|
|
|
|
|
"germany": AzureAuthorityHosts.AZURE_GERMANY,
|
|
|
|
|
}
|
|
|
|
|
|
2024-09-09 09:41:14 +08:00
|
|
|
|
|
|
|
|
@singleton
|
2025-03-05 18:03:53 +08:00
|
|
|
class RAGFlowAzureSpnBlob:
|
2024-09-09 09:41:14 +08:00
|
|
|
def __init__(self):
|
|
|
|
|
self.conn = None
|
2025-11-06 09:36:38 +08:00
|
|
|
self.account_url = os.getenv('ACCOUNT_URL', settings.AZURE["account_url"])
|
|
|
|
|
self.client_id = os.getenv('CLIENT_ID', settings.AZURE["client_id"])
|
|
|
|
|
self.secret = os.getenv('SECRET', settings.AZURE["secret"])
|
|
|
|
|
self.tenant_id = os.getenv('TENANT_ID', settings.AZURE["tenant_id"])
|
|
|
|
|
self.container_name = os.getenv('CONTAINER_NAME', settings.AZURE["container_name"])
|
2026-04-03 12:51:26 +08:00
|
|
|
self.cloud = os.getenv('AZURE_CLOUD', settings.AZURE.get("cloud", "public")).lower()
|
2024-09-09 09:41:14 +08:00
|
|
|
self.__open__()
|
|
|
|
|
|
|
|
|
|
def __open__(self):
|
|
|
|
|
try:
|
|
|
|
|
if self.conn:
|
|
|
|
|
self.__close__()
|
2024-11-12 17:35:13 +08:00
|
|
|
except Exception:
|
2024-09-09 09:41:14 +08:00
|
|
|
pass
|
|
|
|
|
|
|
|
|
|
try:
|
2026-04-03 12:51:26 +08:00
|
|
|
authority = _CLOUD_AUTHORITY_MAP.get(self.cloud, AzureAuthorityHosts.AZURE_PUBLIC_CLOUD)
|
2025-12-29 12:01:18 +08:00
|
|
|
credentials = ClientSecretCredential(tenant_id=self.tenant_id, client_id=self.client_id,
|
2026-04-03 12:51:26 +08:00
|
|
|
client_secret=self.secret, authority=authority)
|
2025-12-29 12:01:18 +08:00
|
|
|
self.conn = FileSystemClient(account_url=self.account_url, file_system_name=self.container_name,
|
|
|
|
|
credential=credentials)
|
2024-11-12 17:35:13 +08:00
|
|
|
except Exception:
|
2024-11-14 17:13:48 +08:00
|
|
|
logging.exception("Fail to connect %s" % self.account_url)
|
2024-09-09 09:41:14 +08:00
|
|
|
|
|
|
|
|
def __close__(self):
|
|
|
|
|
del self.conn
|
|
|
|
|
self.conn = None
|
|
|
|
|
|
|
|
|
|
def health(self):
|
2024-12-08 14:21:12 +08:00
|
|
|
_bucket, fnm, binary = "txtxtxtxt1", "txtxtxtxt1", b"_t@@@1"
|
2024-09-09 09:41:14 +08:00
|
|
|
f = self.conn.create_file(fnm)
|
|
|
|
|
f.append_data(binary, offset=0, length=len(binary))
|
|
|
|
|
return f.flush_data(len(binary))
|
|
|
|
|
|
2026-04-23 20:40:54 +08:00
|
|
|
def put(self, bucket, fnm, binary, tenant_id=None):
|
2024-09-09 09:41:14 +08:00
|
|
|
for _ in range(3):
|
|
|
|
|
try:
|
fix: prepend bucket prefix in Azure Blob (SAS/SPN) to prevent cross-dataset file overwrites (#14174)
Fixes #14159
## Problem
The `put()`, `get()`, `rm()`, and `obj_exist()` methods in both
`azure_spn_conn.py` and `azure_sas_conn.py` ignore the `bucket`
parameter entirely, storing all files flat using only the filename. This
causes files from different datasets to overwrite each other when they
share the same filename.
By contrast, the MinIO and S3 implementations correctly use the bucket
(typically the knowledge base ID) as a path prefix, creating logical
folder isolation like `{kb_id}/{filename}`.
## Solution
Prepend the `bucket` parameter as a path prefix to all file operations
in both Azure storage implementations:
- `azure_spn_conn.py`: `create_file`, `delete_file`, `get_file_client`
now use `f"{bucket}/{fnm}"`
- `azure_sas_conn.py`: `upload_blob`, `delete_blob`, `download_blob`,
`get_blob_client` now use `f"{bucket}/{fnm}"`
This matches the behavior of all other storage backends (MinIO, S3) and
prevents filename collisions across knowledge bases.
## Testing
- Verified the fix aligns with how MinIO/S3 connectors handle the bucket
parameter
- The `health()` method is left unchanged as it uses a fixed test path
for connectivity checks only
Co-authored-by: octo-patch <octo-patch@github.com>
Co-authored-by: Jin Hai <haijin.chn@gmail.com>
2026-05-07 17:13:43 +08:00
|
|
|
f = self.conn.create_file(f"{bucket}/{fnm}")
|
2024-09-09 09:41:14 +08:00
|
|
|
f.append_data(binary, offset=0, length=len(binary))
|
|
|
|
|
return f.flush_data(len(binary))
|
2024-11-12 17:35:13 +08:00
|
|
|
except Exception:
|
2024-11-14 17:13:48 +08:00
|
|
|
logging.exception(f"Fail put {bucket}/{fnm}")
|
2024-09-09 09:41:14 +08:00
|
|
|
self.__open__()
|
|
|
|
|
time.sleep(1)
|
2025-11-12 19:00:15 +08:00
|
|
|
return None
|
|
|
|
|
return None
|
2024-09-09 09:41:14 +08:00
|
|
|
|
|
|
|
|
def rm(self, bucket, fnm):
|
|
|
|
|
try:
|
fix: prepend bucket prefix in Azure Blob (SAS/SPN) to prevent cross-dataset file overwrites (#14174)
Fixes #14159
## Problem
The `put()`, `get()`, `rm()`, and `obj_exist()` methods in both
`azure_spn_conn.py` and `azure_sas_conn.py` ignore the `bucket`
parameter entirely, storing all files flat using only the filename. This
causes files from different datasets to overwrite each other when they
share the same filename.
By contrast, the MinIO and S3 implementations correctly use the bucket
(typically the knowledge base ID) as a path prefix, creating logical
folder isolation like `{kb_id}/{filename}`.
## Solution
Prepend the `bucket` parameter as a path prefix to all file operations
in both Azure storage implementations:
- `azure_spn_conn.py`: `create_file`, `delete_file`, `get_file_client`
now use `f"{bucket}/{fnm}"`
- `azure_sas_conn.py`: `upload_blob`, `delete_blob`, `download_blob`,
`get_blob_client` now use `f"{bucket}/{fnm}"`
This matches the behavior of all other storage backends (MinIO, S3) and
prevents filename collisions across knowledge bases.
## Testing
- Verified the fix aligns with how MinIO/S3 connectors handle the bucket
parameter
- The `health()` method is left unchanged as it uses a fixed test path
for connectivity checks only
Co-authored-by: octo-patch <octo-patch@github.com>
Co-authored-by: Jin Hai <haijin.chn@gmail.com>
2026-05-07 17:13:43 +08:00
|
|
|
self.conn.delete_file(f"{bucket}/{fnm}")
|
2024-11-12 17:35:13 +08:00
|
|
|
except Exception:
|
2024-11-14 17:13:48 +08:00
|
|
|
logging.exception(f"Fail rm {bucket}/{fnm}")
|
2024-09-09 09:41:14 +08:00
|
|
|
|
|
|
|
|
def get(self, bucket, fnm):
|
|
|
|
|
for _ in range(1):
|
|
|
|
|
try:
|
fix: prepend bucket prefix in Azure Blob (SAS/SPN) to prevent cross-dataset file overwrites (#14174)
Fixes #14159
## Problem
The `put()`, `get()`, `rm()`, and `obj_exist()` methods in both
`azure_spn_conn.py` and `azure_sas_conn.py` ignore the `bucket`
parameter entirely, storing all files flat using only the filename. This
causes files from different datasets to overwrite each other when they
share the same filename.
By contrast, the MinIO and S3 implementations correctly use the bucket
(typically the knowledge base ID) as a path prefix, creating logical
folder isolation like `{kb_id}/{filename}`.
## Solution
Prepend the `bucket` parameter as a path prefix to all file operations
in both Azure storage implementations:
- `azure_spn_conn.py`: `create_file`, `delete_file`, `get_file_client`
now use `f"{bucket}/{fnm}"`
- `azure_sas_conn.py`: `upload_blob`, `delete_blob`, `download_blob`,
`get_blob_client` now use `f"{bucket}/{fnm}"`
This matches the behavior of all other storage backends (MinIO, S3) and
prevents filename collisions across knowledge bases.
## Testing
- Verified the fix aligns with how MinIO/S3 connectors handle the bucket
parameter
- The `health()` method is left unchanged as it uses a fixed test path
for connectivity checks only
Co-authored-by: octo-patch <octo-patch@github.com>
Co-authored-by: Jin Hai <haijin.chn@gmail.com>
2026-05-07 17:13:43 +08:00
|
|
|
client = self.conn.get_file_client(f"{bucket}/{fnm}")
|
2024-09-09 09:41:14 +08:00
|
|
|
r = client.download_file()
|
|
|
|
|
return r.read()
|
2024-11-12 17:35:13 +08:00
|
|
|
except Exception:
|
2024-11-14 17:13:48 +08:00
|
|
|
logging.exception(f"fail get {bucket}/{fnm}")
|
2024-09-09 09:41:14 +08:00
|
|
|
self.__open__()
|
|
|
|
|
time.sleep(1)
|
2025-11-12 19:00:15 +08:00
|
|
|
return None
|
2024-09-09 09:41:14 +08:00
|
|
|
|
|
|
|
|
def obj_exist(self, bucket, fnm):
|
|
|
|
|
try:
|
fix: prepend bucket prefix in Azure Blob (SAS/SPN) to prevent cross-dataset file overwrites (#14174)
Fixes #14159
## Problem
The `put()`, `get()`, `rm()`, and `obj_exist()` methods in both
`azure_spn_conn.py` and `azure_sas_conn.py` ignore the `bucket`
parameter entirely, storing all files flat using only the filename. This
causes files from different datasets to overwrite each other when they
share the same filename.
By contrast, the MinIO and S3 implementations correctly use the bucket
(typically the knowledge base ID) as a path prefix, creating logical
folder isolation like `{kb_id}/{filename}`.
## Solution
Prepend the `bucket` parameter as a path prefix to all file operations
in both Azure storage implementations:
- `azure_spn_conn.py`: `create_file`, `delete_file`, `get_file_client`
now use `f"{bucket}/{fnm}"`
- `azure_sas_conn.py`: `upload_blob`, `delete_blob`, `download_blob`,
`get_blob_client` now use `f"{bucket}/{fnm}"`
This matches the behavior of all other storage backends (MinIO, S3) and
prevents filename collisions across knowledge bases.
## Testing
- Verified the fix aligns with how MinIO/S3 connectors handle the bucket
parameter
- The `health()` method is left unchanged as it uses a fixed test path
for connectivity checks only
Co-authored-by: octo-patch <octo-patch@github.com>
Co-authored-by: Jin Hai <haijin.chn@gmail.com>
2026-05-07 17:13:43 +08:00
|
|
|
client = self.conn.get_file_client(f"{bucket}/{fnm}")
|
2024-09-09 09:41:14 +08:00
|
|
|
return client.exists()
|
2024-11-12 17:35:13 +08:00
|
|
|
except Exception:
|
2024-11-14 17:13:48 +08:00
|
|
|
logging.exception(f"Fail put {bucket}/{fnm}")
|
2024-09-09 09:41:14 +08:00
|
|
|
return False
|
|
|
|
|
|
|
|
|
|
def get_presigned_url(self, bucket, fnm, expires):
|
|
|
|
|
for _ in range(10):
|
|
|
|
|
try:
|
|
|
|
|
return self.conn.get_presigned_url("GET", bucket, fnm, expires)
|
2024-11-12 17:35:13 +08:00
|
|
|
except Exception:
|
2024-11-14 17:13:48 +08:00
|
|
|
logging.exception(f"fail get {bucket}/{fnm}")
|
2024-09-09 09:41:14 +08:00
|
|
|
self.__open__()
|
|
|
|
|
time.sleep(1)
|
2025-12-29 12:01:18 +08:00
|
|
|
return None
|