Feat: add splitter (#10161)

### What problem does this PR solve? ### Type of change - [x] New Feature (non-breaking change which adds functionality) --------- Signed-off-by: dependabot[bot] <support@github.com> Co-authored-by: Lynn <lynn_inf@hotmail.com> Co-authored-by: chanx <1243304602@qq.com> Co-authored-by: balibabu <cike8899@users.noreply.github.com> Co-authored-by: 纷繁下的无奈 <zhileihuang@126.com> Co-authored-by: huangzl <huangzl@shinemo.com> Co-authored-by: writinwaters <93570324+writinwaters@users.noreply.github.com> Co-authored-by: Wilmer <33392318@qq.com> Co-authored-by: Adrian Weidig <adrianweidig@gmx.net> Co-authored-by: Zhichang Yu <yuzhichang@gmail.com> Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com> Co-authored-by: Yongteng Lei <yongtengrey@outlook.com> Co-authored-by: Liu An <asiro@qq.com> Co-authored-by: buua436 <66937541+buua436@users.noreply.github.com> Co-authored-by: BadwomanCraZY <511528396@qq.com> Co-authored-by: cucusenok <31804608+cucusenok@users.noreply.github.com> Co-authored-by: Russell Valentine <russ@coldstonelabs.org> Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> Co-authored-by: Billy Bao <newyorkupperbay@gmail.com> Co-authored-by: Zhedong Cen <cenzhedong2@126.com> Co-authored-by: TensorNull <129579691+TensorNull@users.noreply.github.com> Co-authored-by: TensorNull <tensor.null@gmail.com>
2025-12-08 20:42:30 +08:00 · 2025-09-19 10:15:19 +08:00
parent f9c7404bee
commit a1b947ffd6
81 changed files with 3083 additions and 799 deletions
--- a/.github/workflows/release.yml
+++ b/.github/workflows/release.yml
@ -88,7 +88,9 @@ jobs:
        with:
          context: .
          push: true
-          tags: infiniflow/ragflow:${{ env.RELEASE_TAG }}
+          tags: |
+            infiniflow/ragflow:${{ env.RELEASE_TAG }}
+            infiniflow/ragflow:latest-full
          file: Dockerfile
          platforms: linux/amd64

@ -98,7 +100,9 @@ jobs:
        with:
          context: .
          push: true
-          tags: infiniflow/ragflow:${{ env.RELEASE_TAG }}-slim
+          tags: |
+            infiniflow/ragflow:${{ env.RELEASE_TAG }}-slim
+            infiniflow/ragflow:latest-slim
          file: Dockerfile
          build-args: LIGHTEN=1
          platforms: linux/amd64
--- a/agent/templates/sql_assistant.json
+++ b/agent/templates/sql_assistant.json
@ -83,7 +83,7 @@
                            },
                            "password": "20010812Yy!",
                            "port": 3306,
-                            "sql": "Agent:WickedGoatsDivide@content",
+                            "sql": "{Agent:WickedGoatsDivide@content}",
                            "username": "13637682833@163.com"
                        }
                    },
@ -114,9 +114,7 @@
                        "params": {
                            "cross_languages": [],
                            "empty_response": "",
-                            "kb_ids": [
-                                "ed31364c727211f0bdb2bafe6e7908e6"
-                            ],
+                            "kb_ids": [],
                            "keywords_similarity_weight": 0.7,
                            "outputs": {
                                "formalized_content": {
@ -124,7 +122,7 @@
                                    "value": ""
                                }
                            },
-                            "query": "sys.query",
+                            "query": "{sys.query}",
                            "rerank_id": "",
                            "similarity_threshold": 0.2,
                            "top_k": 1024,
@ -145,9 +143,7 @@
                        "params": {
                            "cross_languages": [],
                            "empty_response": "",
-                            "kb_ids": [
-                                "0f968106727311f08357bafe6e7908e6"
-                            ],
+                            "kb_ids": [],
                            "keywords_similarity_weight": 0.7,
                            "outputs": {
                                "formalized_content": {
@ -155,7 +151,7 @@
                                    "value": ""
                                }
                            },
-                            "query": "sys.query",
+                            "query": "{sys.query}",
                            "rerank_id": "",
                            "similarity_threshold": 0.2,
                            "top_k": 1024,
@ -176,9 +172,7 @@
                        "params": {
                            "cross_languages": [],
                            "empty_response": "",
-                            "kb_ids": [
-                                "4ad1f9d0727311f0827dbafe6e7908e6"
-                            ],
+                            "kb_ids": [],
                            "keywords_similarity_weight": 0.7,
                            "outputs": {
                                "formalized_content": {
@ -186,7 +180,7 @@
                                    "value": ""
                                }
                            },
-                            "query": "sys.query",
+                            "query": "{sys.query}",
                            "rerank_id": "",
                            "similarity_threshold": 0.2,
                            "top_k": 1024,
@ -347,9 +341,7 @@
                            "form": {
                                "cross_languages": [],
                                "empty_response": "",
-                                "kb_ids": [
-                                    "ed31364c727211f0bdb2bafe6e7908e6"
-                                ],
+                                "kb_ids": [],
                                "keywords_similarity_weight": 0.7,
                                "outputs": {
                                    "formalized_content": {
@ -357,7 +349,7 @@
                                        "value": ""
                                    }
                                },
-                                "query": "sys.query",
+                                "query": "{sys.query}",
                                "rerank_id": "",
                                "similarity_threshold": 0.2,
                                "top_k": 1024,
@ -387,9 +379,7 @@
                            "form": {
                                "cross_languages": [],
                                "empty_response": "",
-                                "kb_ids": [
-                                    "0f968106727311f08357bafe6e7908e6"
-                                ],
+                                "kb_ids": [],
                                "keywords_similarity_weight": 0.7,
                                "outputs": {
                                    "formalized_content": {
@ -397,7 +387,7 @@
                                        "value": ""
                                    }
                                },
-                                "query": "sys.query",
+                                "query": "{sys.query}",
                                "rerank_id": "",
                                "similarity_threshold": 0.2,
                                "top_k": 1024,
@ -427,9 +417,7 @@
                            "form": {
                                "cross_languages": [],
                                "empty_response": "",
-                                "kb_ids": [
-                                    "4ad1f9d0727311f0827dbafe6e7908e6"
-                                ],
+                                "kb_ids": [],
                                "keywords_similarity_weight": 0.7,
                                "outputs": {
                                    "formalized_content": {
@ -437,7 +425,7 @@
                                        "value": ""
                                    }
                                },
-                                "query": "sys.query",
+                                "query": "{sys.query}",
                                "rerank_id": "",
                                "similarity_threshold": 0.2,
                                "top_k": 1024,
@ -539,7 +527,7 @@
                                },
                                "password": "20010812Yy!",
                                "port": 3306,
-                                "sql": "Agent:WickedGoatsDivide@content",
+                                "sql": "{Agent:WickedGoatsDivide@content}",
                                "username": "13637682833@163.com"
                            },
                            "label": "ExeSQL",
--- a/agent/tools/code_exec.py
+++ b/agent/tools/code_exec.py
@ -157,7 +157,7 @@ class CodeExec(ToolBase, ABC):

        try:
            resp = requests.post(url=f"http://{settings.SANDBOX_HOST}:9385/run", json=code_req, timeout=os.environ.get("COMPONENT_EXEC_TIMEOUT", 10*60))
-            logging.info(f"http://{settings.SANDBOX_HOST}:9385/run", code_req, resp.status_code)
+            logging.info(f"http://{settings.SANDBOX_HOST}:9385/run,  code_req: {code_req}, resp.status_code {resp.status_code}:")
            if resp.status_code != 200:
                resp.raise_for_status()
            body = resp.json()
--- a/agent/tools/exesql.py
+++ b/agent/tools/exesql.py
@ -53,7 +53,7 @@ class ExeSQLParam(ToolParamBase):
        self.max_records = 1024

    def check(self):
-        self.check_valid_value(self.db_type, "Choose DB type", ['mysql', 'postgresql', 'mariadb', 'mssql'])
+        self.check_valid_value(self.db_type, "Choose DB type", ['mysql', 'postgres', 'mariadb', 'mssql'])
        self.check_empty(self.database, "Database name")
        self.check_empty(self.username, "database username")
        self.check_empty(self.host, "IP Address")
@ -111,7 +111,7 @@ class ExeSQL(ToolBase, ABC):
        if self._param.db_type in ["mysql", "mariadb"]:
            db = pymysql.connect(db=self._param.database, user=self._param.username, host=self._param.host,
                                 port=self._param.port, password=self._param.password)
-        elif self._param.db_type == 'postgresql':
+        elif self._param.db_type == 'postgres':
            db = psycopg2.connect(dbname=self._param.database, user=self._param.username, host=self._param.host,
                                  port=self._param.port, password=self._param.password)
        elif self._param.db_type == 'mssql':
--- a/api/apps/HEALTHCHECK_TESTING.md
+++ b/api/apps/HEALTHCHECK_TESTING.md
@ -0,0 +1,105 @@
+# 健康检查与 Kubernetes 探针简明说明
+
+本文件说明：什么是 K8s 探针、如何用 `/v1/system/healthz` 做健康检查，以及下文用例中的关键词含义。
+
+## 什么是 K8s 探针（Probe）
+- 探针是 K8s 用来“探测”容器是否健康/可对外服务的机制。
+- 常见三类：
+  - livenessProbe：活性探针。失败时 K8s 会重启容器，用于“应用卡死/失去连接时自愈”。
+  - readinessProbe：就绪探针。失败时 Endpoint 不会被加入 Service 负载均衡，用于“应用尚未准备好时不接流量”。
+  - startupProbe：启动探针。给慢启动应用更长的初始化窗口，期间不执行 liveness/readiness。
+- 这些探针通常通过 HTTP GET 访问一个公开且轻量的健康端点（无需鉴权），以 HTTP 状态码判定结果：200=通过；5xx/超时=失败。
+
+## 本项目健康端点
+- 已实现：`GET /v1/system/healthz`（无需认证）。
+- 语义：
+  - 200：关键依赖正常。
+  - 500：任一关键依赖异常（当前判定为 DB 或 Chat）。
+  - 响应体：JSON，最小字段 `status, db, chat`；并包含 `redis, doc_engine, storage` 等可观测项。失败项会在 `_meta` 中包含 `error/elapsed`。
+- 示例（DB 故障）：
+```json
+{"status":"nok","chat":"ok","db":"nok"}
+```
+
+## 用例背景（Problem/use case）
+- 现状：Ragflow 跑在 K8s，数据库是 AWS RDS Postgres，凭证由 Secret Manager 管理并每 7 天轮换。轮换后应用连接失效，需要手动重启 Pod 才能重新建立连接。
+- 目标：通过 K8s 探针自动化检测并重启异常 Pod，减少人工操作。
+- 需求：一个“无需鉴权”的公共健康端点，能在依赖异常时返回非 200（如 500）且提供 JSON 详情。
+- 现已满足：`/v1/system/healthz` 正是为此设计。
+
+## 关键术语解释（对应你提供的描述）
+- Ragflow instance：部署在 K8s 的 Ragflow 服务。
+- AWS RDS Postgres：托管的 PostgreSQL 数据库实例。
+- Secret Manager rotation：Secrets 定期轮换（每 7 天），会导致旧连接失效。
+- Probes（K8s 探针）：liveness/readiness，用于自动重启或摘除不健康实例。
+- Public endpoint without API key：无需 Authorization 的 HTTP 路由，便于探针直接访问。
+- Dependencies statuses：依赖健康状态（db、chat、redis、doc_engine、storage 等）。
+- HTTP 500 with JSON：当依赖异常时返回 500，并附带 JSON 说明哪个子系统失败。
+
+## 快速测试
+- 正常：
+```bash
+curl -i http://<host>/v1/system/healthz
+```
+- 制造 DB 故障（docker-compose 示例）：
+```bash
+docker compose stop db && curl -i http://<host>/v1/system/healthz
+```
+（预期 500，JSON 中 `db:"nok"`）
+
+## 更完整的测试清单
+### 1) 仅查看 HTTP 状态码
+```bash
+curl -s -o /dev/null -w "%{http_code}\n" http://<host>/v1/system/healthz
+```
+期望：`200` 或 `500`。
+
+### 2) Windows PowerShell
+```powershell
+# 状态码
+(Invoke-WebRequest -Uri "http://<host>/v1/system/healthz" -Method GET -TimeoutSec 3 -ErrorAction SilentlyContinue).StatusCode
+# 完整响应
+Invoke-RestMethod -Uri "http://<host>/v1/system/healthz" -Method GET
+```
+
+### 3) 通过 kubectl 端口转发本地测试
+```bash
+# 前端/网关暴露端口不同环境自行调整
+kubectl port-forward deploy/<your-deploy> 8080:80 -n <ns>
+curl -i http://127.0.0.1:8080/v1/system/healthz
+```
+
+### 4) 制造常见失败场景
+- DB 失败（推荐）：
+```bash
+docker compose stop db
+curl -i http://<host>/v1/system/healthz   # 预期 500
+```
+- Chat 失败（可选）：将 `CHAT_CFG` 的 `factory`/`base_url` 设为无效并重启后端，再请求应为 500，且 `chat:"nok"`。
+- Redis/存储/文档引擎：停用对应服务后再次请求，可在 JSON 中看到相应字段为 `"nok"`（不影响 200/500 判定）。
+
+### 5) 浏览器验证
+- 直接打开 `http://<host>/v1/system/healthz`，在 DevTools Network 查看 200/500；页面正文就是 JSON。
+- 反向代理注意：若有自定义 500 错页，需对 `/healthz` 关闭错误页拦截（如 `proxy_intercept_errors off;`）。
+
+## K8s 探针示例
+```yaml
+readinessProbe:
+  httpGet:
+    path: /v1/system/healthz
+    port: 80
+  initialDelaySeconds: 5
+  periodSeconds: 10
+  timeoutSeconds: 2
+  failureThreshold: 1
+livenessProbe:
+  httpGet:
+    path: /v1/system/healthz
+    port: 80
+  initialDelaySeconds: 10
+  periodSeconds: 10
+  timeoutSeconds: 2
+  failureThreshold: 3
+```
+
+提示：如有反向代理（Nginx）自定义 500 错页，需对 `/healthz` 关闭错误页拦截，以便保留 JSON。
--- a/api/apps/canvas_app.py
+++ b/api/apps/canvas_app.py
@ -28,6 +28,7 @@ from api.db import CanvasCategory, FileType
 from api.db.services.canvas_service import CanvasTemplateService, UserCanvasService, API4ConversationService
 from api.db.services.document_service import DocumentService
 from api.db.services.file_service import FileService
+from api.db.services.task_service import queue_dataflow
 from api.db.services.user_service import TenantService
 from api.db.services.user_canvas_version import UserCanvasVersionService
 from api.settings import RetCode
@ -48,14 +49,6 @@ def templates():
    return get_json_result(data=[c.to_dict() for c in CanvasTemplateService.query(canvas_category=CanvasCategory.Agent)])


-@manager.route('/list', methods=['GET'])  # noqa: F821
-@login_required
-def canvas_list():
-    return get_json_result(data=sorted([c.to_dict() for c in \
-                                 UserCanvasService.query(user_id=current_user.id, canvas_category=CanvasCategory.Agent)], key=lambda x: x["update_time"]*-1)
-                           )
-
-
@manager.route('/rm', methods=['POST'])  # noqa: F821
@validate_request("canvas_ids")
@login_required
@ -77,9 +70,10 @@ def save():
    if not isinstance(req["dsl"], str):
        req["dsl"] = json.dumps(req["dsl"], ensure_ascii=False)
    req["dsl"] = json.loads(req["dsl"])
+    cate = req.get("canvas_category", CanvasCategory.Agent)
    if "id" not in req:
        req["user_id"] = current_user.id
-        if UserCanvasService.query(user_id=current_user.id, title=req["title"].strip(), canvas_category=CanvasCategory.Agent):
+        if UserCanvasService.query(user_id=current_user.id, title=req["title"].strip(), canvas_category=cate):
            return get_data_error_result(message=f"{req['title'].strip()} already exists.")
        req["id"] = get_uuid()
        if not UserCanvasService.save(**req):
@ -148,6 +142,14 @@ def run():
    if not isinstance(cvs.dsl, str):
        cvs.dsl = json.dumps(cvs.dsl, ensure_ascii=False)

+    if cvs.canvas_category == CanvasCategory.DataFlow:
+        task_id = get_uuid()
+        flow_id = get_uuid()
+        ok, error_message = queue_dataflow(dsl=cvs.dsl, tenant_id=user_id, file=files[0], task_id=task_id, flow_id=flow_id, priority=0)
+        if not ok:
+            return server_error_response(error_message)
+        return get_json_result(data={"task_id": task_id, "message_id": flow_id})
+
    try:
        canvas = Canvas(cvs.dsl, current_user.id, req["id"])
    except Exception as e:
@ -332,7 +334,7 @@ def test_db_connect():
        if req["db_type"] in ["mysql", "mariadb"]:
            db = MySQLDatabase(req["database"], user=req["username"], host=req["host"], port=req["port"],
                               password=req["password"])
-        elif req["db_type"] == 'postgresql':
+        elif req["db_type"] == 'postgres':
            db = PostgresqlDatabase(req["database"], user=req["username"], host=req["host"], port=req["port"],
                                    password=req["password"])
        elif req["db_type"] == 'mssql':
@ -383,22 +385,31 @@ def getversion( version_id):
        return get_json_result(data=f"Error getting history file: {e}")


-@manager.route('/listteam', methods=['GET'])  # noqa: F821
+@manager.route('/list', methods=['GET'])  # noqa: F821
@login_required
 def list_canvas():
    keywords = request.args.get("keywords", "")
    page_number = int(request.args.get("page", 1))
    items_per_page = int(request.args.get("page_size", 150))
    orderby = request.args.get("orderby", "create_time")
-    desc = request.args.get("desc", True)
-    try:
+    canvas_category = request.args.get("canvas_category")
+    if request.args.get("desc", "true").lower() == "false":
+        desc = False
+    else:
+        desc = True
+    owner_ids = request.args.get("owner_ids", [])
+    if not owner_ids:
        tenants = TenantService.get_joined_tenants_by_user_id(current_user.id)
+        tenants = [m["tenant_id"] for m in tenants]
        canvas, total = UserCanvasService.get_by_tenant_ids(
-            [m["tenant_id"] for m in tenants], current_user.id, page_number,
-            items_per_page, orderby, desc, keywords, canvas_category=CanvasCategory.Agent)
+            tenants, current_user.id, page_number,
+            items_per_page, orderby, desc, keywords, canvas_category)
+    else:
+        tenants = owner_ids
+        canvas, total = UserCanvasService.get_by_tenant_ids(
+            tenants, current_user.id, 0,
+            0, orderby, desc, keywords, canvas_category)
    return get_json_result(data={"canvas": canvas, "total": total})
-    except Exception as e:
-        return server_error_response(e)


@manager.route('/setting', methods=['POST'])  # noqa: F821
--- a/api/apps/document_app.py
+++ b/api/apps/document_app.py
@ -182,6 +182,7 @@ def create():
                "id": get_uuid(),
                "kb_id": kb.id,
                "parser_id": kb.parser_id,
+                "pipeline_id": kb.pipeline_id,
                "parser_config": kb.parser_config,
                "created_by": current_user.id,
                "type": FileType.VIRTUAL,
@ -546,31 +547,22 @@ def get(doc_id):

@manager.route("/change_parser", methods=["POST"])  # noqa: F821
@login_required
-@validate_request("doc_id", "parser_id")
+@validate_request("doc_id")
 def change_parser():
    req = request.json

    if not DocumentService.accessible(req["doc_id"], current_user.id):
        return get_json_result(data=False, message="No authorization.", code=settings.RetCode.AUTHENTICATION_ERROR)
-    try:
+
    e, doc = DocumentService.get_by_id(req["doc_id"])
    if not e:
        return get_data_error_result(message="Document not found!")
-        if doc.parser_id.lower() == req["parser_id"].lower():
-            if "parser_config" in req:
-                if req["parser_config"] == doc.parser_config:
-                    return get_json_result(data=True)
-            else:
-                return get_json_result(data=True)
-
-        if (doc.type == FileType.VISUAL and req["parser_id"] != "picture") or (re.search(r"\.(ppt|pptx|pages)$", doc.name) and req["parser_id"] != "presentation"):
-            return get_data_error_result(message="Not supported yet!")

+    def reset_doc():
+        nonlocal doc
        e = DocumentService.update_by_id(doc.id, {"parser_id": req["parser_id"], "progress": 0, "progress_msg": "", "run": TaskStatus.UNSTART.value})
        if not e:
            return get_data_error_result(message="Document not found!")
-        if "parser_config" in req:
-            DocumentService.update_parser_config(doc.id, req["parser_config"])
        if doc.token_num > 0:
            e = DocumentService.increment_chunk_num(doc.id, doc.kb_id, doc.token_num * -1, doc.chunk_num * -1, doc.process_duration * -1)
            if not e:
@ -581,6 +573,26 @@ def change_parser():
            if settings.docStoreConn.indexExist(search.index_name(tenant_id), doc.kb_id):
                settings.docStoreConn.delete({"doc_id": doc.id}, search.index_name(tenant_id), doc.kb_id)

+    try:
+        if req.get("pipeline_id"):
+            if doc.pipeline_id == req["pipeline_id"]:
+                return get_json_result(data=True)
+            DocumentService.update_by_id(doc.id, {"pipeline_id": req["pipeline_id"]})
+            reset_doc()
+            return get_json_result(data=True)
+
+        if doc.parser_id.lower() == req["parser_id"].lower():
+            if "parser_config" in req:
+                if req["parser_config"] == doc.parser_config:
+                    return get_json_result(data=True)
+            else:
+                return get_json_result(data=True)
+
+        if (doc.type == FileType.VISUAL and req["parser_id"] != "picture") or (re.search(r"\.(ppt|pptx|pages)$", doc.name) and req["parser_id"] != "presentation"):
+            return get_data_error_result(message="Not supported yet!")
+        if "parser_config" in req:
+            DocumentService.update_parser_config(doc.id, req["parser_config"])
+        reset_doc()
        return get_json_result(data=True)
    except Exception as e:
        return server_error_response(e)
--- a/api/apps/kb_app.py
+++ b/api/apps/kb_app.py
@ -64,7 +64,7 @@ def create():
        e, t = TenantService.get_by_id(current_user.id)
        if not e:
            return get_data_error_result(message="Tenant not found.")
-        req["embd_id"] = t.embd_id
+        #req["embd_id"] = t.embd_id
        if not KnowledgebaseService.save(**req):
            return get_data_error_result()
        return get_json_result(data={"kb_id": req["id"]})
@ -379,3 +379,19 @@ def get_meta():
                code=settings.RetCode.AUTHENTICATION_ERROR
            )
    return get_json_result(data=DocumentService.get_meta_by_kbs(kb_ids))
+
+
+@manager.route("/basic_info", methods=["GET"])  # noqa: F821
+@login_required
+def get_basic_info():
+    kb_id = request.args.get("kb_id", "")
+    if not KnowledgebaseService.accessible(kb_id, current_user.id):
+        return get_json_result(
+            data=False,
+            message='No authorization.',
+            code=settings.RetCode.AUTHENTICATION_ERROR
+        )
+
+    basic_info = DocumentService.knowledgebase_basic_info(kb_id)
+
+    return get_json_result(data=basic_info)
--- a/api/apps/sdk/session.py
+++ b/api/apps/sdk/session.py
@ -414,7 +414,7 @@ def agents_completion_openai_compatibility(tenant_id, agent_id):
                tenant_id,
                agent_id,
                question,
-                session_id=req.get("session_id", req.get("id", "") or req.get("metadata", {}).get("id", "")),
+                session_id=req.pop("session_id", req.get("id", "")) or req.get("metadata", {}).get("id", ""),
                stream=True,
                **req,
            ),
@ -432,7 +432,7 @@ def agents_completion_openai_compatibility(tenant_id, agent_id):
                tenant_id,
                agent_id,
                question,
-                session_id=req.get("session_id", req.get("id", "") or req.get("metadata", {}).get("id", "")),
+                session_id=req.pop("session_id", req.get("id", "")) or req.get("metadata", {}).get("id", ""),
                stream=False,
                **req,
            )
--- a/api/apps/system_app.py
+++ b/api/apps/system_app.py
@ -36,6 +36,8 @@ from rag.utils.storage_factory import STORAGE_IMPL, STORAGE_IMPL_TYPE
 from timeit import default_timer as timer

 from rag.utils.redis_conn import REDIS_CONN
+from flask import jsonify
+from api.utils.health import run_health_checks

@manager.route("/version", methods=["GET"])  # noqa: F821
@login_required
@ -169,6 +171,12 @@ def status():
    return get_json_result(data=res)


+@manager.route("/healthz", methods=["GET"])  # noqa: F821
+def healthz():
+    result, all_ok = run_health_checks()
+    return jsonify(result), (200 if all_ok else 500)
+
+
@manager.route("/new_token", methods=["POST"])  # noqa: F821
@login_required
 def new_token():
--- a/api/db/db_models.py
+++ b/api/db/db_models.py
@ -646,6 +646,7 @@ class Knowledgebase(DataBaseModel):
    vector_similarity_weight = FloatField(default=0.3, index=True)

    parser_id = CharField(max_length=32, null=False, help_text="default parser ID", default=ParserType.NAIVE.value, index=True)
+    pipeline_id = CharField(max_length=32, null=True, help_text="Pipeline ID", index=True)
    parser_config = JSONField(null=False, default={"pages": [[1, 1000000]]})
    pagerank = IntegerField(default=0, index=False)
    status = CharField(max_length=1, null=True, help_text="is it validate(0: wasted, 1: validate)", default="1", index=True)
@ -662,6 +663,7 @@ class Document(DataBaseModel):
    thumbnail = TextField(null=True, help_text="thumbnail base64 string")
    kb_id = CharField(max_length=256, null=False, index=True)
    parser_id = CharField(max_length=32, null=False, help_text="default parser ID", index=True)
+    pipeline_id = CharField(max_length=32, null=True, help_text="pipleline ID", index=True)
    parser_config = JSONField(null=False, default={"pages": [[1, 1000000]]})
    source_type = CharField(max_length=128, null=False, default="local", help_text="where dose this document come from", index=True)
    type = CharField(max_length=32, null=False, help_text="file extension", index=True)
@ -1020,7 +1022,6 @@ def migrate_db():
        migrate(migrator.add_column("dialog", "meta_data_filter", JSONField(null=True, default={})))
    except Exception:
        pass
-
    try:
        migrate(migrator.alter_column_type("canvas_template", "title", JSONField(null=True, default=dict, help_text="Canvas title")))
    except Exception:
@ -1037,4 +1038,12 @@ def migrate_db():
        migrate(migrator.add_column("canvas_template", "canvas_category", CharField(max_length=32, null=False, default="agent_canvas", help_text="agent_canvas|dataflow_canvas", index=True)))
    except Exception:
        pass
+    try:
+        migrate(migrator.add_column("knowledgebase", "pipeline_id", CharField(max_length=32, null=True, help_text="default parser ID", index=True)))
+    except Exception:
+        pass
+    try:
+        migrate(migrator.add_column("document", "pipeline_id", CharField(max_length=32, null=True, help_text="default parser ID", index=True)))
+    except Exception:
+        pass
    logging.disable(logging.NOTSET)
--- a/api/db/services/canvas_service.py
+++ b/api/db/services/canvas_service.py
@ -95,7 +95,7 @@ class UserCanvasService(CommonService):
    @DB.connection_context()
    def get_by_tenant_ids(cls, joined_tenant_ids, user_id,
                          page_number, items_per_page,
-                          orderby, desc, keywords, canvas_category=CanvasCategory.Agent,
+                          orderby, desc, keywords, canvas_category=None
                          ):
        fields = [
            cls.model.id,
@ -122,6 +122,7 @@ class UserCanvasService(CommonService):
                                                                TenantPermission.TEAM.value)) | (
                    cls.model.user_id == user_id))
            )
+        if canvas_category:
            agents = agents.where(cls.model.canvas_category == canvas_category)
        if desc:
            agents = agents.order_by(cls.model.getter_by(orderby).desc())
--- a/api/db/services/document_service.py
+++ b/api/db/services/document_service.py
@ -24,7 +24,7 @@ from io import BytesIO

 import trio
 import xxhash
-from peewee import fn
+from peewee import fn, Case

 from api import settings
 from api.constants import IMG_BASE64_PREFIX, FILE_NAME_LEN_LIMIT
@ -674,6 +674,53 @@ class DocumentService(CommonService):
        return False


+    @classmethod
+    @DB.connection_context()
+    def knowledgebase_basic_info(cls, kb_id: str) -> dict[str, int]:
+        # cancelled: run == "2" but progress can vary
+        cancelled = (
+            cls.model.select(fn.COUNT(1))
+            .where((cls.model.kb_id == kb_id) & (cls.model.run == TaskStatus.CANCEL))
+            .scalar()
+        )
+
+        row = (
+            cls.model.select(
+                # finished: progress == 1
+                fn.COALESCE(fn.SUM(Case(None, [(cls.model.progress == 1, 1)], 0)), 0).alias("finished"),
+
+                # failed: progress == -1
+                fn.COALESCE(fn.SUM(Case(None, [(cls.model.progress == -1, 1)], 0)), 0).alias("failed"),
+
+                # processing: 0 <= progress < 1
+                fn.COALESCE(
+                    fn.SUM(
+                        Case(
+                            None,
+                            [
+                                (((cls.model.progress == 0) | ((cls.model.progress > 0) & (cls.model.progress < 1))), 1),
+                            ],
+                            0,
+                        )
+                    ),
+                    0,
+                ).alias("processing"),
+            )
+            .where(
+                (cls.model.kb_id == kb_id)
+                & ((cls.model.run.is_null(True)) | (cls.model.run != TaskStatus.CANCEL))
+            )
+            .dicts()
+            .get()
+        )
+
+        return {
+            "processing": int(row["processing"]),
+            "finished": int(row["finished"]),
+            "failed": int(row["failed"]),
+            "cancelled": int(cancelled),
+        }
+
 def queue_raptor_o_graphrag_tasks(doc, ty, priority):
    chunking_config = DocumentService.get_chunking_config(doc["id"])
    hasher = xxhash.xxh64()
@ -702,6 +749,8 @@ def queue_raptor_o_graphrag_tasks(doc, ty, priority):

 def get_queue_length(priority):
    group_info = REDIS_CONN.queue_info(get_svr_queue_name(priority), SVR_CONSUMER_GROUP_NAME)
+    if not group_info:
+        return 0
    return int(group_info.get("lag", 0) or 0)


@ -847,3 +896,4 @@ def doc_upload_and_parse(conversation_id, file_objs, user_id):
            doc_id, kb.id, token_counts[doc_id], chunk_counts[doc_id], 0)

    return [d["id"] for d, _ in files]
+
--- a/api/db/services/file_service.py
+++ b/api/db/services/file_service.py
@ -440,6 +440,7 @@ class FileService(CommonService):
                    "id": doc_id,
                    "kb_id": kb.id,
                    "parser_id": self.get_parser(filetype, filename, kb.parser_id),
+                    "pipeline_id": kb.pipeline_id,
                    "parser_config": kb.parser_config,
                    "created_by": user_id,
                    "type": filetype,
--- a/api/db/services/task_service.py
+++ b/api/db/services/task_service.py
@ -472,7 +472,7 @@ def has_canceled(task_id):
    return False


-def queue_dataflow(dsl:str, tenant_id:str, doc_id:str, task_id:str, flow_id:str, priority: int, callback=None) -> tuple[bool, str]:
+def queue_dataflow(dsl:str, tenant_id:str, task_id:str, flow_id:str=None, doc_id:str=None, file:dict=None, priority: int=0, callback=None) -> tuple[bool, str]:
    """
    Returns a tuple (success: bool, error_message: str).
    """
@ -499,6 +499,7 @@ def queue_dataflow(dsl:str, tenant_id:str, doc_id:str, task_id:str, flow_id:str,
    task["task_type"] = "dataflow"
    task["dsl"] = dsl
    task["dataflow_id"] = get_uuid() if not flow_id else flow_id
+    task["file"] = file

    if not REDIS_CONN.queue_product(
        get_svr_queue_name(priority), message=task
--- a/api/utils/base64_image.py
+++ b/api/utils/base64_image.py
@ -1,3 +1,51 @@
 import base64
+from functools import partial
+from io import BytesIO
+
+from PIL import Image
+
 test_image_base64 = "iVBORw0KGgoAAAANSUhEUgAAAGQAAABkCAIAAAD/gAIDAAAA6ElEQVR4nO3QwQ3AIBDAsIP9d25XIC+EZE8QZc18w5l9O+AlZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBWYFZgVmBT+IYAHHLHkdEgAAAABJRU5ErkJggg=="
 test_image = base64.b64decode(test_image_base64)
+
+async def image2id(d: dict, storage_put_func: partial, bucket:str, objname:str):
+    import logging
+    from io import BytesIO
+    import trio
+    from rag.svr.task_executor import minio_limiter
+    if not d.get("image"):
+        return
+
+    with BytesIO() as output_buffer:
+        if isinstance(d["image"], bytes):
+            output_buffer.write(d["image"])
+            output_buffer.seek(0)
+        else:
+            # If the image is in RGBA mode, convert it to RGB mode before saving it in JPEG format.
+            if d["image"].mode in ("RGBA", "P"):
+                converted_image = d["image"].convert("RGB")
+                d["image"] = converted_image
+            try:
+                d["image"].save(output_buffer, format='JPEG')
+            except OSError as e:
+                logging.warning(
+                    "Saving image exception, ignore: {}".format(str(e)))
+
+        async with minio_limiter:
+            await trio.to_thread.run_sync(lambda: storage_put_func(bucket=bucket, fnm=objname, binary=output_buffer.getvalue()))
+        d["img_id"] = f"{bucket}-{objname}"
+        if not isinstance(d["image"], bytes):
+            d["image"].close()
+        del d["image"]  # Remove image reference
+
+
+def id2image(image_id:str|None, storage_get_func: partial):
+    if not image_id:
+        return
+    arr = image_id.split("-")
+    if len(arr) != 2:
+        return
+    bkt, nm = image_id.split("-")
+    blob = storage_get_func(bucket=bkt, filename=nm)
+    if not blob:
+        return
+    return Image.open(BytesIO(blob))
--- a/api/utils/health.py
+++ b/api/utils/health.py
@ -0,0 +1,104 @@
+from timeit import default_timer as timer
+
+from api import settings
+from api.db.db_models import DB
+from rag.utils.redis_conn import REDIS_CONN
+from rag.utils.storage_factory import STORAGE_IMPL
+
+
+def _ok_nok(ok: bool) -> str:
+    return "ok" if ok else "nok"
+
+
+def check_db() -> tuple[bool, dict]:
+    st = timer()
+    try:
+        # lightweight probe; works for MySQL/Postgres
+        DB.execute_sql("SELECT 1")
+        return True, {"elapsed": f"{(timer() - st) * 1000.0:.1f}"}
+    except Exception as e:
+        return False, {"elapsed": f"{(timer() - st) * 1000.0:.1f}", "error": str(e)}
+
+
+def check_redis() -> tuple[bool, dict]:
+    st = timer()
+    try:
+        ok = bool(REDIS_CONN.health())
+        return ok, {"elapsed": f"{(timer() - st) * 1000.0:.1f}"}
+    except Exception as e:
+        return False, {"elapsed": f"{(timer() - st) * 1000.0:.1f}", "error": str(e)}
+
+
+def check_doc_engine() -> tuple[bool, dict]:
+    st = timer()
+    try:
+        meta = settings.docStoreConn.health()
+        # treat any successful call as ok
+        return True, {"elapsed": f"{(timer() - st) * 1000.0:.1f}", **(meta or {})}
+    except Exception as e:
+        return False, {"elapsed": f"{(timer() - st) * 1000.0:.1f}", "error": str(e)}
+
+
+def check_storage() -> tuple[bool, dict]:
+    st = timer()
+    try:
+        STORAGE_IMPL.health()
+        return True, {"elapsed": f"{(timer() - st) * 1000.0:.1f}"}
+    except Exception as e:
+        return False, {"elapsed": f"{(timer() - st) * 1000.0:.1f}", "error": str(e)}
+
+
+def check_chat() -> tuple[bool, dict]:
+    st = timer()
+    try:
+        cfg = getattr(settings, "CHAT_CFG", None)
+        ok = bool(cfg and cfg.get("factory"))
+        return ok, {"elapsed": f"{(timer() - st) * 1000.0:.1f}"}
+    except Exception as e:
+        return False, {"elapsed": f"{(timer() - st) * 1000.0:.1f}", "error": str(e)}
+
+
+def run_health_checks() -> tuple[dict, bool]:
+    result: dict[str, str | dict] = {}
+
+    db_ok, db_meta = check_db()
+    chat_ok, chat_meta = check_chat()
+
+    result["db"] = _ok_nok(db_ok)
+    if not db_ok:
+        result.setdefault("_meta", {})["db"] = db_meta
+
+    result["chat"] = _ok_nok(chat_ok)
+    if not chat_ok:
+        result.setdefault("_meta", {})["chat"] = chat_meta
+
+    # Optional probes (do not change minimal contract but exposed for observability)
+    try:
+        redis_ok, redis_meta = check_redis()
+        result["redis"] = _ok_nok(redis_ok)
+        if not redis_ok:
+            result.setdefault("_meta", {})["redis"] = redis_meta
+    except Exception:
+        result["redis"] = "nok"
+
+    try:
+        doc_ok, doc_meta = check_doc_engine()
+        result["doc_engine"] = _ok_nok(doc_ok)
+        if not doc_ok:
+            result.setdefault("_meta", {})["doc_engine"] = doc_meta
+    except Exception:
+        result["doc_engine"] = "nok"
+
+    try:
+        sto_ok, sto_meta = check_storage()
+        result["storage"] = _ok_nok(sto_ok)
+        if not sto_ok:
+            result.setdefault("_meta", {})["storage"] = sto_meta
+    except Exception:
+        result["storage"] = "nok"
+
+    all_ok = (result.get("db") == "ok") and (result.get("chat") == "ok")
+    result["status"] = "ok" if all_ok else "nok"
+    return result, all_ok
+
+
--- a/conf/llm_factories.json
+++ b/conf/llm_factories.json
@ -219,6 +219,70 @@
                }
            ]
        },
+        {
+            "name": "TokenPony",
+            "logo": "",
+            "tags": "LLM",
+            "status": "1",
+            "llm": [
+                {
+                    "llm_name": "qwen3-8b",
+                    "tags": "LLM,CHAT,131k",
+                    "max_tokens": 131000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "deepseek-v3-0324",
+                    "tags": "LLM,CHAT,128k",
+                    "max_tokens": 128000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "qwen3-32b",
+                    "tags": "LLM,CHAT,131k",
+                    "max_tokens": 131000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "kimi-k2-instruct",
+                    "tags": "LLM,CHAT,128K",
+                    "max_tokens": 128000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "deepseek-r1-0528",
+                    "tags": "LLM,CHAT,164k",
+                    "max_tokens": 164000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "qwen3-coder-480b",
+                    "tags": "LLM,CHAT,1024k",
+                    "max_tokens": 1024000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "glm-4.5",
+                    "tags": "LLM,CHAT,131K",
+                    "max_tokens": 131000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "deepseek-v3.1",
+                    "tags": "LLM,CHAT,128k",
+                    "max_tokens": 128000,
+                    "model_type": "chat",
+                    "is_tools": true
+                }
+            ]
+        },
        {
            "name": "Tongyi-Qianwen",
            "logo": "",
@ -4477,6 +4541,273 @@
                }
            ]
        },
+        {
+            "name": "CometAPI",
+            "logo": "",
+            "tags": "LLM,TEXT EMBEDDING,IMAGE2TEXT",
+            "status": "1",
+            "llm": [
+                {
+                    "llm_name": "gpt-5-chat-latest",
+                    "tags": "LLM,CHAT,400k",
+                    "max_tokens": 400000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "chatgpt-4o-latest",
+                    "tags": "LLM,CHAT,128k",
+                    "max_tokens": 128000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "gpt-5-mini",
+                    "tags": "LLM,CHAT,400k",
+                    "max_tokens": 400000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "gpt-5-nano",
+                    "tags": "LLM,CHAT,400k",
+                    "max_tokens": 400000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "gpt-5",
+                    "tags": "LLM,CHAT,400k",
+                    "max_tokens": 400000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "gpt-4.1-mini",
+                    "tags": "LLM,CHAT,1M",
+                    "max_tokens": 1047576,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "gpt-4.1-nano",
+                    "tags": "LLM,CHAT,1M",
+                    "max_tokens": 1047576,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "gpt-4.1",
+                    "tags": "LLM,CHAT,1M",
+                    "max_tokens": 1047576,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "gpt-4o-mini",
+                    "tags": "LLM,CHAT,128k",
+                    "max_tokens": 128000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "o4-mini-2025-04-16",
+                    "tags": "LLM,CHAT,200k",
+                    "max_tokens": 200000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "o3-pro-2025-06-10",
+                    "tags": "LLM,CHAT,200k",
+                    "max_tokens": 200000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "claude-opus-4-1-20250805",
+                    "tags": "LLM,CHAT,200k,IMAGE2TEXT",
+                    "max_tokens": 200000,
+                    "model_type": "image2text",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "claude-opus-4-1-20250805-thinking",
+                    "tags": "LLM,CHAT,200k,IMAGE2TEXT",
+                    "max_tokens": 200000,
+                    "model_type": "image2text",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "claude-sonnet-4-20250514",
+                    "tags": "LLM,CHAT,200k,IMAGE2TEXT",
+                    "max_tokens": 200000,
+                    "model_type": "image2text",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "claude-sonnet-4-20250514-thinking",
+                    "tags": "LLM,CHAT,200k,IMAGE2TEXT",
+                    "max_tokens": 200000,
+                    "model_type": "image2text",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "claude-3-7-sonnet-latest",
+                    "tags": "LLM,CHAT,200k",
+                    "max_tokens": 200000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "claude-3-5-haiku-latest",
+                    "tags": "LLM,CHAT,200k",
+                    "max_tokens": 200000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "gemini-2.5-pro",
+                    "tags": "LLM,CHAT,1M,IMAGE2TEXT",
+                    "max_tokens": 1000000,
+                    "model_type": "image2text",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "gemini-2.5-flash",
+                    "tags": "LLM,CHAT,1M,IMAGE2TEXT",
+                    "max_tokens": 1000000,
+                    "model_type": "image2text",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "gemini-2.5-flash-lite",
+                    "tags": "LLM,CHAT,1M,IMAGE2TEXT",
+                    "max_tokens": 1000000,
+                    "model_type": "image2text",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "gemini-2.0-flash",
+                    "tags": "LLM,CHAT,1M,IMAGE2TEXT",
+                    "max_tokens": 1000000,
+                    "model_type": "image2text",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "grok-4-0709",
+                    "tags": "LLM,CHAT,131k",
+                    "max_tokens": 131072,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "grok-3",
+                    "tags": "LLM,CHAT,131k",
+                    "max_tokens": 131072,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "grok-3-mini",
+                    "tags": "LLM,CHAT,131k",
+                    "max_tokens": 131072,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "grok-2-image-1212",
+                    "tags": "LLM,CHAT,32k,IMAGE2TEXT",
+                    "max_tokens": 32768,
+                    "model_type": "image2text",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "deepseek-v3.1",
+                    "tags": "LLM,CHAT,64k",
+                    "max_tokens": 64000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "deepseek-v3",
+                    "tags": "LLM,CHAT,64k",
+                    "max_tokens": 64000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "deepseek-r1-0528",
+                    "tags": "LLM,CHAT,164k",
+                    "max_tokens": 164000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "deepseek-chat",
+                    "tags": "LLM,CHAT,32k",
+                    "max_tokens": 32000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "deepseek-reasoner",
+                    "tags": "LLM,CHAT,64k",
+                    "max_tokens": 64000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "qwen3-30b-a3b",
+                    "tags": "LLM,CHAT,128k",
+                    "max_tokens": 128000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "qwen3-coder-plus-2025-07-22",
+                    "tags": "LLM,CHAT,128k",
+                    "max_tokens": 128000,
+                    "model_type": "chat",
+                    "is_tools": true
+                },
+                {
+                    "llm_name": "text-embedding-ada-002",
+                    "tags": "TEXT EMBEDDING,8K",
+                    "max_tokens": 8191,
+                    "model_type": "embedding",
+                    "is_tools": false
+                },
+                {
+                    "llm_name": "text-embedding-3-small",
+                    "tags": "TEXT EMBEDDING,8K",
+                    "max_tokens": 8191,
+                    "model_type": "embedding",
+                    "is_tools": false
+                },
+                {
+                    "llm_name": "text-embedding-3-large",
+                    "tags": "TEXT EMBEDDING,8K",
+                    "max_tokens": 8191,
+                    "model_type": "embedding",
+                    "is_tools": false
+                },
+                {
+                    "llm_name": "whisper-1",
+                    "tags": "SPEECH2TEXT",
+                    "max_tokens": 26214400,
+                    "model_type": "speech2text",
+                    "is_tools": false
+                },
+                {
+                    "llm_name": "tts-1",
+                    "tags": "TTS",
+                    "max_tokens": 2048,
+                    "model_type": "tts",
+                    "is_tools": false
+                }
+            ]
+        },
        {
            "name": "Meituan",
            "logo": "",
--- a/deepdoc/parser/html_parser.py
+++ b/deepdoc/parser/html_parser.py
@ -37,7 +37,7 @@ TITLE_TAGS = {"h1": "#", "h2": "##", "h3": "###", "h4": "#####", "h5": "#####",


 class RAGFlowHtmlParser:
-    def __call__(self, fnm, binary=None, chunk_token_num=None):
+    def __call__(self, fnm, binary=None, chunk_token_num=512):
        if binary:
            encoding = find_codec(binary)
            txt = binary.decode(encoding, errors="ignore")
--- a/deepdoc/parser/pdf_parser.py
+++ b/deepdoc/parser/pdf_parser.py
@ -34,7 +34,7 @@ from pypdf import PdfReader as pdf2_read

 from api import settings
 from api.utils.file_utils import get_project_base_directory
-from deepdoc.vision import OCR, LayoutRecognizer, Recognizer, TableStructureRecognizer
+from deepdoc.vision import OCR, AscendLayoutRecognizer, LayoutRecognizer, Recognizer, TableStructureRecognizer
 from rag.app.picture import vision_llm_chunk as picture_vision_llm_chunk
 from rag.nlp import rag_tokenizer
 from rag.prompts import vision_llm_describe_prompt
@ -64,33 +64,38 @@ class RAGFlowPdfParser:
        if PARALLEL_DEVICES > 1:
            self.parallel_limiter = [trio.CapacityLimiter(1) for _ in range(PARALLEL_DEVICES)]

+        layout_recognizer_type = os.getenv("LAYOUT_RECOGNIZER_TYPE", "onnx").lower()
+        if layout_recognizer_type not in ["onnx", "ascend"]:
+            raise RuntimeError("Unsupported layout recognizer type.")
+
        if hasattr(self, "model_speciess"):
-            self.layouter = LayoutRecognizer("layout." + self.model_speciess)
+            recognizer_domain = "layout." + self.model_speciess
        else:
-            self.layouter = LayoutRecognizer("layout")
+            recognizer_domain = "layout"
+
+        if layout_recognizer_type == "ascend":
+            logging.debug("Using Ascend LayoutRecognizer")
+            self.layouter = AscendLayoutRecognizer(recognizer_domain)
+        else:  # onnx
+            logging.debug("Using Onnx LayoutRecognizer")
+            self.layouter = LayoutRecognizer(recognizer_domain)
        self.tbl_det = TableStructureRecognizer()

        self.updown_cnt_mdl = xgb.Booster()
        if not settings.LIGHTEN:
            try:
                import torch.cuda
+
                if torch.cuda.is_available():
                    self.updown_cnt_mdl.set_param({"device": "cuda"})
            except Exception:
                logging.exception("RAGFlowPdfParser __init__")
        try:
-            model_dir = os.path.join(
-                get_project_base_directory(),
-                "rag/res/deepdoc")
-            self.updown_cnt_mdl.load_model(os.path.join(
-                model_dir, "updown_concat_xgb.model"))
+            model_dir = os.path.join(get_project_base_directory(), "rag/res/deepdoc")
+            self.updown_cnt_mdl.load_model(os.path.join(model_dir, "updown_concat_xgb.model"))
        except Exception:
-            model_dir = snapshot_download(
-                repo_id="InfiniFlow/text_concat_xgb_v1.0",
-                local_dir=os.path.join(get_project_base_directory(), "rag/res/deepdoc"),
-                local_dir_use_symlinks=False)
-            self.updown_cnt_mdl.load_model(os.path.join(
-                model_dir, "updown_concat_xgb.model"))
+            model_dir = snapshot_download(repo_id="InfiniFlow/text_concat_xgb_v1.0", local_dir=os.path.join(get_project_base_directory(), "rag/res/deepdoc"), local_dir_use_symlinks=False)
+            self.updown_cnt_mdl.load_model(os.path.join(model_dir, "updown_concat_xgb.model"))

        self.page_from = 0
        self.column_num = 1
@ -102,13 +107,10 @@ class RAGFlowPdfParser:
        return c["bottom"] - c["top"]

    def _x_dis(self, a, b):
-        return min(abs(a["x1"] - b["x0"]), abs(a["x0"] - b["x1"]),
-                   abs(a["x0"] + a["x1"] - b["x0"] - b["x1"]) / 2)
+        return min(abs(a["x1"] - b["x0"]), abs(a["x0"] - b["x1"]), abs(a["x0"] + a["x1"] - b["x0"] - b["x1"]) / 2)

-    def _y_dis(
-            self, a, b):
-        return (
-            b["top"] + b["bottom"] - a["top"] - a["bottom"]) / 2
+    def _y_dis(self, a, b):
+        return (b["top"] + b["bottom"] - a["top"] - a["bottom"]) / 2

    def _match_proj(self, b):
        proj_patt = [
@ -130,10 +132,7 @@ class RAGFlowPdfParser:
        LEN = 6
        tks_down = rag_tokenizer.tokenize(down["text"][:LEN]).split()
        tks_up = rag_tokenizer.tokenize(up["text"][-LEN:]).split()
-        tks_all = up["text"][-LEN:].strip() \
-            + (" " if re.match(r"[a-zA-Z0-9]+",
-                               up["text"][-1] + down["text"][0]) else "") \
-            + down["text"][:LEN].strip()
+        tks_all = up["text"][-LEN:].strip() + (" " if re.match(r"[a-zA-Z0-9]+", up["text"][-1] + down["text"][0]) else "") + down["text"][:LEN].strip()
        tks_all = rag_tokenizer.tokenize(tks_all).split()
        fea = [
            up.get("R", -1) == down.get("R", -1),
@ -144,39 +143,30 @@ class RAGFlowPdfParser:
            down["layout_type"] == "text",
            up["layout_type"] == "table",
            down["layout_type"] == "table",
-            True if re.search(
-                r"([。？！；!?;+)）]|[a-z]\.)$",
-                up["text"]) else False,
+            True if re.search(r"([。？！；!?;+)）]|[a-z]\.)$", up["text"]) else False,
            True if re.search(r"[，：‘“、0-9（+-]$", up["text"]) else False,
-            True if re.search(
-                r"(^.?[/,?;:\]，。；：’”？！》】）-])",
-                down["text"]) else False,
+            True if re.search(r"(^.?[/,?;:\]，。；：’”？！》】）-])", down["text"]) else False,
            True if re.match(r"[\(（][^\(\)（）]+[）\)]$", up["text"]) else False,
            True if re.search(r"[，,][^。.]+$", up["text"]) else False,
            True if re.search(r"[，,][^。.]+$", up["text"]) else False,
-            True if re.search(r"[\(（][^\)）]+$", up["text"])
-            and re.search(r"[\)）]", down["text"]) else False,
+            True if re.search(r"[\(（][^\)）]+$", up["text"]) and re.search(r"[\)）]", down["text"]) else False,
            self._match_proj(down),
            True if re.match(r"[A-Z]", down["text"]) else False,
            True if re.match(r"[A-Z]", up["text"][-1]) else False,
            True if re.match(r"[a-z0-9]", up["text"][-1]) else False,
            True if re.match(r"[0-9.%,-]+$", down["text"]) else False,
-            up["text"].strip()[-2:] == down["text"].strip()[-2:] if len(up["text"].strip()
-                                                                        ) > 1 and len(
-                down["text"].strip()) > 1 else False,
+            up["text"].strip()[-2:] == down["text"].strip()[-2:] if len(up["text"].strip()) > 1 and len(down["text"].strip()) > 1 else False,
            up["x0"] > down["x1"],
-            abs(self.__height(up) - self.__height(down)) / min(self.__height(up),
-                                                               self.__height(down)),
+            abs(self.__height(up) - self.__height(down)) / min(self.__height(up), self.__height(down)),
            self._x_dis(up, down) / max(w, 0.000001),
-            (len(up["text"]) - len(down["text"])) /
-            max(len(up["text"]), len(down["text"])),
+            (len(up["text"]) - len(down["text"])) / max(len(up["text"]), len(down["text"])),
            len(tks_all) - len(tks_up) - len(tks_down),
            len(tks_down) - len(tks_up),
            tks_down[-1] == tks_up[-1] if tks_down and tks_up else False,
            max(down["in_row"], up["in_row"]),
            abs(down["in_row"] - up["in_row"]),
            len(tks_down) == 1 and rag_tokenizer.tag(tks_down[0]).find("n") >= 0,
-            len(tks_up) == 1 and rag_tokenizer.tag(tks_up[0]).find("n") >= 0
+            len(tks_up) == 1 and rag_tokenizer.tag(tks_up[0]).find("n") >= 0,
        ]
        return fea

@ -187,9 +177,7 @@ class RAGFlowPdfParser:
        for i in range(len(arr) - 1):
            for j in range(i, -1, -1):
                # restore the order using th
-                if abs(arr[j + 1]["x0"] - arr[j]["x0"]) < threshold \
-                        and arr[j + 1]["top"] < arr[j]["top"] \
-                        and arr[j + 1]["page_number"] == arr[j]["page_number"]:
+                if abs(arr[j + 1]["x0"] - arr[j]["x0"]) < threshold and arr[j + 1]["top"] < arr[j]["top"] and arr[j + 1]["page_number"] == arr[j]["page_number"]:
                    tmp = arr[j]
                    arr[j] = arr[j + 1]
                    arr[j + 1] = tmp
@ -197,8 +185,7 @@ class RAGFlowPdfParser:

    def _has_color(self, o):
        if o.get("ncs", "") == "DeviceGray":
-            if o["stroking_color"] and o["stroking_color"][0] == 1 and o["non_stroking_color"] and \
-                    o["non_stroking_color"][0] == 1:
+            if o["stroking_color"] and o["stroking_color"][0] == 1 and o["non_stroking_color"] and o["non_stroking_color"][0] == 1:
                if re.match(r"[a-zT_\[\]\(\)-]+", o.get("text", "")):
                    return False
        return True
@ -216,8 +203,7 @@ class RAGFlowPdfParser:
            if not tbls:
                continue
            for tb in tbls:  # for table
-                left, top, right, bott = tb["x0"] - MARGIN, tb["top"] - MARGIN, \
-                    tb["x1"] + MARGIN, tb["bottom"] + MARGIN
+                left, top, right, bott = tb["x0"] - MARGIN, tb["top"] - MARGIN, tb["x1"] + MARGIN, tb["bottom"] + MARGIN
                left *= ZM
                top *= ZM
                right *= ZM
@ -232,14 +218,13 @@ class RAGFlowPdfParser:
        tbcnt = np.cumsum(tbcnt)
        for i in range(len(tbcnt) - 1):  # for page
            pg = []
-            for j, tb_items in enumerate(
-                    recos[tbcnt[i]: tbcnt[i + 1]]):  # for table
+            for j, tb_items in enumerate(recos[tbcnt[i] : tbcnt[i + 1]]):  # for table
                poss = pos[tbcnt[i] : tbcnt[i + 1]]
                for it in tb_items:  # for table components
-                    it["x0"] = (it["x0"] + poss[j][0])
-                    it["x1"] = (it["x1"] + poss[j][0])
-                    it["top"] = (it["top"] + poss[j][1])
-                    it["bottom"] = (it["bottom"] + poss[j][1])
+                    it["x0"] = it["x0"] + poss[j][0]
+                    it["x1"] = it["x1"] + poss[j][0]
+                    it["top"] = it["top"] + poss[j][1]
+                    it["bottom"] = it["bottom"] + poss[j][1]
                    for n in ["x0", "x1", "top", "bottom"]:
                        it[n] /= ZM
                    it["top"] += self.page_cum_height[i]
@ -250,8 +235,7 @@ class RAGFlowPdfParser:
            self.tb_cpns.extend(pg)

        def gather(kwd, fzy=10, ption=0.6):
-            eles = Recognizer.sort_Y_firstly(
-                [r for r in self.tb_cpns if re.match(kwd, r["label"])], fzy)
+            eles = Recognizer.sort_Y_firstly([r for r in self.tb_cpns if re.match(kwd, r["label"])], fzy)
            eles = Recognizer.layouts_cleanup(self.boxes, eles, 5, ption)
            return Recognizer.sort_Y_firstly(eles, 0)

@ -259,8 +243,7 @@ class RAGFlowPdfParser:
        headers = gather(r".*header$")
        rows = gather(r".* (row|header)")
        spans = gather(r".*spanning")
-        clmns = sorted([r for r in self.tb_cpns if re.match(
-            r"table column$", r["label"])], key=lambda x: (x["pn"], x["layoutno"], x["x0"]))
+        clmns = sorted([r for r in self.tb_cpns if re.match(r"table column$", r["label"])], key=lambda x: (x["pn"], x["layoutno"], x["x0"]))
        clmns = Recognizer.layouts_cleanup(self.boxes, clmns, 5, 0.5)
        for b in self.boxes:
            if b.get("layout_type", "") != "table":
@ -271,8 +254,7 @@ class RAGFlowPdfParser:
                b["R_top"] = rows[ii]["top"]
                b["R_bott"] = rows[ii]["bottom"]

-            ii = Recognizer.find_overlapped_with_threshold(
-                b, headers, thr=0.3)
+            ii = Recognizer.find_overlapped_with_threshold(b, headers, thr=0.3)
            if ii is not None:
                b["H_top"] = headers[ii]["top"]
                b["H_bott"] = headers[ii]["bottom"]
@ -305,12 +287,12 @@ class RAGFlowPdfParser:
            return
        bxs = [(line[0], line[1][0]) for line in bxs]
        bxs = Recognizer.sort_Y_firstly(
-            [{"x0": b[0][0] / ZM, "x1": b[1][0] / ZM,
-              "top": b[0][1] / ZM, "text": "", "txt": t,
-              "bottom": b[-1][1] / ZM,
-              "chars": [],
-              "page_number": pagenum} for b, t in bxs if b[0][0] <= b[1][0] and b[0][1] <= b[-1][1]],
-            self.mean_height[pagenum-1] / 3
+            [
+                {"x0": b[0][0] / ZM, "x1": b[1][0] / ZM, "top": b[0][1] / ZM, "text": "", "txt": t, "bottom": b[-1][1] / ZM, "chars": [], "page_number": pagenum}
+                for b, t in bxs
+                if b[0][0] <= b[1][0] and b[0][1] <= b[-1][1]
+            ],
+            self.mean_height[pagenum - 1] / 3,
        )

        # merge chars in the same rect
@ -321,7 +303,7 @@ class RAGFlowPdfParser:
                continue
            ch = c["bottom"] - c["top"]
            bh = bxs[ii]["bottom"] - bxs[ii]["top"]
-            if abs(ch - bh) / max(ch, bh) >= 0.7 and c["text"] != ' ':
+            if abs(ch - bh) / max(ch, bh) >= 0.7 and c["text"] != " ":
                self.lefted_chars.append(c)
                continue
            bxs[ii]["chars"].append(c)
@ -345,8 +327,7 @@ class RAGFlowPdfParser:
        img_np = np.array(img)
        for b in bxs:
            if not b["text"]:
-                left, right, top, bott = b["x0"] * ZM, b["x1"] * \
-                                         ZM, b["top"] * ZM, b["bottom"] * ZM
+                left, right, top, bott = b["x0"] * ZM, b["x1"] * ZM, b["top"] * ZM, b["bottom"] * ZM
                b["box_image"] = self.ocr.get_rotate_crop_image(img_np, np.array([[left, top], [right, top], [right, bott], [left, bott]], dtype=np.float32))
                boxes_to_reg.append(b)
            del b["txt"]
@ -357,20 +338,16 @@ class RAGFlowPdfParser:
        logging.info(f"__ocr recognize {len(bxs)} boxes cost {timer() - start}s")
        bxs = [b for b in bxs if b["text"]]
        if self.mean_height[pagenum - 1] == 0:
-            self.mean_height[pagenum-1] = np.median([b["bottom"] - b["top"]
-                                              for b in bxs])
+            self.mean_height[pagenum - 1] = np.median([b["bottom"] - b["top"] for b in bxs])
        self.boxes.append(bxs)

    def _layouts_rec(self, ZM, drop=True):
        assert len(self.page_images) == len(self.boxes)
-        self.boxes, self.page_layout = self.layouter(
-            self.page_images, self.boxes, ZM, drop=drop)
+        self.boxes, self.page_layout = self.layouter(self.page_images, self.boxes, ZM, drop=drop)
        # cumlative Y
        for i in range(len(self.boxes)):
-            self.boxes[i]["top"] += \
-                self.page_cum_height[self.boxes[i]["page_number"] - 1]
-            self.boxes[i]["bottom"] += \
-                self.page_cum_height[self.boxes[i]["page_number"] - 1]
+            self.boxes[i]["top"] += self.page_cum_height[self.boxes[i]["page_number"] - 1]
+            self.boxes[i]["bottom"] += self.page_cum_height[self.boxes[i]["page_number"] - 1]

    def _text_merge(self):
        # merge adjusted boxes
@ -390,12 +367,10 @@ class RAGFlowPdfParser:
        while i < len(bxs) - 1:
            b = bxs[i]
            b_ = bxs[i + 1]
-            if b.get("layoutno", "0") != b_.get("layoutno", "1") or b.get("layout_type", "") in ["table", "figure",
-                                                                                                 "equation"]:
+            if b.get("layoutno", "0") != b_.get("layoutno", "1") or b.get("layout_type", "") in ["table", "figure", "equation"]:
                i += 1
                continue
-            if abs(self._y_dis(b, b_)
-                   ) < self.mean_height[bxs[i]["page_number"] - 1] / 3:
+            if abs(self._y_dis(b, b_)) < self.mean_height[bxs[i]["page_number"] - 1] / 3:
                # merge
                bxs[i]["x1"] = b_["x1"]
                bxs[i]["top"] = (b["top"] + b_["top"]) / 2
@ -408,16 +383,14 @@ class RAGFlowPdfParser:

            dis_thr = 1
            dis = b["x1"] - b_["x0"]
-            if b.get("layout_type", "") != "text" or b_.get(
-                    "layout_type", "") != "text":
+            if b.get("layout_type", "") != "text" or b_.get("layout_type", "") != "text":
                if end_with(b, "，") or start_with(b_, "（，"):
                    dis_thr = -8
                else:
                    i += 1
                    continue

-            if abs(self._y_dis(b, b_)) < self.mean_height[bxs[i]["page_number"] - 1] / 5 \
-                    and dis >= dis_thr and b["x1"] < b_["x1"]:
+            if abs(self._y_dis(b, b_)) < self.mean_height[bxs[i]["page_number"] - 1] / 5 and dis >= dis_thr and b["x1"] < b_["x1"]:
                # merge
                bxs[i]["x1"] = b_["x1"]
                bxs[i]["top"] = (b["top"] + b_["top"]) / 2
@ -429,23 +402,22 @@ class RAGFlowPdfParser:
        self.boxes = bxs

    def _naive_vertical_merge(self, zoomin=3):
-        bxs = Recognizer.sort_Y_firstly(
-            self.boxes, np.median(
-                self.mean_height) / 3)
+        import math
+        bxs = Recognizer.sort_Y_firstly(self.boxes, np.median(self.mean_height) / 3)

        column_width = np.median([b["x1"] - b["x0"] for b in self.boxes])
+        if not column_width or math.isnan(column_width):
+            column_width = self.mean_width[0]
        self.column_num = int(self.page_images[0].size[0] / zoomin / column_width)
        if column_width < self.page_images[0].size[0] / zoomin / self.column_num:
-            logging.info("Multi-column................... {} {}".format(column_width,
-                  self.page_images[0].size[0] / zoomin / self.column_num))
+            logging.info("Multi-column................... {} {}".format(column_width, self.page_images[0].size[0] / zoomin / self.column_num))
            self.boxes = self.sort_X_by_page(self.boxes, column_width / self.column_num)

        i = 0
        while i + 1 < len(bxs):
            b = bxs[i]
            b_ = bxs[i + 1]
-            if b["page_number"] < b_["page_number"] and re.match(
-                    r"[0-9  •一—-]+$", b["text"]):
+            if b["page_number"] < b_["page_number"] and re.match(r"[0-9  •一—-]+$", b["text"]):
                bxs.pop(i)
                continue
            if not b["text"].strip():
@ -453,8 +425,7 @@ class RAGFlowPdfParser:
                continue
            concatting_feats = [
                b["text"].strip()[-1] in ",;:'\"，、‘“；：-",
-                len(b["text"].strip()) > 1 and b["text"].strip(
-                )[-2] in ",;:'\"，‘“、；：",
+                len(b["text"].strip()) > 1 and b["text"].strip()[-2] in ",;:'\"，‘“、；：",
                b_["text"].strip() and b_["text"].strip()[0] in "。；？！?”）),，、：",
            ]
            # features for not concating
@ -462,21 +433,20 @@ class RAGFlowPdfParser:
                b.get("layoutno", 0) != b_.get("layoutno", 0),
                b["text"].strip()[-1] in "。？！?",
                self.is_english and b["text"].strip()[-1] in ".!?",
-                b["page_number"] == b_["page_number"] and b_["top"] -
-                b["bottom"] > self.mean_height[b["page_number"] - 1] * 1.5,
-                b["page_number"] < b_["page_number"] and abs(
-                    b["x0"] - b_["x0"]) > self.mean_width[b["page_number"] - 1] * 4,
+                b["page_number"] == b_["page_number"] and b_["top"] - b["bottom"] > self.mean_height[b["page_number"] - 1] * 1.5,
+                b["page_number"] < b_["page_number"] and abs(b["x0"] - b_["x0"]) > self.mean_width[b["page_number"] - 1] * 4,
            ]
            # split features
-            detach_feats = [b["x1"] < b_["x0"],
-                            b["x0"] > b_["x1"]]
+            detach_feats = [b["x1"] < b_["x0"], b["x0"] > b_["x1"]]
            if (any(feats) and not any(concatting_feats)) or any(detach_feats):
-                logging.debug("{} {} {} {}".format(
+                logging.debug(
+                    "{} {} {} {}".format(
                        b["text"],
                        b_["text"],
                        any(feats),
                        any(concatting_feats),
-                ))
+                    )
+                )
                i += 1
                continue
            # merge up and down
@ -529,14 +499,11 @@ class RAGFlowPdfParser:
                    if not concat_between_pages and down["page_number"] > up["page_number"]:
                        break

-                    if up.get("R", "") != down.get(
-                            "R", "") and up["text"][-1] != "，":
+                    if up.get("R", "") != down.get("R", "") and up["text"][-1] != "，":
                        i += 1
                        continue

-                    if re.match(r"[0-9]{2,3}/[0-9]{3}$", up["text"]) \
-                            or re.match(r"[0-9]{2,3}/[0-9]{3}$", down["text"]) \
-                            or not down["text"].strip():
+                    if re.match(r"[0-9]{2,3}/[0-9]{3}$", up["text"]) or re.match(r"[0-9]{2,3}/[0-9]{3}$", down["text"]) or not down["text"].strip():
                        i += 1
                        continue

@ -544,14 +511,12 @@ class RAGFlowPdfParser:
                        i += 1
                        continue

-                    if up["x1"] < down["x0"] - 10 * \
-                            mw or up["x0"] > down["x1"] + 10 * mw:
+                    if up["x1"] < down["x0"] - 10 * mw or up["x0"] > down["x1"] + 10 * mw:
                        i += 1
                        continue

                    if i - dp < 5 and up.get("layout_type") == "text":
-                        if up.get("layoutno", "1") == down.get(
-                                "layoutno", "2"):
+                        if up.get("layoutno", "1") == down.get("layoutno", "2"):
                            dfs(down, i + 1)
                            boxes.pop(i)
                            return
@ -559,8 +524,7 @@ class RAGFlowPdfParser:
                        continue

                    fea = self._updown_concat_features(up, down)
-                    if self.updown_cnt_mdl.predict(
-                            xgb.DMatrix([fea]))[0] <= 0.5:
+                    if self.updown_cnt_mdl.predict(xgb.DMatrix([fea]))[0] <= 0.5:
                        i += 1
                        continue
                    dfs(down, i + 1)
@ -584,16 +548,14 @@ class RAGFlowPdfParser:
                c["text"] = c["text"].strip()
                if not c["text"]:
                    continue
-                if t["text"] and re.match(
-                        r"[0-9\.a-zA-Z]+$", t["text"][-1] + c["text"][-1]):
+                if t["text"] and re.match(r"[0-9\.a-zA-Z]+$", t["text"][-1] + c["text"][-1]):
                    t["text"] += " "
                t["text"] += c["text"]
                t["x0"] = min(t["x0"], c["x0"])
                t["x1"] = max(t["x1"], c["x1"])
                t["page_number"] = min(t["page_number"], c["page_number"])
                t["bottom"] = c["bottom"]
-                if not t["layout_type"] \
-                        and c["layout_type"]:
+                if not t["layout_type"] and c["layout_type"]:
                    t["layout_type"] = c["layout_type"]
            boxes.append(t)

@ -605,25 +567,20 @@ class RAGFlowPdfParser:
        findit = False
        i = 0
        while i < len(self.boxes):
-            if not re.match(r"(contents|目录|目次|table of contents|致谢|acknowledge)$",
-                            re.sub(r"( | |\u3000)+", "", self.boxes[i]["text"].lower())):
+            if not re.match(r"(contents|目录|目次|table of contents|致谢|acknowledge)$", re.sub(r"( | |\u3000)+", "", self.boxes[i]["text"].lower())):
                i += 1
                continue
            findit = True
-            eng = re.match(
-                r"[0-9a-zA-Z :'.-]{5,}",
-                self.boxes[i]["text"].strip())
+            eng = re.match(r"[0-9a-zA-Z :'.-]{5,}", self.boxes[i]["text"].strip())
            self.boxes.pop(i)
            if i >= len(self.boxes):
                break
-            prefix = self.boxes[i]["text"].strip()[:3] if not eng else " ".join(
-                self.boxes[i]["text"].strip().split()[:2])
+            prefix = self.boxes[i]["text"].strip()[:3] if not eng else " ".join(self.boxes[i]["text"].strip().split()[:2])
            while not prefix:
                self.boxes.pop(i)
                if i >= len(self.boxes):
                    break
-                prefix = self.boxes[i]["text"].strip()[:3] if not eng else " ".join(
-                    self.boxes[i]["text"].strip().split()[:2])
+                prefix = self.boxes[i]["text"].strip()[:3] if not eng else " ".join(self.boxes[i]["text"].strip().split()[:2])
            self.boxes.pop(i)
            if i >= len(self.boxes) or not prefix:
                break
@ -662,10 +619,12 @@ class RAGFlowPdfParser:
                self.boxes.pop(i + 1)
                continue

-            if b["text"].strip()[0] != b_["text"].strip()[0] \
-                    or b["text"].strip()[0].lower() in set("qwertyuopasdfghjklzxcvbnm") \
-                    or rag_tokenizer.is_chinese(b["text"].strip()[0]) \
-                    or b["top"] > b_["bottom"]:
+            if (
+                b["text"].strip()[0] != b_["text"].strip()[0]
+                or b["text"].strip()[0].lower() in set("qwertyuopasdfghjklzxcvbnm")
+                or rag_tokenizer.is_chinese(b["text"].strip()[0])
+                or b["top"] > b_["bottom"]
+            ):
                i += 1
                continue
            b_["text"] = b["text"] + "\n" + b_["text"]
@ -685,12 +644,8 @@ class RAGFlowPdfParser:
            if "layoutno" not in self.boxes[i]:
                i += 1
                continue
-            lout_no = str(self.boxes[i]["page_number"]) + \
-                "-" + str(self.boxes[i]["layoutno"])
-            if TableStructureRecognizer.is_caption(self.boxes[i]) or self.boxes[i]["layout_type"] in ["table caption",
-                                                                                                      "title",
-                                                                                                      "figure caption",
-                                                                                                      "reference"]:
+            lout_no = str(self.boxes[i]["page_number"]) + "-" + str(self.boxes[i]["layoutno"])
+            if TableStructureRecognizer.is_caption(self.boxes[i]) or self.boxes[i]["layout_type"] in ["table caption", "title", "figure caption", "reference"]:
                nomerge_lout_no.append(lst_lout_no)
            if self.boxes[i]["layout_type"] == "table":
                if re.match(r"(数据|资料|图表)*来源[:： ]", self.boxes[i]["text"]):
@ -716,8 +671,7 @@ class RAGFlowPdfParser:

        # merge table on different pages
        nomerge_lout_no = set(nomerge_lout_no)
-        tbls = sorted([(k, bxs) for k, bxs in tables.items()],
-                      key=lambda x: (x[1][0]["top"], x[1][0]["x0"]))
+        tbls = sorted([(k, bxs) for k, bxs in tables.items()], key=lambda x: (x[1][0]["top"], x[1][0]["x0"]))

        i = len(tbls) - 1
        while i - 1 >= 0:
@ -758,9 +712,7 @@ class RAGFlowPdfParser:
                        if b.get("layout_type", "").find("caption") >= 0:
                            continue
                        y_dis = self._y_dis(c, b)
-                        x_dis = self._x_dis(
-                            c, b) if not x_overlapped(
-                            c, b) else 0
+                        x_dis = self._x_dis(c, b) if not x_overlapped(c, b) else 0
                        dis = y_dis * y_dis + x_dis * x_dis
                        if dis < minv:
                            mink = k
@ -774,18 +726,10 @@ class RAGFlowPdfParser:
            #    continue
            if tv < fv and tk:
                tables[tk].insert(0, c)
-                logging.debug(
-                    "TABLE:" +
-                    self.boxes[i]["text"] +
-                    "; Cap: " +
-                    tk)
+                logging.debug("TABLE:" + self.boxes[i]["text"] + "; Cap: " + tk)
            elif fk:
                figures[fk].insert(0, c)
-                logging.debug(
-                    "FIGURE:" +
-                    self.boxes[i]["text"] +
-                    "; Cap: " +
-                    tk)
+                logging.debug("FIGURE:" + self.boxes[i]["text"] + "; Cap: " + tk)
            self.boxes.pop(i)

        def cropout(bxs, ltype, poss):
@ -794,29 +738,19 @@ class RAGFlowPdfParser:
            if len(pn) < 2:
                pn = list(pn)[0]
                ht = self.page_cum_height[pn]
-                b = {
-                    "x0": np.min([b["x0"] for b in bxs]),
-                    "top": np.min([b["top"] for b in bxs]) - ht,
-                    "x1": np.max([b["x1"] for b in bxs]),
-                    "bottom": np.max([b["bottom"] for b in bxs]) - ht
-                }
+                b = {"x0": np.min([b["x0"] for b in bxs]), "top": np.min([b["top"] for b in bxs]) - ht, "x1": np.max([b["x1"] for b in bxs]), "bottom": np.max([b["bottom"] for b in bxs]) - ht}
                louts = [layout for layout in self.page_layout[pn] if layout["type"] == ltype]
                ii = Recognizer.find_overlapped(b, louts, naive=True)
                if ii is not None:
                    b = louts[ii]
                else:
-                    logging.warning(
-                        f"Missing layout match: {pn + 1},%s" %
-                        (bxs[0].get(
-                            "layoutno", "")))
+                    logging.warning(f"Missing layout match: {pn + 1},%s" % (bxs[0].get("layoutno", "")))

                left, top, right, bott = b["x0"], b["top"], b["x1"], b["bottom"]
                if right < left:
                    right = left + 1
                poss.append((pn + self.page_from, left, right, top, bott))
-                return self.page_images[pn] \
-                    .crop((left * ZM, top * ZM,
-                           right * ZM, bott * ZM))
+                return self.page_images[pn].crop((left * ZM, top * ZM, right * ZM, bott * ZM))
            pn = {}
            for b in bxs:
                p = b["page_number"] - 1
@ -825,10 +759,7 @@ class RAGFlowPdfParser:
                pn[p].append(b)
            pn = sorted(pn.items(), key=lambda x: x[0])
            imgs = [cropout(arr, ltype, poss) for p, arr in pn]
-            pic = Image.new("RGB",
-                            (int(np.max([i.size[0] for i in imgs])),
-                             int(np.sum([m.size[1] for m in imgs]))),
-                            (245, 245, 245))
+            pic = Image.new("RGB", (int(np.max([i.size[0] for i in imgs])), int(np.sum([m.size[1] for m in imgs]))), (245, 245, 245))
            height = 0
            for img in imgs:
                pic.paste(img, (0, int(height)))
@ -848,30 +779,20 @@ class RAGFlowPdfParser:
            poss = []

            if separate_tables_figures:
-                figure_results.append(
-                    (cropout(
-                        bxs,
-                        "figure", poss),
-                     [txt]))
+                figure_results.append((cropout(bxs, "figure", poss), [txt]))
                figure_positions.append(poss)
            else:
-                res.append(
-                    (cropout(
-                        bxs,
-                        "figure", poss),
-                     [txt]))
+                res.append((cropout(bxs, "figure", poss), [txt]))
                positions.append(poss)

        for k, bxs in tables.items():
            if not bxs:
                continue
-            bxs = Recognizer.sort_Y_firstly(bxs, np.mean(
-                [(b["bottom"] - b["top"]) / 2 for b in bxs]))
+            bxs = Recognizer.sort_Y_firstly(bxs, np.mean([(b["bottom"] - b["top"]) / 2 for b in bxs]))

            poss = []

-            res.append((cropout(bxs, "table", poss),
-                        self.tbl_det.construct_table(bxs, html=return_html, is_english=self.is_english)))
+            res.append((cropout(bxs, "table", poss), self.tbl_det.construct_table(bxs, html=return_html, is_english=self.is_english)))
            positions.append(poss)

        if separate_tables_figures:
@ -905,7 +826,7 @@ class RAGFlowPdfParser:
            (r"[0-9]+）", 10),
            (r"[\(（][0-9]+[）\)]", 11),
            (r"[零一二三四五六七八九十百]+是", 12),
-            (r"[⚫•➢✓]", 12)
+            (r"[⚫•➢✓]", 12),
        ]:
            if re.match(p, line):
                return j
@ -924,12 +845,9 @@ class RAGFlowPdfParser:
            if pn[-1] - 1 >= page_images_cnt:
                return ""

-        return "@@{}\t{:.1f}\t{:.1f}\t{:.1f}\t{:.1f}##" \
-            .format("-".join([str(p) for p in pn]),
-                    bx["x0"], bx["x1"], top, bott)
+        return "@@{}\t{:.1f}\t{:.1f}\t{:.1f}\t{:.1f}##".format("-".join([str(p) for p in pn]), bx["x0"], bx["x1"], top, bott)

    def __filterout_scraps(self, boxes, ZM):
-
        def width(b):
            return b["x1"] - b["x0"]

@ -939,8 +857,7 @@ class RAGFlowPdfParser:
        def usefull(b):
            if b.get("layout_type"):
                return True
-            if width(
-                    b) > self.page_images[b["page_number"] - 1].size[0] / ZM / 3:
+            if width(b) > self.page_images[b["page_number"] - 1].size[0] / ZM / 3:
                return True
            if b["bottom"] - b["top"] > self.mean_height[b["page_number"] - 1]:
                return True
@ -952,30 +869,22 @@ class RAGFlowPdfParser:
            widths = []
            pw = self.page_images[boxes[0]["page_number"] - 1].size[0] / ZM
            mh = self.mean_height[boxes[0]["page_number"] - 1]
-            mj = self.proj_match(
-                boxes[0]["text"]) or boxes[0].get(
-                "layout_type",
-                "") == "title"
+            mj = self.proj_match(boxes[0]["text"]) or boxes[0].get("layout_type", "") == "title"

            def dfs(line, st):
                nonlocal mh, pw, lines, widths
                lines.append(line)
                widths.append(width(line))
-                mmj = self.proj_match(
-                    line["text"]) or line.get(
-                    "layout_type",
-                    "") == "title"
+                mmj = self.proj_match(line["text"]) or line.get("layout_type", "") == "title"
                for i in range(st + 1, min(st + 20, len(boxes))):
                    if (boxes[i]["page_number"] - line["page_number"]) > 0:
                        break
-                    if not mmj and self._y_dis(
-                            line, boxes[i]) >= 3 * mh and height(line) < 1.5 * mh:
+                    if not mmj and self._y_dis(line, boxes[i]) >= 3 * mh and height(line) < 1.5 * mh:
                        break

                    if not usefull(boxes[i]):
                        continue
-                    if mmj or \
-                            (self._x_dis(boxes[i], line) < pw / 10): \
+                    if mmj or (self._x_dis(boxes[i], line) < pw / 10):
                        # and abs(width(boxes[i])-width_mean)/max(width(boxes[i]),width_mean)<0.5):
                        # concat following
                        dfs(boxes[i], i)
@ -992,11 +901,9 @@ class RAGFlowPdfParser:
            boxes.pop(0)
            mw = np.mean(widths)
            if mj or mw / pw >= 0.35 or mw > 200:
-                res.append(
-                    "\n".join([c["text"] + self._line_tag(c, ZM) for c in lines]))
+                res.append("\n".join([c["text"] + self._line_tag(c, ZM) for c in lines]))
            else:
-                logging.debug("REMOVED: " +
-                              "<<".join([c["text"] for c in lines]))
+                logging.debug("REMOVED: " + "<<".join([c["text"] for c in lines]))

        return "\n\n".join(res)

@ -1004,16 +911,14 @@ class RAGFlowPdfParser:
    def total_page_number(fnm, binary=None):
        try:
            with sys.modules[LOCK_KEY_pdfplumber]:
-                pdf = pdfplumber.open(
-                    fnm) if not binary else pdfplumber.open(BytesIO(binary))
+                pdf = pdfplumber.open(fnm) if not binary else pdfplumber.open(BytesIO(binary))
            total_page = len(pdf.pages)
            pdf.close()
            return total_page
        except Exception:
            logging.exception("total_page_number")

-    def __images__(self, fnm, zoomin=3, page_from=0,
-                   page_to=299, callback=None):
+    def __images__(self, fnm, zoomin=3, page_from=0, page_to=299, callback=None):
        self.lefted_chars = []
        self.mean_height = []
        self.mean_width = []
@ -1025,10 +930,9 @@ class RAGFlowPdfParser:
        start = timer()
        try:
            with sys.modules[LOCK_KEY_pdfplumber]:
-                with (pdfplumber.open(fnm) if isinstance(fnm, str) else pdfplumber.open(BytesIO(fnm))) as pdf:
+                with pdfplumber.open(fnm) if isinstance(fnm, str) else pdfplumber.open(BytesIO(fnm)) as pdf:
                    self.pdf = pdf
-                    self.page_images = [p.to_image(resolution=72 * zoomin, antialias=True).annotated for i, p in
-                                        enumerate(self.pdf.pages[page_from:page_to])]
+                    self.page_images = [p.to_image(resolution=72 * zoomin, antialias=True).annotated for i, p in enumerate(self.pdf.pages[page_from:page_to])]

                    try:
                        self.page_chars = [[c for c in page.dedupe_chars().chars if self._has_color(c)] for page in self.pdf.pages[page_from:page_to]]
@ -1044,11 +948,11 @@ class RAGFlowPdfParser:

        self.outlines = []
        try:
-            with (pdf2_read(fnm if isinstance(fnm, str)
-                            else BytesIO(fnm))) as pdf:
+            with pdf2_read(fnm if isinstance(fnm, str) else BytesIO(fnm)) as pdf:
                self.pdf = pdf

                outlines = self.pdf.outline
+
                def dfs(arr, depth):
                    for a in arr:
                        if isinstance(a, dict):
@ -1065,11 +969,11 @@ class RAGFlowPdfParser:
            logging.warning("Miss outlines")

        logging.debug("Images converted.")
-        self.is_english = [re.search(r"[a-zA-Z0-9,/¸;:'\[\]\(\)!@#$%^&*\"?<>._-]{30,}", "".join(
-            random.choices([c["text"] for c in self.page_chars[i]], k=min(100, len(self.page_chars[i]))))) for i in
-            range(len(self.page_chars))]
-        if sum([1 if e else 0 for e in self.is_english]) > len(
-                self.page_images) / 2:
+        self.is_english = [
+            re.search(r"[a-zA-Z0-9,/¸;:'\[\]\(\)!@#$%^&*\"?<>._-]{30,}", "".join(random.choices([c["text"] for c in self.page_chars[i]], k=min(100, len(self.page_chars[i])))))
+            for i in range(len(self.page_chars))
+        ]
+        if sum([1 if e else 0 for e in self.is_english]) > len(self.page_images) / 2:
            self.is_english = True
        else:
            self.is_english = False
@ -1077,10 +981,12 @@ class RAGFlowPdfParser:
        async def __img_ocr(i, id, img, chars, limiter):
            j = 0
            while j + 1 < len(chars):
-                if chars[j]["text"] and chars[j + 1]["text"] \
-                        and re.match(r"[0-9a-zA-Z,.:;!%]+", chars[j]["text"] + chars[j + 1]["text"]) \
-                        and chars[j + 1]["x0"] - chars[j]["x1"] >= min(chars[j + 1]["width"],
-                                                                       chars[j]["width"]) / 2:
+                if (
+                    chars[j]["text"]
+                    and chars[j + 1]["text"]
+                    and re.match(r"[0-9a-zA-Z,.:;!%]+", chars[j]["text"] + chars[j + 1]["text"])
+                    and chars[j + 1]["x0"] - chars[j]["x1"] >= min(chars[j + 1]["width"], chars[j]["width"]) / 2
+                ):
                    chars[j]["text"] += " "
                j += 1

@ -1096,12 +1002,8 @@ class RAGFlowPdfParser:
        async def __img_ocr_launcher():
            def __ocr_preprocess():
                chars = self.page_chars[i] if not self.is_english else []
-                self.mean_height.append(
-                    np.median(sorted([c["height"] for c in chars])) if chars else 0
-                )
-                self.mean_width.append(
-                    np.median(sorted([c["width"] for c in chars])) if chars else 8
-                )
+                self.mean_height.append(np.median(sorted([c["height"] for c in chars])) if chars else 0)
+                self.mean_width.append(np.median(sorted([c["width"] for c in chars])) if chars else 8)
                self.page_cum_height.append(img.size[1] / zoomin)
                return chars

@ -1110,8 +1012,7 @@ class RAGFlowPdfParser:
                    for i, img in enumerate(self.page_images):
                        chars = __ocr_preprocess()

-                        nursery.start_soon(__img_ocr, i, i % PARALLEL_DEVICES, img, chars,
-                                           self.parallel_limiter[i % PARALLEL_DEVICES])
+                        nursery.start_soon(__img_ocr, i, i % PARALLEL_DEVICES, img, chars, self.parallel_limiter[i % PARALLEL_DEVICES])
                        await trio.sleep(0.1)
            else:
                for i, img in enumerate(self.page_images):
@ -1124,11 +1025,9 @@ class RAGFlowPdfParser:

        logging.info(f"__images__ {len(self.page_images)} pages cost {timer() - start}s")

-        if not self.is_english and not any(
-                [c for c in self.page_chars]) and self.boxes:
+        if not self.is_english and not any([c for c in self.page_chars]) and self.boxes:
            bxes = [b for bxs in self.boxes for b in bxs]
-            self.is_english = re.search(r"[\na-zA-Z0-9,/¸;:'\[\]\(\)!@#$%^&*\"?<>._-]{30,}",
-                                        "".join([b["text"] for b in random.choices(bxes, k=min(30, len(bxes)))]))
+            self.is_english = re.search(r"[\na-zA-Z0-9,/¸;:'\[\]\(\)!@#$%^&*\"?<>._-]{30,}", "".join([b["text"] for b in random.choices(bxes, k=min(30, len(bxes)))]))

        logging.debug("Is it English:", self.is_english)

@ -1144,8 +1043,7 @@ class RAGFlowPdfParser:
        self._text_merge()
        self._concat_downward()
        self._filter_forpages()
-        tbls = self._extract_table_figure(
-            need_image, zoomin, return_html, False)
+        tbls = self._extract_table_figure(need_image, zoomin, return_html, False)
        return self.__filterout_scraps(deepcopy(self.boxes), zoomin), tbls

    def parse_into_bboxes(self, fnm, callback=None, zoomin=3):
@ -1179,9 +1077,8 @@ class RAGFlowPdfParser:
                import math
                pn1, left1, right1, top1, bottom1 = rect1
                pn2, left2, right2, top2, bottom2 = rect2
-                if (right1 >= left2 and right2 >= left1 and
-                        bottom1 >= top2 and bottom2 >= top1):
-                    return 0 + (pn1-pn2)*10000
+                if right1 >= left2 and right2 >= left1 and bottom1 >= top2 and bottom2 >= top1:
+                    return 0
                if right1 < left2:
                    dx = left2 - right1
                elif right2 < left1:
@ -1194,17 +1091,20 @@ class RAGFlowPdfParser:
                    dy = top1 - bottom2
                else:
                    dy = 0
-                return math.sqrt(dx*dx + dy*dy) + (pn1-pn2)*10000
+                return math.sqrt(dx*dx + dy*dy)# + (pn2-pn1)*10000

            for (img, txt), poss in tbls_or_figs:
                bboxes = [(i, (b["page_number"], b["x0"], b["x1"], b["top"], b["bottom"])) for i, b in enumerate(self.boxes)]
-                dists = [(min_rectangle_distance((pn, left, right, top, bott), rect),i) for i, rect in bboxes for pn, left, right, top, bott in poss]
+                dists = [(min_rectangle_distance((pn, left, right, top+self.page_cum_height[pn], bott+self.page_cum_height[pn]), rect),i) for i, rect in bboxes for pn, left, right, top, bott in poss]
                min_i = np.argmin(dists, axis=0)[0]
                min_i, rect = bboxes[dists[min_i][-1]]
                if isinstance(txt, list):
                    txt = "\n".join(txt)
+                pn, left, right, top, bott = poss[0]
+                if self.boxes[min_i]["bottom"] < top+self.page_cum_height[pn]:
+                    min_i += 1
                self.boxes.insert(min_i, {
-                    "page_number": rect[0], "x0": rect[1], "x1": rect[2], "top": rect[3], "bottom": rect[4], "layout_type": layout_type, "text": txt, "image": img
+                    "page_number": pn+1, "x0": left, "x1": right, "top": top+self.page_cum_height[pn], "bottom": bott+self.page_cum_height[pn], "layout_type": layout_type, "text": txt, "image": img
                })

        for b in self.boxes:
@ -1225,12 +1125,9 @@ class RAGFlowPdfParser:
    def extract_positions(txt):
        poss = []
        for tag in re.findall(r"@@[0-9-]+\t[0-9.\t]+##", txt):
-            pn, left, right, top, bottom = tag.strip(
-                "#").strip("@").split("\t")
-            left, right, top, bottom = float(left), float(
-                right), float(top), float(bottom)
-            poss.append(([int(p) - 1 for p in pn.split("-")],
-                         left, right, top, bottom))
+            pn, left, right, top, bottom = tag.strip("#").strip("@").split("\t")
+            left, right, top, bottom = float(left), float(right), float(top), float(bottom)
+            poss.append(([int(p) - 1 for p in pn.split("-")], left, right, top, bottom))
        return poss

    def crop(self, text, ZM=3, need_position=False):
@ -1241,15 +1138,12 @@ class RAGFlowPdfParser:
                return None, None
            return

-        max_width = max(
-            np.max([right - left for (_, left, right, _, _) in poss]), 6)
+        max_width = max(np.max([right - left for (_, left, right, _, _) in poss]), 6)
        GAP = 6
        pos = poss[0]
-        poss.insert(0, ([pos[0][0]], pos[1], pos[2], max(
-            0, pos[3] - 120), max(pos[3] - GAP, 0)))
+        poss.insert(0, ([pos[0][0]], pos[1], pos[2], max(0, pos[3] - 120), max(pos[3] - GAP, 0)))
        pos = poss[-1]
-        poss.append(([pos[0][-1]], pos[1], pos[2], min(self.page_images[pos[0][-1]].size[1] / ZM, pos[4] + GAP),
-                     min(self.page_images[pos[0][-1]].size[1] / ZM, pos[4] + 120)))
+        poss.append(([pos[0][-1]], pos[1], pos[2], min(self.page_images[pos[0][-1]].size[1] / ZM, pos[4] + GAP), min(self.page_images[pos[0][-1]].size[1] / ZM, pos[4] + 120)))

        positions = []
        for ii, (pns, left, right, top, bottom) in enumerate(poss):
@ -1257,28 +1151,14 @@ class RAGFlowPdfParser:
            bottom *= ZM
            for pn in pns[1:]:
                bottom += self.page_images[pn - 1].size[1]
-            imgs.append(
-                self.page_images[pns[0]].crop((left * ZM, top * ZM,
-                                               right *
-                                               ZM, min(
-                                                   bottom, self.page_images[pns[0]].size[1])
-                                               ))
-            )
+            imgs.append(self.page_images[pns[0]].crop((left * ZM, top * ZM, right * ZM, min(bottom, self.page_images[pns[0]].size[1]))))
            if 0 < ii < len(poss) - 1:
-                positions.append((pns[0] + self.page_from, left, right, top, min(
-                    bottom, self.page_images[pns[0]].size[1]) / ZM))
+                positions.append((pns[0] + self.page_from, left, right, top, min(bottom, self.page_images[pns[0]].size[1]) / ZM))
            bottom -= self.page_images[pns[0]].size[1]
            for pn in pns[1:]:
-                imgs.append(
-                    self.page_images[pn].crop((left * ZM, 0,
-                                               right * ZM,
-                                               min(bottom,
-                                                   self.page_images[pn].size[1])
-                                               ))
-                )
+                imgs.append(self.page_images[pn].crop((left * ZM, 0, right * ZM, min(bottom, self.page_images[pn].size[1]))))
                if 0 < ii < len(poss) - 1:
-                    positions.append((pn + self.page_from, left, right, 0, min(
-                        bottom, self.page_images[pn].size[1]) / ZM))
+                    positions.append((pn + self.page_from, left, right, 0, min(bottom, self.page_images[pn].size[1]) / ZM))
                bottom -= self.page_images[pn].size[1]

        if not imgs:
@ -1290,14 +1170,12 @@ class RAGFlowPdfParser:
            height += img.size[1] + GAP
        height = int(height)
        width = int(np.max([i.size[0] for i in imgs]))
-        pic = Image.new("RGB",
-                        (width, height),
-                        (245, 245, 245))
+        pic = Image.new("RGB", (width, height), (245, 245, 245))
        height = 0
        for ii, img in enumerate(imgs):
            if ii == 0 or ii + 1 == len(imgs):
-                img = img.convert('RGBA')
-                overlay = Image.new('RGBA', img.size, (0, 0, 0, 0))
+                img = img.convert("RGBA")
+                overlay = Image.new("RGBA", img.size, (0, 0, 0, 0))
                overlay.putalpha(128)
                img = Image.alpha_composite(img, overlay).convert("RGB")
            pic.paste(img, (0, int(height)))
@ -1312,14 +1190,12 @@ class RAGFlowPdfParser:
        pn = bx["page_number"]
        top = bx["top"] - self.page_cum_height[pn - 1]
        bott = bx["bottom"] - self.page_cum_height[pn - 1]
-        poss.append((pn, bx["x0"], bx["x1"], top, min(
-            bott, self.page_images[pn - 1].size[1] / ZM)))
+        poss.append((pn, bx["x0"], bx["x1"], top, min(bott, self.page_images[pn - 1].size[1] / ZM)))
        while bott * ZM > self.page_images[pn - 1].size[1]:
            bott -= self.page_images[pn - 1].size[1] / ZM
            top = 0
            pn += 1
-            poss.append((pn, bx["x0"], bx["x1"], top, min(
-                bott, self.page_images[pn - 1].size[1] / ZM)))
+            poss.append((pn, bx["x0"], bx["x1"], top, min(bott, self.page_images[pn - 1].size[1] / ZM)))
        return poss


@ -1328,9 +1204,7 @@ class PlainParser:
        self.outlines = []
        lines = []
        try:
-            self.pdf = pdf2_read(
-                filename if isinstance(
-                    filename, str) else BytesIO(filename))
+            self.pdf = pdf2_read(filename if isinstance(filename, str) else BytesIO(filename))
            for page in self.pdf.pages[from_page:to_page]:
                lines.extend([t for t in page.extract_text().split("\n")])

@ -1367,10 +1241,8 @@ class VisionParser(RAGFlowPdfParser):
    def __images__(self, fnm, zoomin=3, page_from=0, page_to=299, callback=None):
        try:
            with sys.modules[LOCK_KEY_pdfplumber]:
-                self.pdf = pdfplumber.open(fnm) if isinstance(
-                    fnm, str) else pdfplumber.open(BytesIO(fnm))
-                self.page_images = [p.to_image(resolution=72 * zoomin).annotated for i, p in
-                                    enumerate(self.pdf.pages[page_from:page_to])]
+                self.pdf = pdfplumber.open(fnm) if isinstance(fnm, str) else pdfplumber.open(BytesIO(fnm))
+                self.page_images = [p.to_image(resolution=72 * zoomin).annotated for i, p in enumerate(self.pdf.pages[page_from:page_to])]
                self.total_page = len(self.pdf.pages)
        except Exception:
            self.page_images = None
@ -1401,7 +1273,7 @@ class VisionParser(RAGFlowPdfParser):
                callback=callback,
            )
            if kwargs.get("callback"):
-                kwargs["callback"](idx*1./len(self.page_images), f"Processed: {idx+1}/{len(self.page_images)}")
+                kwargs["callback"](idx * 1.0 / len(self.page_images), f"Processed: {idx + 1}/{len(self.page_images)}")

            if text:
                width, height = self.page_images[idx].size
--- a/deepdoc/vision/init.py
+++ b/deepdoc/vision/init.py
@ -16,24 +16,28 @@
 import io
 import sys
 import threading
+
 import pdfplumber

 from .ocr import OCR
 from .recognizer import Recognizer
+from .layout_recognizer import AscendLayoutRecognizer
 from .layout_recognizer import LayoutRecognizer4YOLOv10 as LayoutRecognizer
 from .table_structure_recognizer import TableStructureRecognizer

-
 LOCK_KEY_pdfplumber = "global_shared_lock_pdfplumber"
 if LOCK_KEY_pdfplumber not in sys.modules:
    sys.modules[LOCK_KEY_pdfplumber] = threading.Lock()


 def init_in_out(args):
-    from PIL import Image
    import os
    import traceback
+
+    from PIL import Image
+
    from api.utils.file_utils import traversal_files
+
    images = []
    outputs = []

@ -44,8 +48,7 @@ def init_in_out(args):
        nonlocal outputs, images
        with sys.modules[LOCK_KEY_pdfplumber]:
            pdf = pdfplumber.open(fnm)
-            images = [p.to_image(resolution=72 * zoomin).annotated for i, p in
-                                enumerate(pdf.pages)]
+            images = [p.to_image(resolution=72 * zoomin).annotated for i, p in enumerate(pdf.pages)]

        for i, page in enumerate(images):
            outputs.append(os.path.split(fnm)[-1] + f"_{i}.jpg")
@ -57,10 +60,10 @@ def init_in_out(args):
            pdf_pages(fnm)
            return
        try:
-            fp = open(fnm, 'rb')
+            fp = open(fnm, "rb")
            binary = fp.read()
            fp.close()
-            images.append(Image.open(io.BytesIO(binary)).convert('RGB'))
+            images.append(Image.open(io.BytesIO(binary)).convert("RGB"))
            outputs.append(os.path.split(fnm)[-1])
        except Exception:
            traceback.print_exc()
@ -81,6 +84,7 @@ __all__ = [
    "OCR",
    "Recognizer",
    "LayoutRecognizer",
+    "AscendLayoutRecognizer",
    "TableStructureRecognizer",
    "init_in_out",
 ]
--- a/deepdoc/vision/layout_recognizer.py
+++ b/deepdoc/vision/layout_recognizer.py
@ -14,6 +14,8 @@
 #  limitations under the License.
 #

+import logging
+import math
 import os
 import re
 from collections import Counter
@ -45,28 +47,22 @@ class LayoutRecognizer(Recognizer):

    def __init__(self, domain):
        try:
-            model_dir = os.path.join(
-                get_project_base_directory(),
-                "rag/res/deepdoc")
+            model_dir = os.path.join(get_project_base_directory(), "rag/res/deepdoc")
            super().__init__(self.labels, domain, model_dir)
        except Exception:
-            model_dir = snapshot_download(repo_id="InfiniFlow/deepdoc",
-                                          local_dir=os.path.join(get_project_base_directory(), "rag/res/deepdoc"),
-                                          local_dir_use_symlinks=False)
+            model_dir = snapshot_download(repo_id="InfiniFlow/deepdoc", local_dir=os.path.join(get_project_base_directory(), "rag/res/deepdoc"), local_dir_use_symlinks=False)
            super().__init__(self.labels, domain, model_dir)

        self.garbage_layouts = ["footer", "header", "reference"]
        self.client = None
        if os.environ.get("TENSORRT_DLA_SVR"):
            from deepdoc.vision.dla_cli import DLAClient
+
            self.client = DLAClient(os.environ["TENSORRT_DLA_SVR"])

    def __call__(self, image_list, ocr_res, scale_factor=3, thr=0.2, batch_size=16, drop=True):
        def __is_garbage(b):
-            patt = [r"^•+$", "^[0-9]{1,2} / ?[0-9]{1,2}$",
-                    r"^[0-9]{1,2} of [0-9]{1,2}$", "^http://[^ ]{12,}",
-                    "\\(cid *: *[0-9]+ *\\)"
-                    ]
+            patt = [r"^•+$", "^[0-9]{1,2} / ?[0-9]{1,2}$", r"^[0-9]{1,2} of [0-9]{1,2}$", "^http://[^ ]{12,}", "\\(cid *: *[0-9]+ *\\)"]
            return any([re.search(p, b["text"]) for p in patt])

        if self.client:
@ -82,18 +78,23 @@ class LayoutRecognizer(Recognizer):
        page_layout = []
        for pn, lts in enumerate(layouts):
            bxs = ocr_res[pn]
-            lts = [{"type": b["type"],
+            lts = [
+                {
+                    "type": b["type"],
                    "score": float(b["score"]),
-                    "x0": b["bbox"][0] / scale_factor, "x1": b["bbox"][2] / scale_factor,
-                    "top": b["bbox"][1] / scale_factor, "bottom": b["bbox"][-1] / scale_factor,
+                    "x0": b["bbox"][0] / scale_factor,
+                    "x1": b["bbox"][2] / scale_factor,
+                    "top": b["bbox"][1] / scale_factor,
+                    "bottom": b["bbox"][-1] / scale_factor,
                    "page_number": pn,
-                    } for b in lts if float(b["score"]) >= 0.4 or b["type"] not in self.garbage_layouts]
-            lts = self.sort_Y_firstly(lts, np.mean(
-                [lt["bottom"] - lt["top"] for lt in lts]) / 2)
+                }
+                for b in lts
+                if float(b["score"]) >= 0.4 or b["type"] not in self.garbage_layouts
+            ]
+            lts = self.sort_Y_firstly(lts, np.mean([lt["bottom"] - lt["top"] for lt in lts]) / 2)
            lts = self.layouts_cleanup(bxs, lts)
            page_layout.append(lts)

-            # Tag layout type, layouts are ready
            def findLayout(ty):
                nonlocal bxs, lts, self
                lts_ = [lt for lt in lts if lt["type"] == ty]
@ -106,21 +107,17 @@ class LayoutRecognizer(Recognizer):
                        bxs.pop(i)
                        continue

-                    ii = self.find_overlapped_with_threshold(bxs[i], lts_,
-                                                              thr=0.4)
-                    if ii is None:  # belong to nothing
+                    ii = self.find_overlapped_with_threshold(bxs[i], lts_, thr=0.4)
+                    if ii is None:
                        bxs[i]["layout_type"] = ""
                        i += 1
                        continue
                    lts_[ii]["visited"] = True
                    keep_feats = [
-                        lts_[
-                            ii]["type"] == "footer" and bxs[i]["bottom"] < image_list[pn].size[1] * 0.9 / scale_factor,
-                        lts_[
-                            ii]["type"] == "header" and bxs[i]["top"] > image_list[pn].size[1] * 0.1 / scale_factor,
+                        lts_[ii]["type"] == "footer" and bxs[i]["bottom"] < image_list[pn].size[1] * 0.9 / scale_factor,
+                        lts_[ii]["type"] == "header" and bxs[i]["top"] > image_list[pn].size[1] * 0.1 / scale_factor,
                    ]
-                    if drop and lts_[
-                            ii]["type"] in self.garbage_layouts and not any(keep_feats):
+                    if drop and lts_[ii]["type"] in self.garbage_layouts and not any(keep_feats):
                        if lts_[ii]["type"] not in garbages:
                            garbages[lts_[ii]["type"]] = []
                        garbages[lts_[ii]["type"]].append(bxs[i]["text"])
@ -128,17 +125,14 @@ class LayoutRecognizer(Recognizer):
                        continue

                    bxs[i]["layoutno"] = f"{ty}-{ii}"
-                    bxs[i]["layout_type"] = lts_[ii]["type"] if lts_[
-                        ii]["type"] != "equation" else "figure"
+                    bxs[i]["layout_type"] = lts_[ii]["type"] if lts_[ii]["type"] != "equation" else "figure"
                    i += 1

-            for lt in ["footer", "header", "reference", "figure caption",
-                       "table caption", "title", "table", "text", "figure", "equation"]:
+            for lt in ["footer", "header", "reference", "figure caption", "table caption", "title", "table", "text", "figure", "equation"]:
                findLayout(lt)

            # add box to figure layouts which has not text box
-            for i, lt in enumerate(
-                    [lt for lt in lts if lt["type"] in ["figure", "equation"]]):
+            for i, lt in enumerate([lt for lt in lts if lt["type"] in ["figure", "equation"]]):
                if lt.get("visited"):
                    continue
                lt = deepcopy(lt)
@ -206,9 +200,7 @@ class LayoutRecognizer4YOLOv10(LayoutRecognizer):
            img = cv2.resize(img, new_unpad, interpolation=cv2.INTER_LINEAR)
            top, bottom = int(round(dh - 0.1)) if self.center else 0, int(round(dh + 0.1))
            left, right = int(round(dw - 0.1)) if self.center else 0, int(round(dw + 0.1))
-            img = cv2.copyMakeBorder(
-                img, top, bottom, left, right, cv2.BORDER_CONSTANT, value=(114, 114, 114)
-            )  # add border
+            img = cv2.copyMakeBorder(img, top, bottom, left, right, cv2.BORDER_CONSTANT, value=(114, 114, 114))  # add border
            img /= 255.0
            img = img.transpose(2, 0, 1)
            img = img[np.newaxis, :, :, :].astype(np.float32)
@ -230,8 +222,7 @@ class LayoutRecognizer4YOLOv10(LayoutRecognizer):
        boxes[:, 2] -= inputs["scale_factor"][2]
        boxes[:, 1] -= inputs["scale_factor"][3]
        boxes[:, 3] -= inputs["scale_factor"][3]
-        input_shape = np.array([inputs["scale_factor"][0], inputs["scale_factor"][1], inputs["scale_factor"][0],
-                                inputs["scale_factor"][1]])
+        input_shape = np.array([inputs["scale_factor"][0], inputs["scale_factor"][1], inputs["scale_factor"][0], inputs["scale_factor"][1]])
        boxes = np.multiply(boxes, input_shape, dtype=np.float32)

        unique_class_ids = np.unique(class_ids)
@ -243,8 +234,223 @@ class LayoutRecognizer4YOLOv10(LayoutRecognizer):
            class_keep_boxes = nms(class_boxes, class_scores, 0.45)
            indices.extend(class_indices[class_keep_boxes])

-        return [{
-            "type": self.label_list[class_ids[i]].lower(),
-            "bbox": [float(t) for t in boxes[i].tolist()],
-            "score": float(scores[i])
-        } for i in indices]
+        return [{"type": self.label_list[class_ids[i]].lower(), "bbox": [float(t) for t in boxes[i].tolist()], "score": float(scores[i])} for i in indices]
+
+
+class AscendLayoutRecognizer(Recognizer):
+    labels = [
+        "title",
+        "Text",
+        "Reference",
+        "Figure",
+        "Figure caption",
+        "Table",
+        "Table caption",
+        "Table caption",
+        "Equation",
+        "Figure caption",
+    ]
+
+    def __init__(self, domain):
+        from ais_bench.infer.interface import InferSession
+
+        model_dir = os.path.join(get_project_base_directory(), "rag/res/deepdoc")
+        model_file_path = os.path.join(model_dir, domain + ".om")
+
+        if not os.path.exists(model_file_path):
+            raise ValueError(f"Model file not found: {model_file_path}")
+
+        device_id = int(os.getenv("ASCEND_LAYOUT_RECOGNIZER_DEVICE_ID", 0))
+        self.session = InferSession(device_id=device_id, model_path=model_file_path)
+        self.input_shape = self.session.get_inputs()[0].shape[2:4]  # H,W
+        self.garbage_layouts = ["footer", "header", "reference"]
+
+    def preprocess(self, image_list):
+        inputs = []
+        H, W = self.input_shape
+        for img in image_list:
+            h, w = img.shape[:2]
+            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB).astype(np.float32)
+
+            r = min(H / h, W / w)
+            new_unpad = (int(round(w * r)), int(round(h * r)))
+            dw, dh = (W - new_unpad[0]) / 2.0, (H - new_unpad[1]) / 2.0
+
+            img = cv2.resize(img, new_unpad, interpolation=cv2.INTER_LINEAR)
+            top, bottom = int(round(dh - 0.1)), int(round(dh + 0.1))
+            left, right = int(round(dw - 0.1)), int(round(dw + 0.1))
+            img = cv2.copyMakeBorder(img, top, bottom, left, right, cv2.BORDER_CONSTANT, value=(114, 114, 114))
+
+            img /= 255.0
+            img = img.transpose(2, 0, 1)[np.newaxis, :, :, :].astype(np.float32)
+
+            inputs.append(
+                {
+                    "image": img,
+                    "scale_factor": [w / new_unpad[0], h / new_unpad[1]],
+                    "pad": [dw, dh],
+                    "orig_shape": [h, w],
+                }
+            )
+        return inputs
+
+    def postprocess(self, boxes, inputs, thr=0.25):
+        arr = np.squeeze(boxes)
+        if arr.ndim == 1:
+            arr = arr.reshape(1, -1)
+
+        results = []
+        if arr.shape[1] == 6:
+            # [x1,y1,x2,y2,score,cls]
+            m = arr[:, 4] >= thr
+            arr = arr[m]
+            if arr.size == 0:
+                return []
+            xyxy = arr[:, :4].astype(np.float32)
+            scores = arr[:, 4].astype(np.float32)
+            cls_ids = arr[:, 5].astype(np.int32)
+
+            if "pad" in inputs:
+                dw, dh = inputs["pad"]
+                sx, sy = inputs["scale_factor"]
+                xyxy[:, [0, 2]] -= dw
+                xyxy[:, [1, 3]] -= dh
+                xyxy *= np.array([sx, sy, sx, sy], dtype=np.float32)
+            else:
+                # backup
+                sx, sy = inputs["scale_factor"]
+                xyxy *= np.array([sx, sy, sx, sy], dtype=np.float32)
+
+            keep_indices = []
+            for c in np.unique(cls_ids):
+                idx = np.where(cls_ids == c)[0]
+                k = nms(xyxy[idx], scores[idx], 0.45)
+                keep_indices.extend(idx[k])
+
+            for i in keep_indices:
+                cid = int(cls_ids[i])
+                if 0 <= cid < len(self.labels):
+                    results.append({"type": self.labels[cid].lower(), "bbox": [float(t) for t in xyxy[i].tolist()], "score": float(scores[i])})
+            return results
+
+        raise ValueError(f"Unexpected output shape: {arr.shape}")
+
+    def __call__(self, image_list, ocr_res, scale_factor=3, thr=0.2, batch_size=16, drop=True):
+        import re
+        from collections import Counter
+
+        assert len(image_list) == len(ocr_res)
+
+        images = [np.array(im) if not isinstance(im, np.ndarray) else im for im in image_list]
+        layouts_all_pages = []  # list of list[{"type","score","bbox":[x1,y1,x2,y2]}]
+
+        conf_thr = max(thr, 0.08)
+
+        batch_loop_cnt = math.ceil(float(len(images)) / batch_size)
+        for bi in range(batch_loop_cnt):
+            s = bi * batch_size
+            e = min((bi + 1) * batch_size, len(images))
+            batch_images = images[s:e]
+
+            inputs_list = self.preprocess(batch_images)
+            logging.debug("preprocess done")
+
+            for ins in inputs_list:
+                feeds = [ins["image"]]
+                out_list = self.session.infer(feeds=feeds, mode="static")
+
+                for out in out_list:
+                    lts = self.postprocess(out, ins, conf_thr)
+
+                    page_lts = []
+                    for b in lts:
+                        if float(b["score"]) >= 0.4 or b["type"] not in self.garbage_layouts:
+                            x0, y0, x1, y1 = b["bbox"]
+                            page_lts.append(
+                                {
+                                    "type": b["type"],
+                                    "score": float(b["score"]),
+                                    "x0": float(x0) / scale_factor,
+                                    "x1": float(x1) / scale_factor,
+                                    "top": float(y0) / scale_factor,
+                                    "bottom": float(y1) / scale_factor,
+                                    "page_number": len(layouts_all_pages),
+                                }
+                            )
+                    layouts_all_pages.append(page_lts)
+
+        def _is_garbage_text(box):
+            patt = [r"^•+$", r"^[0-9]{1,2} / ?[0-9]{1,2}$", r"^[0-9]{1,2} of [0-9]{1,2}$", r"^http://[^ ]{12,}", r"\(cid *: *[0-9]+ *\)"]
+            return any(re.search(p, box.get("text", "")) for p in patt)
+
+        boxes_out = []
+        page_layout = []
+        garbages = {}
+
+        for pn, lts in enumerate(layouts_all_pages):
+            if lts:
+                avg_h = np.mean([lt["bottom"] - lt["top"] for lt in lts])
+                lts = self.sort_Y_firstly(lts, avg_h / 2 if avg_h > 0 else 0)
+
+            bxs = ocr_res[pn]
+            lts = self.layouts_cleanup(bxs, lts)
+            page_layout.append(lts)
+
+            def _tag_layout(ty):
+                nonlocal bxs, lts
+                lts_of_ty = [lt for lt in lts if lt["type"] == ty]
+                i = 0
+                while i < len(bxs):
+                    if bxs[i].get("layout_type"):
+                        i += 1
+                        continue
+                    if _is_garbage_text(bxs[i]):
+                        bxs.pop(i)
+                        continue
+
+                    ii = self.find_overlapped_with_threshold(bxs[i], lts_of_ty, thr=0.4)
+                    if ii is None:
+                        bxs[i]["layout_type"] = ""
+                        i += 1
+                        continue
+
+                    lts_of_ty[ii]["visited"] = True
+
+                    keep_feats = [
+                        lts_of_ty[ii]["type"] == "footer" and bxs[i]["bottom"] < image_list[pn].shape[0] * 0.9 / scale_factor,
+                        lts_of_ty[ii]["type"] == "header" and bxs[i]["top"] > image_list[pn].shape[0] * 0.1 / scale_factor,
+                    ]
+                    if drop and lts_of_ty[ii]["type"] in self.garbage_layouts and not any(keep_feats):
+                        garbages.setdefault(lts_of_ty[ii]["type"], []).append(bxs[i].get("text", ""))
+                        bxs.pop(i)
+                        continue
+
+                    bxs[i]["layoutno"] = f"{ty}-{ii}"
+                    bxs[i]["layout_type"] = lts_of_ty[ii]["type"] if lts_of_ty[ii]["type"] != "equation" else "figure"
+                    i += 1
+
+            for ty in ["footer", "header", "reference", "figure caption", "table caption", "title", "table", "text", "figure", "equation"]:
+                _tag_layout(ty)
+
+            figs = [lt for lt in lts if lt["type"] in ["figure", "equation"]]
+            for i, lt in enumerate(figs):
+                if lt.get("visited"):
+                    continue
+                lt = deepcopy(lt)
+                lt.pop("type", None)
+                lt["text"] = ""
+                lt["layout_type"] = "figure"
+                lt["layoutno"] = f"figure-{i}"
+                bxs.append(lt)
+
+            boxes_out.extend(bxs)
+
+        garbag_set = set()
+        for k, lst in garbages.items():
+            cnt = Counter(lst)
+            for g, c in cnt.items():
+                if c > 1:
+                    garbag_set.add(g)
+
+        ocr_res_new = [b for b in boxes_out if b["text"].strip() not in garbag_set]
+        return ocr_res_new, page_layout
--- a/deepdoc/vision/ocr.py
+++ b/deepdoc/vision/ocr.py
@ -13,7 +13,7 @@
 #  See the License for the specific language governing permissions and
 #  limitations under the License.
 #
-
+import gc
 import logging
 import copy
 import time
@ -348,6 +348,13 @@ class TextRecognizer:

        return img

+    def close(self):
+        # close session and release manually
+        logging.info('Close TextRecognizer.')
+        if hasattr(self, "predictor"):
+            del self.predictor
+        gc.collect()
+
    def __call__(self, img_list):
        img_num = len(img_list)
        # Calculate the aspect ratio of all text bars
@ -395,6 +402,9 @@ class TextRecognizer:

        return rec_res, time.time() - st

+    def __del__(self):
+        self.close()
+

 class TextDetector:
    def __init__(self, model_dir, device_id: int | None = None):
@ -479,6 +489,12 @@ class TextDetector:
        dt_boxes = np.array(dt_boxes_new)
        return dt_boxes

+    def close(self):
+        logging.info("Close TextDetector.")
+        if hasattr(self, "predictor"):
+            del self.predictor
+        gc.collect()
+
    def __call__(self, img):
        ori_im = img.copy()
        data = {'image': img}
@ -508,6 +524,9 @@ class TextDetector:

        return dt_boxes, time.time() - st

+    def __del__(self):
+        self.close()
+

 class OCR:
    def __init__(self, model_dir=None):
--- a/deepdoc/vision/recognizer.py
+++ b/deepdoc/vision/recognizer.py
@ -13,7 +13,7 @@
 #  See the License for the specific language governing permissions and
 #  limitations under the License.
 #
-
+import gc
 import logging
 import os
 import math
@ -406,6 +406,12 @@ class Recognizer:
            "score": float(scores[i])
        } for i in indices]

+    def close(self):
+        logging.info("Close recognizer.")
+        if hasattr(self, "ort_sess"):
+            del self.ort_sess
+        gc.collect()
+
    def __call__(self, image_list, thr=0.7, batch_size=16):
        res = []
        images = []
@ -430,5 +436,7 @@ class Recognizer:

        return res

+    def __del__(self):
+        self.close()


--- a/deepdoc/vision/table_structure_recognizer.py
+++ b/deepdoc/vision/table_structure_recognizer.py
@ -23,6 +23,7 @@ from huggingface_hub import snapshot_download

 from api.utils.file_utils import get_project_base_directory
 from rag.nlp import rag_tokenizer
+
 from .recognizer import Recognizer


@ -38,31 +39,49 @@ class TableStructureRecognizer(Recognizer):

    def __init__(self):
        try:
-            super().__init__(self.labels, "tsr", os.path.join(
-                    get_project_base_directory(),
-                    "rag/res/deepdoc"))
+            super().__init__(self.labels, "tsr", os.path.join(get_project_base_directory(), "rag/res/deepdoc"))
        except Exception:
-            super().__init__(self.labels, "tsr", snapshot_download(repo_id="InfiniFlow/deepdoc",
+            super().__init__(
+                self.labels,
+                "tsr",
+                snapshot_download(
+                    repo_id="InfiniFlow/deepdoc",
                    local_dir=os.path.join(get_project_base_directory(), "rag/res/deepdoc"),
-                                              local_dir_use_symlinks=False))
+                    local_dir_use_symlinks=False,
+                ),
+            )

    def __call__(self, images, thr=0.2):
+        table_structure_recognizer_type = os.getenv("TABLE_STRUCTURE_RECOGNIZER_TYPE", "onnx").lower()
+        if table_structure_recognizer_type not in ["onnx", "ascend"]:
+            raise RuntimeError("Unsupported table structure recognizer type.")
+
+        if table_structure_recognizer_type == "onnx":
+            logging.debug("Using Onnx table structure recognizer", flush=True)
            tbls = super().__call__(images, thr)
+        else:  # ascend
+            logging.debug("Using Ascend table structure recognizer", flush=True)
+            tbls = self._run_ascend_tsr(images, thr)
+
        res = []
        # align left&right for rows, align top&bottom for columns
        for tbl in tbls:
-            lts = [{"label": b["type"],
+            lts = [
+                {
+                    "label": b["type"],
                    "score": b["score"],
-                    "x0": b["bbox"][0], "x1": b["bbox"][2],
-                    "top": b["bbox"][1], "bottom": b["bbox"][-1]
-                    } for b in tbl]
+                    "x0": b["bbox"][0],
+                    "x1": b["bbox"][2],
+                    "top": b["bbox"][1],
+                    "bottom": b["bbox"][-1],
+                }
+                for b in tbl
+            ]
            if not lts:
                continue

-            left = [b["x0"] for b in lts if b["label"].find(
-                "row") > 0 or b["label"].find("header") > 0]
-            right = [b["x1"] for b in lts if b["label"].find(
-                "row") > 0 or b["label"].find("header") > 0]
+            left = [b["x0"] for b in lts if b["label"].find("row") > 0 or b["label"].find("header") > 0]
+            right = [b["x1"] for b in lts if b["label"].find("row") > 0 or b["label"].find("header") > 0]
            if not left:
                continue
            left = np.mean(left) if len(left) > 4 else np.min(left)
@ -93,11 +112,8 @@ class TableStructureRecognizer(Recognizer):

    @staticmethod
    def is_caption(bx):
-        patt = [
-            r"[图表]+[ 0-9:：]{2,}"
-        ]
-        if any([re.match(p, bx["text"].strip()) for p in patt]) \
-                or bx.get("layout_type", "").find("caption") >= 0:
+        patt = [r"[图表]+[ 0-9:：]{2,}"]
+        if any([re.match(p, bx["text"].strip()) for p in patt]) or bx.get("layout_type", "").find("caption") >= 0:
            return True
        return False

@ -115,7 +131,7 @@ class TableStructureRecognizer(Recognizer):
            (r"^[0-9A-Z/\._~-]+$", "Ca"),
            (r"^[A-Z]*[a-z' -]+$", "En"),
            (r"^[0-9.,+-]+[0-9A-Za-z/$￥%<>（）()' -]+$", "NE"),
-            (r"^.{1}$", "Sg")
+            (r"^.{1}$", "Sg"),
        ]
        for p, n in patt:
            if re.search(p, b["text"].strip()):
@ -163,14 +179,12 @@ class TableStructureRecognizer(Recognizer):
        for b in boxes[1:]:
            b["rn"] = len(rows) - 1
            lst_r = rows[-1]
-            if lst_r[-1].get("R", "") != b.get("R", "") \
-                    or (b["top"] >= btm - 3 and lst_r[-1].get("R", "-1") != b.get("R", "-2")
-                        ):  # new row
+            if lst_r[-1].get("R", "") != b.get("R", "") or (b["top"] >= btm - 3 and lst_r[-1].get("R", "-1") != b.get("R", "-2")):  # new row
                btm = b["bottom"]
                b["rn"] += 1
                rows.append([b])
                continue
-            btm = (btm + b["bottom"]) / 2.
+            btm = (btm + b["bottom"]) / 2.0
            rows[-1].append(b)

        colwm = [b["C_right"] - b["C_left"] for b in boxes if "C" in b]
@ -186,14 +200,14 @@ class TableStructureRecognizer(Recognizer):
        for b in boxes[1:]:
            b["cn"] = len(cols) - 1
            lst_c = cols[-1]
-            if (int(b.get("C", "1")) - int(lst_c[-1].get("C", "1")) == 1 and b["page_number"] == lst_c[-1][
-                "page_number"]) \
-                    or (b["x0"] >= right and lst_c[-1].get("C", "-1") != b.get("C", "-2")):  # new col
+            if (int(b.get("C", "1")) - int(lst_c[-1].get("C", "1")) == 1 and b["page_number"] == lst_c[-1]["page_number"]) or (
+                b["x0"] >= right and lst_c[-1].get("C", "-1") != b.get("C", "-2")
+            ):  # new col
                right = b["x1"]
                b["cn"] += 1
                cols.append([b])
                continue
-            right = (right + b["x1"]) / 2.
+            right = (right + b["x1"]) / 2.0
            cols[-1].append(b)

        tbl = [[[] for _ in range(len(cols))] for _ in range(len(rows))]
@ -214,10 +228,8 @@ class TableStructureRecognizer(Recognizer):
                if e > 1:
                    j += 1
                    continue
-                f = (j > 0 and tbl[ii][j - 1] and tbl[ii]
-                     [j - 1][0].get("text")) or j == 0
-                ff = (j + 1 < len(tbl[ii]) and tbl[ii][j + 1] and tbl[ii]
-                      [j + 1][0].get("text")) or j + 1 >= len(tbl[ii])
+                f = (j > 0 and tbl[ii][j - 1] and tbl[ii][j - 1][0].get("text")) or j == 0
+                ff = (j + 1 < len(tbl[ii]) and tbl[ii][j + 1] and tbl[ii][j + 1][0].get("text")) or j + 1 >= len(tbl[ii])
                if f and ff:
                    j += 1
                    continue
@ -228,13 +240,11 @@ class TableStructureRecognizer(Recognizer):
                if j > 0 and not f:
                    for i in range(len(tbl)):
                        if tbl[i][j - 1]:
-                            left = min(left, np.min(
-                                [bx["x0"] - a["x1"] for a in tbl[i][j - 1]]))
+                            left = min(left, np.min([bx["x0"] - a["x1"] for a in tbl[i][j - 1]]))
                if j + 1 < len(tbl[0]) and not ff:
                    for i in range(len(tbl)):
                        if tbl[i][j + 1]:
-                            right = min(right, np.min(
-                                [a["x0"] - bx["x1"] for a in tbl[i][j + 1]]))
+                            right = min(right, np.min([a["x0"] - bx["x1"] for a in tbl[i][j + 1]]))
                assert left < 100000 or right < 100000
                if left < right:
                    for jj in range(j, len(tbl[0])):
@ -260,8 +270,7 @@ class TableStructureRecognizer(Recognizer):
                    for i in range(len(tbl)):
                        tbl[i].pop(j)
                cols.pop(j)
-        assert len(cols) == len(tbl[0]), "Column NO. miss matched: %d vs %d" % (
-            len(cols), len(tbl[0]))
+        assert len(cols) == len(tbl[0]), "Column NO. miss matched: %d vs %d" % (len(cols), len(tbl[0]))

        if len(cols) >= 4:
            # remove single in row
@ -277,10 +286,8 @@ class TableStructureRecognizer(Recognizer):
                if e > 1:
                    i += 1
                    continue
-                f = (i > 0 and tbl[i - 1][jj] and tbl[i - 1]
-                     [jj][0].get("text")) or i == 0
-                ff = (i + 1 < len(tbl) and tbl[i + 1][jj] and tbl[i + 1]
-                      [jj][0].get("text")) or i + 1 >= len(tbl)
+                f = (i > 0 and tbl[i - 1][jj] and tbl[i - 1][jj][0].get("text")) or i == 0
+                ff = (i + 1 < len(tbl) and tbl[i + 1][jj] and tbl[i + 1][jj][0].get("text")) or i + 1 >= len(tbl)
                if f and ff:
                    i += 1
                    continue
@ -292,13 +299,11 @@ class TableStructureRecognizer(Recognizer):
                if i > 0 and not f:
                    for j in range(len(tbl[i - 1])):
                        if tbl[i - 1][j]:
-                            up = min(up, np.min(
-                                [bx["top"] - a["bottom"] for a in tbl[i - 1][j]]))
+                            up = min(up, np.min([bx["top"] - a["bottom"] for a in tbl[i - 1][j]]))
                if i + 1 < len(tbl) and not ff:
                    for j in range(len(tbl[i + 1])):
                        if tbl[i + 1][j]:
-                            down = min(down, np.min(
-                                [a["top"] - bx["bottom"] for a in tbl[i + 1][j]]))
+                            down = min(down, np.min([a["top"] - bx["bottom"] for a in tbl[i + 1][j]]))
                assert up < 100000 or down < 100000
                if up < down:
                    for ii in range(i, len(tbl)):
@ -333,22 +338,15 @@ class TableStructureRecognizer(Recognizer):
                cnt += 1
                if max_type == "Nu" and arr[0]["btype"] == "Nu":
                    continue
-                if any([a.get("H") for a in arr]) \
-                        or (max_type == "Nu" and arr[0]["btype"] != "Nu"):
+                if any([a.get("H") for a in arr]) or (max_type == "Nu" and arr[0]["btype"] != "Nu"):
                    h += 1
            if h / cnt > 0.5:
                hdset.add(i)

        if html:
-            return TableStructureRecognizer.__html_table(cap, hdset,
-                                                         TableStructureRecognizer.__cal_spans(boxes, rows,
-                                                                                              cols, tbl, True)
-                                                         )
+            return TableStructureRecognizer.__html_table(cap, hdset, TableStructureRecognizer.__cal_spans(boxes, rows, cols, tbl, True))

-        return TableStructureRecognizer.__desc_table(cap, hdset,
-                                                     TableStructureRecognizer.__cal_spans(boxes, rows, cols, tbl,
-                                                                                          False),
-                                                     is_english)
+        return TableStructureRecognizer.__desc_table(cap, hdset, TableStructureRecognizer.__cal_spans(boxes, rows, cols, tbl, False), is_english)

    @staticmethod
    def __html_table(cap, hdset, tbl):
@ -367,10 +365,8 @@ class TableStructureRecognizer(Recognizer):
                    continue
                txt = ""
                if arr:
-                    h = min(np.min([c["bottom"] - c["top"]
-                            for c in arr]) / 2, 10)
-                    txt = " ".join([c["text"]
-                                   for c in Recognizer.sort_Y_firstly(arr, h)])
+                    h = min(np.min([c["bottom"] - c["top"] for c in arr]) / 2, 10)
+                    txt = " ".join([c["text"] for c in Recognizer.sort_Y_firstly(arr, h)])
                txts.append(txt)
                sp = ""
                if arr[0].get("colspan"):
@ -436,15 +432,11 @@ class TableStructureRecognizer(Recognizer):
                    if headers[j][k].find(headers[j - 1][k]) >= 0:
                        continue
                    if len(headers[j][k]) > len(headers[j - 1][k]):
-                        headers[j][k] += (de if headers[j][k]
-                                          else "") + headers[j - 1][k]
+                        headers[j][k] += (de if headers[j][k] else "") + headers[j - 1][k]
                    else:
-                        headers[j][k] = headers[j - 1][k] \
-                            + (de if headers[j - 1][k] else "") \
-                            + headers[j][k]
+                        headers[j][k] = headers[j - 1][k] + (de if headers[j - 1][k] else "") + headers[j][k]

-        logging.debug(
-            f">>>>>>>>>>>>>>>>>{cap}：SIZE:{rowno}X{clmno} Header: {hdr_rowno}")
+        logging.debug(f">>>>>>>>>>>>>>>>>{cap}：SIZE:{rowno}X{clmno} Header: {hdr_rowno}")
        row_txt = []
        for i in range(rowno):
            if i in hdr_rowno:
@ -503,14 +495,10 @@ class TableStructureRecognizer(Recognizer):
    @staticmethod
    def __cal_spans(boxes, rows, cols, tbl, html=True):
        # caculate span
-        clft = [np.mean([c.get("C_left", c["x0"]) for c in cln])
-                for cln in cols]
-        crgt = [np.mean([c.get("C_right", c["x1"]) for c in cln])
-                for cln in cols]
-        rtop = [np.mean([c.get("R_top", c["top"]) for c in row])
-                for row in rows]
-        rbtm = [np.mean([c.get("R_btm", c["bottom"])
-                         for c in row]) for row in rows]
+        clft = [np.mean([c.get("C_left", c["x0"]) for c in cln]) for cln in cols]
+        crgt = [np.mean([c.get("C_right", c["x1"]) for c in cln]) for cln in cols]
+        rtop = [np.mean([c.get("R_top", c["top"]) for c in row]) for row in rows]
+        rbtm = [np.mean([c.get("R_btm", c["bottom"]) for c in row]) for row in rows]
        for b in boxes:
            if "SP" not in b:
                continue
@ -585,3 +573,40 @@ class TableStructureRecognizer(Recognizer):
                tbl[rowspan[0]][colspan[0]] = arr

        return tbl
+
+    def _run_ascend_tsr(self, image_list, thr=0.2, batch_size=16):
+        import math
+
+        from ais_bench.infer.interface import InferSession
+
+        model_dir = os.path.join(get_project_base_directory(), "rag/res/deepdoc")
+        model_file_path = os.path.join(model_dir, "tsr.om")
+
+        if not os.path.exists(model_file_path):
+            raise ValueError(f"Model file not found: {model_file_path}")
+
+        device_id = int(os.getenv("ASCEND_LAYOUT_RECOGNIZER_DEVICE_ID", 0))
+        session = InferSession(device_id=device_id, model_path=model_file_path)
+
+        images = [np.array(im) if not isinstance(im, np.ndarray) else im for im in image_list]
+        results = []
+
+        conf_thr = max(thr, 0.08)
+
+        batch_loop_cnt = math.ceil(float(len(images)) / batch_size)
+        for bi in range(batch_loop_cnt):
+            s = bi * batch_size
+            e = min((bi + 1) * batch_size, len(images))
+            batch_images = images[s:e]
+
+            inputs_list = self.preprocess(batch_images)
+            for ins in inputs_list:
+                feeds = []
+                if "image" in ins:
+                    feeds.append(ins["image"])
+                else:
+                    feeds.append(ins[self.input_names[0]])
+                output_list = session.infer(feeds=feeds, mode="static")
+                bb = self.postprocess(output_list, ins, conf_thr)
+                results.append(bb)
+        return results
--- a/docs/guides/agent/agent_component_reference/agent.mdx
+++ b/docs/guides/agent/agent_component_reference/agent.mdx
@ -26,6 +26,84 @@ An **Agent** component is essential when you need the LLM to assist with summari

 2. If your Agent involves dataset retrieval, ensure you [have properly configured your target knowledge base(s)](../../dataset/configure_knowledge_base.md).

+## Quickstart
+
+### 1. Click on an **Agent** component to show its configuration panel  
+
+The corresponding configuration panel appears to the right of the canvas. Use this panel to define and fine-tune the **Agent** component's behavior.
+
+### 2. Select your model
+
+Click **Model**, and select a chat model from the dropdown menu. 
+
+:::tip NOTE
+If no model appears, check if your have added a chat model on the **Model providers** page.
+:::
+
+### 3. Update system prompt (Optional)
+
+The system prompt typically defines your model's role. You can either keep the system prompt as is or customize it to override the default.
+
+
+### 4. Update user prompt
+
+The user prompt typically defines your model's task. You will find the `sys.query` variable auto-populated. Type `/` or click **(x)** to view or add variables.
+
+In this quickstart, we assume your **Agent** component is used standalone (without tools or sub-Agents below), then you may also need to specify retrieved chunks using the `formalized_content` variable:
+
+![](https://raw.githubusercontent.com/infiniflow/ragflow-docs/main/images/standalone_user_prompt_variable.jpg)
+
+### 5. Skip Tools and Agent
+
+The **+ Add tools** and **+ Add agent** sections are used *only* when you need to configure your **Agent** component as a planner (with tools or sub-Agents beneath). In this quickstart, we assume your **Agent** component is used standalone (without tools or sub-Agents beneath). 
+
+### 6. Choose the next component
+
+When necessary, click the **+** button on the **Agent** component to choose the next component in the worflow from the dropdown list.
+
+## Connect to an MCP server as a client
+
+:::danger IMPORTANT
+In this section, we assume your **Agent** will be configured as a planner, with a Tavily tool beneath it.
+:::
+
+### 1. Navigate to the MCP configuration page
+
+![](https://raw.githubusercontent.com/infiniflow/ragflow-docs/main/images/mcp_page.jpg)
+
+### 2. Configure your Tavily MCP server 
+
+Update your MCP server's name, URL (including the API key), server type, and other necessary settings. When configured correctly, the available tools will be displayed.
+
+![](https://raw.githubusercontent.com/infiniflow/ragflow-docs/main/images/edit_mcp_server.jpg)
+
+### 3. Navigate to your Agent's editing page
+
+### 4. Connect to your MCP server
+
+1. Click **+ Add tools**:
+
+![](https://raw.githubusercontent.com/infiniflow/ragflow-docs/main/images/add_tools.jpg)
+
+2. Click **MCP** to show the available MCP servers.
+
+3. Select your MCP server:
+
+   *The target MCP server appears below your Agent component, and your Agent will autonomously decide when to invoke the available tools it offers.*
+
+![](https://raw.githubusercontent.com/infiniflow/ragflow-docs/main/images/choose_tavily_mcp_server.jpg)
+
+### 5. Update system prompt to specify trigger conditions (Optional)
+
+To ensure reliable tool calls, you may specify within the system prompt which tasks should trigger each tool call.
+
+### 6. View the availabe tools of your MCP server
+
+On the canvas, click the newly-populated Tavily server to view and select its available tools:
+
+![](https://raw.githubusercontent.com/infiniflow/ragflow-docs/main/images/tavily_mcp_server.jpg)
+
+
 ## Configurations

 ### Model
@ -69,7 +147,7 @@ An **Agent** component relies on keys (variables) to specify its data inputs. It

 #### Advanced usage

-From v0.20.5 onwards, four framework-level prompt blocks are available in the **System prompt** field. Type `/` or click **(x)** to view them; they appear under the **Framework** entry in the dropdown menu.
+From v0.20.5 onwards, four framework-level prompt blocks are available in the **System prompt** field, enabling you to customize and *override* prompts at the framework level. Type `/` or click **(x)** to view them; they appear under the **Framework** entry in the dropdown menu.

 - `task_analysis` prompt block
  - This block is responsible for analyzing tasks — either a user task or a task assigned by the lead Agent when the **Agent** component is acting as a Sub-Agent.
@ -100,6 +178,12 @@ From v0.20.5 onwards, four framework-level prompt blocks are available in the **
 - `citation_guidelines` prompt block
  - Reference design: [citation_prompt.md](https://github.com/infiniflow/ragflow/blob/main/rag/prompts/citation_prompt.md)

+*The screenshots below show the framework prompt blocks available to an **Agent** component, both as a standalone and as a planner (with a Tavily tool below):*
+
+![standalone](https://raw.githubusercontent.com/infiniflow/ragflow-docs/main/images/standalone_agent_framework_block.jpg)
+
+![planner](https://raw.githubusercontent.com/infiniflow/ragflow-docs/main/images/planner_agent_framework_blocks.jpg)
+
 ### User prompt

 The user-defined prompt. Defaults to `sys.query`, the user query. As a general rule, when using the **Agent** component as a standalone module (not as a planner), you usually need to specify the corresponding **Retrieval** component’s output variable (`formalized_content`) here as part of the input to the LLM.
@ -129,7 +213,7 @@ Defines the maximum number of attempts the agent will make to retry a failed tas

 The waiting period in seconds that the agent observes before retrying a failed task, helping to prevent immediate repeated attempts and allowing system conditions to improve. Defaults to 1 second.

-### Max rounds
+### Max reflection rounds

 Defines the maximum number reflection rounds of the selected chat model. Defaults to 1 round.

--- a/docs/guides/agent/agent_component_reference/execute_sql.md
+++ b/docs/guides/agent/agent_component_reference/execute_sql.md
@ -0,0 +1,79 @@
+---
+sidebar_position: 25
+slug: /execute_sql
+---
+
+# Execute SQL tool
+
+A tool that execute SQL queries on a specified relational database.
+
+---
+
+The **Execute SQL** tool enables you to connect to a relational database and run SQL queries, whether entered directly or generated by the system’s Text2SQL capability via an **Agent** component. 
+
+## Prerequisites
+
+- A database instance properly configured and running.
+- The database must be one of the following types:
+  - MySQL
+  - PostgreSQL
+  - MariaDB
+  - Microsoft SQL Server
+
+## Examples
+
+You can pair an **Agent** component with the **Execute SQL** tool, with the **Agent** generating SQL statements and the **Execute SQL** tool handling database connection and query execution. An example of this setup can be found in the **SQL Assistant** Agent template shown below:
+
+![](https://raw.githubusercontent.com/infiniflow/ragflow-docs/main/images/exeSQL.jpg)
+
+## Configurations
+
+### SQL statement
+
+This text input field allows you to write static SQL queries, such as `SELECT * FROM my_table`, and dynamic SQL queries using variables.
+
+:::tip NOTE
+Click **(x)** or type `/` to insert variables.
+:::
+
+For dynamic SQL queries, you can include variables in your SQL queries, such as `SELECT * FROM /sys.query`; if an **Agent** component is paired with the **Execute SQL** tool to generate SQL tasks (see the [Examples](#examples) section), you can directly insert that **Agent**'s output, `content`, into this field.
+
+### Database type
+
+The supported database type. Currently the following database types are available:
+
+- MySQL
+- PostreSQL
+- MariaDB
+- Microsoft SQL Server (Myssql)
+
+### Database
+
+Appears only when you select **Split** as method.
+
+### Username
+
+The username with access privileges to the database.
+
+### Host
+
+The IP address of the database server.
+
+### Port
+
+The port number on which the database server is listening.
+
+### Password
+
+The password for the database user.
+
+### Max records
+
+The maximum number of records returned by the SQL query to control response size and improve efficiency. Defaults to `1024`.
+
+### Output
+
+The **Execute SQL** tool provides two output variables:
+
+- `formalized_content`: A string. If you reference this variable in a **Message** component, the returned records are displayed as a table.
+- `json`: An object array. If you reference this variable in a **Message** component, the returned records will be presented as key-value pairs.
--- a/docs/guides/chat/start_chat.md
+++ b/docs/guides/chat/start_chat.md
@ -106,7 +106,7 @@ RAGFlow offers HTTP and Python APIs for you to integrate RAGFlow's capabilities

 You can use iframe to embed the created chat assistant into a third-party webpage:

-1. Before proceeding, you must [acquire an API key](../models/llm_api_key_setup.md); otherwise, an error message would appear.
+1. Before proceeding, you must [acquire an API key](../../develop/acquire_ragflow_api_key.md); otherwise, an error message would appear.
 2. Hover over an intended chat assistant **>** **Edit** to show the **iframe** window:

   ![chat-embed](https://raw.githubusercontent.com/infiniflow/ragflow-docs/main/images/embed_chat_into_webpage.jpg)
--- a/docs/guides/models/deploy_local_llm.mdx
+++ b/docs/guides/models/deploy_local_llm.mdx
@ -91,7 +91,7 @@ In RAGFlow, click on your logo on the top right of the page **>** **Model provid
 In the popup window, complete basic settings for Ollama:

 1. Ensure that your model name and type match those been pulled at step 1 (Deploy Ollama using Docker). For example, (`llama3.2` and `chat`) or (`bge-m3` and `embedding`).
-2. In Ollama base URL, put the URL you found in step 2 followed by `/v1`, i.e. `http://host.docker.internal:11434/v1`, `http://localhost:11434/v1` or `http://${IP_OF_OLLAMA_MACHINE}:11434/v1`.
+2. Put in the Ollama base URL, i.e. `http://host.docker.internal:11434`, `http://localhost:11434` or `http://${IP_OF_OLLAMA_MACHINE}:11434`.
 3. OPTIONAL: Switch on the toggle under **Does it support Vision?** if your model includes an image-to-text model.


--- a/docs/references/python_api_reference.md
+++ b/docs/references/python_api_reference.md
@ -977,7 +977,7 @@ The languages that should be translated into, in order to achieve keywords retri

 ##### metadata_condition: `dict`

-filter condition for meta_fields
+filter condition for `meta_fields`.

 #### Returns

--- a/docs/references/supported_models.mdx
+++ b/docs/references/supported_models.mdx
@ -65,6 +65,7 @@ A complete list of models supported by RAGFlow, which will continue to expand.
 | 01.AI                 | :heavy_check_mark: |                    |                    |                    |                    |                    |
 | DeepInfra             | :heavy_check_mark: | :heavy_check_mark: |                    |                    | :heavy_check_mark: | :heavy_check_mark: |
 | 302.AI                | :heavy_check_mark: | :heavy_check_mark: | :heavy_check_mark: | :heavy_check_mark: |                    |                    |
+| CometAPI              | :heavy_check_mark: | :heavy_check_mark: | :heavy_check_mark: | :heavy_check_mark: |                    |                    |

 ```mdx-code-block
 </APITable>
--- a/docs/release_notes.md
+++ b/docs/release_notes.md
@ -28,11 +28,11 @@ Released on September 10, 2025.

 ### Improvements

- Agent Performance Optimized: Improved planning and reflection speed for simple tasks; optimized concurrent tool calls for parallelizable scenarios, significantly reducing overall response time.
- Agent Prompt Framework exposed: Developers can now customize and override framework-level prompts in the system prompt section, enhancing flexibility and control.
- Execute SQL Component Enhanced: Replaced the original variable reference component with a text input field, allowing free-form SQL writing with variable support.
- Chat: Re-enabled Reasoning and Cross-language search.
- Retrieval API Enhanced: Added metadata filtering support to the [Retrieve chunks](https://ragflow.io/docs/dev/http_api_reference#retrieve-chunks) method.
+- Agent: 
+  - Agent Performance Optimized: Improves planning and reflection speed for simple tasks; optimizes concurrent tool calls for parallelizable scenarios, significantly reducing overall response time.
+  - Four framework-level prompt blocks are available in the **System prompt** section, enabling customization and overriding of prompts at the framework level, thereby enhancing flexibility and control. See [here](./guides/agent/agent_component_reference/agent.mdx#system-prompt).
+  - **Execute SQL** component enhanced: Replaces the original variable reference component with a text input field, allowing users to write free-form SQL queries and reference variables.
+- Chat: Re-enables **Reasoning** and **Cross-language search**.

 ### Added models

@ -45,7 +45,21 @@ Released on September 10, 2025.

 - Dataset: Deleted files remained searchable.
 - Chat: Unable to chat with an Ollama model.
- Agent: Resolved issues including cite toggle failure, task mode requiring dialogue triggers, repeated answers in multi-turn dialogues, and duplicate summarization of parallel execution results.
+- Agent:
+  - A **Cite** toggle failure.
+  - An Agent in task mode still required a dialogue to trigger.
+  - Repeated answers in multi-turn dialogues.
+  - Duplicate summarization of parallel execution results.
+
+### API changes
+
+#### HTTP APIs
+
+- Adds a body parameter `"metadata_condition"` to the [Retrieve chunks](./references/http_api_reference.md#retrieve-chunks) method, enabling metadata-based chunk filtering during retrieval. [#9877](https://github.com/infiniflow/ragflow/pull/9877)
+
+#### Python APIs
+
+- Adds a parameter `metadata_condition` to the [Retrieve chunks](./references/python_api_reference.md#retrieve-chunks) method, enabling metadata-based chunk filtering during retrieval. [#9877](https://github.com/infiniflow/ragflow/pull/9877)

 ## v0.20.4

--- a/rag/app/naive.py
+++ b/rag/app/naive.py
@ -507,16 +507,29 @@ def chunk(filename, binary=None, from_page=0, to_page=100000,
        markdown_parser = Markdown(int(parser_config.get("chunk_token_num", 128)))
        sections, tables = markdown_parser(filename, binary, separate_tables=False)

+        try:
+            vision_model = LLMBundle(kwargs["tenant_id"], LLMType.IMAGE2TEXT)
+            callback(0.2, "Visual model detected. Attempting to enhance figure extraction...")
+        except Exception:
+            vision_model = None
+        
+        if vision_model:
            # Process images for each section
            section_images = []
-        for section_text, _ in sections:
+            for idx, (section_text, _) in enumerate(sections):
                images = markdown_parser.get_pictures(section_text) if section_text else None
+
                if images:
                    # If multiple images found, combine them using concat_img
                    combined_image = reduce(concat_img, images) if len(images) > 1 else images[0]
                    section_images.append(combined_image)
+                    markdown_vision_parser = VisionFigureParser(vision_model=vision_model, figures_data= [((combined_image, ["markdown image"]), [(0, 0, 0, 0, 0)])], **kwargs)
+                    boosted_figures = markdown_vision_parser(callback=callback)
+                    sections[idx] = (section_text + "\n\n" + "\n\n".join([fig[0][1] for fig in boosted_figures]), sections[idx][1])
                else:
                    section_images.append(None)
+        else:
+            logging.warning("No visual model detected. Skipping figure parsing enhancement.")

        res = tokenize_table(tables, doc, is_english)
        callback(0.8, "Finish parsing.")
--- a/rag/flow/base.py
+++ b/rag/flow/base.py
@ -13,7 +13,6 @@
 #  See the License for the specific language governing permissions and
 #  limitations under the License.
 #
-import logging
 import os
 import time
 from functools import partial
@ -44,17 +43,17 @@ class ProcessBase(ComponentBase):
        self.set_output("_created_time", time.perf_counter())
        for k, v in kwargs.items():
            self.set_output(k, v)
-        try:
+        #try:
        with trio.fail_after(self._param.timeout):
            await self._invoke(**kwargs)
            self.callback(1, "Done")
-        except Exception as e:
-            if self.get_exception_default_value():
-                self.set_exception_default_value()
-            else:
-                self.set_output("_ERROR", str(e))
-            logging.exception(e)
-            self.callback(-1, str(e))
+        #except Exception as e:
+        #    if self.get_exception_default_value():
+        #        self.set_exception_default_value()
+        #    else:
+        #        self.set_output("_ERROR", str(e))
+        #    logging.exception(e)
+        #    self.callback(-1, str(e))
        self.set_output("_elapsed_time", time.perf_counter() - self.output("_created_time"))
        return self.output()

--- a/rag/flow/chunker/chunker.py
+++ b/rag/flow/chunker/chunker.py
@ -12,18 +12,19 @@
 #  WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
 #  See the License for the specific language governing permissions and
 #  limitations under the License.
+import json
 import random
-
 import trio
-
 from api.db import LLMType
 from api.db.services.llm_service import LLMBundle
 from deepdoc.parser.pdf_parser import RAGFlowPdfParser
 from graphrag.utils import chat_limiter, get_llm_cache, set_llm_cache
 from rag.flow.base import ProcessBase, ProcessParamBase
 from rag.flow.chunker.schema import ChunkerFromUpstream
-from rag.nlp import naive_merge, naive_merge_with_images
-from rag.prompts.prompts import keyword_extraction, question_proposal
+from rag.nlp import naive_merge, naive_merge_with_images, concat_img
+from rag.prompts.prompts import keyword_extraction, question_proposal, detect_table_of_contents, \
+    table_of_contents_index, toc_transformer
+from rag.utils import num_tokens_from_string


 class ChunkerParam(ProcessParamBase):
@ -43,6 +44,7 @@ class ChunkerParam(ProcessParamBase):
            "paper",
            "laws",
            "presentation",
+            "toc" # table of contents
            # Other
            # "Tag" # TODO: Other method
        ]
@ -54,7 +56,7 @@ class ChunkerParam(ProcessParamBase):
        self.auto_keywords = 0
        self.auto_questions = 0
        self.tag_sets = []
-        self.llm_setting = {"llm_name": "", "lang": "Chinese"}
+        self.llm_setting = {"llm_id": "", "lang": "Chinese"}

    def check(self):
        self.check_valid_value(self.method.lower(), "Chunk method abnormal.", self.method_options)
@ -142,6 +144,91 @@ class Chunker(ProcessBase):
    def _one(self, from_upstream: ChunkerFromUpstream):
        pass

+    def _toc(self, from_upstream: ChunkerFromUpstream):
+        self.callback(random.randint(1, 5) / 100.0, "Start to chunk via `ToC`.")
+        if from_upstream.output_format in ["markdown", "text", "html"]:
+            return
+
+        # json
+        sections, section_images, page_1024, tc_arr = [], [], [""], [0]
+        for o in from_upstream.json_result or []:
+            txt = o.get("text", "")
+            tc = num_tokens_from_string(txt)
+            page_1024[-1] += "\n" + txt
+            tc_arr[-1] += tc
+            if tc_arr[-1] > 1024:
+                page_1024.append("")
+                tc_arr.append(0)
+            sections.append((o.get("text", ""), o.get("position_tag", "")))
+            section_images.append(o.get("image"))
+            print(len(sections), o)
+
+        llm_setting = self._param.llm_setting
+        chat_mdl = LLMBundle(self._canvas._tenant_id, LLMType.CHAT, llm_name=llm_setting["llm_id"], lang=llm_setting["lang"])
+        self.callback(random.randint(5, 15) / 100.0, "Start to detect table of contents...")
+        toc_secs = detect_table_of_contents(page_1024, chat_mdl)
+        if toc_secs:
+            self.callback(random.randint(25, 35) / 100.0, "Start to extract table of contents...")
+            toc_arr = toc_transformer(toc_secs, chat_mdl)
+            toc_arr = [it for it in toc_arr if it.get("structure")]
+            print(json.dumps(toc_arr, ensure_ascii=False, indent=2), flush=True)
+            self.callback(random.randint(35, 75) / 100.0, "Start to link table of contents...")
+            toc_arr = table_of_contents_index(toc_arr, [t for t,_ in sections], chat_mdl)
+            for i in range(len(toc_arr)-1):
+                if not toc_arr[i].get("indices"):
+                    continue
+
+                for j in range(i+1, len(toc_arr)):
+                    if toc_arr[j].get("indices"):
+                        if toc_arr[j]["indices"][0] - toc_arr[i]["indices"][-1] > 1:
+                            toc_arr[i]["indices"].extend([x for x in range(toc_arr[i]["indices"][-1]+1, toc_arr[j]["indices"][0])])
+                        break
+            # put all sections ahead of toc_arr[0] into it
+            # for i in range(len(toc_arr)):
+            #     if toc_arr[i].get("indices") and toc_arr[i]["indices"][0]:
+            #         toc_arr[i]["indices"] = [x for x in range(toc_arr[i]["indices"][-1]+1)]
+            #         break
+            # put all sections after toc_arr[-1] into it
+            for i in range(len(toc_arr)-1, -1, -1):
+                if toc_arr[i].get("indices") and toc_arr[i]["indices"][-1]:
+                    toc_arr[i]["indices"] = [x for x in range(toc_arr[i]["indices"][0], len(sections))]
+                    break
+            print(">>>>>>>>>>>>>>>>>>>>>>>>>>>>>>>>>>>>>>>>>>\n", json.dumps(toc_arr, ensure_ascii=False, indent=2), flush=True)
+
+            chunks, images = [], []
+            for it in toc_arr:
+                if not it.get("indices"):
+                    continue
+                txt = ""
+                img = None
+                for i in it["indices"]:
+                    idx = i
+                    txt += "\n" + sections[idx][0] + "\t" + sections[idx][1]
+                    if img and section_images[idx]:
+                        img = concat_img(img, section_images[idx])
+                    elif section_images[idx]:
+                        img = section_images[idx]
+
+                it["indices"] = []
+                if not txt:
+                    continue
+                it["indices"] = [len(chunks)]
+                print(it, "KKKKKKKKKKKKKKKKKKKKKKKKKKKKKKKKKK\n", txt)
+                chunks.append(txt)
+                images.append(img)
+            self.callback(1, "Done")
+            return [
+                {
+                    "text": RAGFlowPdfParser.remove_tag(c),
+                    "image": img,
+                    "positions": RAGFlowPdfParser.extract_positions(c),
+                }
+                for c, img in zip(chunks, images)
+            ]
+
+        self.callback(message="No table of contents detected.")
+
+
    async def _invoke(self, **kwargs):
        function_map = {
            "general": self._general,
@ -154,6 +241,7 @@ class Chunker(ProcessBase):
            "laws": self._laws,
            "presentation": self._presentation,
            "one": self._one,
+            "toc": self._toc,
        }

        try:
@ -167,7 +255,7 @@ class Chunker(ProcessBase):

        async def auto_keywords():
            nonlocal chunks, llm_setting
-            chat_mdl = LLMBundle(self._canvas._tenant_id, LLMType.CHAT, llm_name=llm_setting["llm_name"], lang=llm_setting["lang"])
+            chat_mdl = LLMBundle(self._canvas._tenant_id, LLMType.CHAT, llm_name=llm_setting["llm_id"], lang=llm_setting["lang"])

            async def doc_keyword_extraction(chat_mdl, ck, topn):
                cached = get_llm_cache(chat_mdl.llm_name, ck["text"], "keywords", {"topn": topn})
@ -184,7 +272,7 @@ class Chunker(ProcessBase):

        async def auto_questions():
            nonlocal chunks, llm_setting
-            chat_mdl = LLMBundle(self._canvas._tenant_id, LLMType.CHAT, llm_name=llm_setting["llm_name"], lang=llm_setting["lang"])
+            chat_mdl = LLMBundle(self._canvas._tenant_id, LLMType.CHAT, llm_name=llm_setting["llm_id"], lang=llm_setting["lang"])

            async def doc_question_proposal(chat_mdl, d, topn):
                cached = get_llm_cache(chat_mdl.llm_name, ck["text"], "question", {"topn": topn})
--- a/rag/flow/chunker/schema.py
+++ b/rag/flow/chunker/schema.py
@ -22,7 +22,7 @@ class ChunkerFromUpstream(BaseModel):
    elapsed_time: float | None = Field(default=None, alias="_elapsed_time")

    name: str
-    blob: bytes
+    file: dict | None = Field(default=None)

    output_format: Literal["json", "markdown", "text", "html"] | None = Field(default=None)

--- a/rag/flow/file.py
+++ b/rag/flow/file.py
@ -14,10 +14,7 @@
 #  limitations under the License.
 #
 from api.db.services.document_service import DocumentService
-from api.db.services.file2document_service import File2DocumentService
-from api.db.services.file_service import FileService
 from rag.flow.base import ProcessBase, ProcessParamBase
-from rag.utils.storage_factory import STORAGE_IMPL


 class FileParam(ProcessParamBase):
@ -41,10 +38,13 @@ class File(ProcessBase):
                self.set_output("_ERROR", f"Document({self._canvas._doc_id}) not found!")
                return

-            b, n = File2DocumentService.get_storage_address(doc_id=self._canvas._doc_id)
-            self.set_output("blob", STORAGE_IMPL.get(b, n))
+            #b, n = File2DocumentService.get_storage_address(doc_id=self._canvas._doc_id)
+            #self.set_output("blob", STORAGE_IMPL.get(b, n))
            self.set_output("name", doc.name)
        else:
            file = kwargs.get("file")
            self.set_output("name", file["name"])
-            self.set_output("blob", FileService.get_blob(file["created_by"], file["id"]))
+            self.set_output("file", file)
+            #self.set_output("blob", FileService.get_blob(file["created_by"], file["id"]))
+
+        self.callback(1, "File fetched.")
--- a/rag/flow/hierarchical_merger/init.py
+++ b/rag/flow/hierarchical_merger/init.py
@ -0,0 +1,15 @@
+#
+#  Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
+#
+#  Licensed under the Apache License, Version 2.0 (the "License");
+#  you may not use this file except in compliance with the License.
+#  You may obtain a copy of the License at
+#
+#      http://www.apache.org/licenses/LICENSE-2.0
+#
+#  Unless required by applicable law or agreed to in writing, software
+#  distributed under the License is distributed on an "AS IS" BASIS,
+#  WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+#  See the License for the specific language governing permissions and
+#  limitations under the License.
+
--- a/rag/flow/hierarchical_merger/hierarchical_merger.py
+++ b/rag/flow/hierarchical_merger/hierarchical_merger.py
@ -0,0 +1,178 @@
+#
+#  Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
+#
+#  Licensed under the Apache License, Version 2.0 (the "License");
+#  you may not use this file except in compliance with the License.
+#  You may obtain a copy of the License at
+#
+#      http://www.apache.org/licenses/LICENSE-2.0
+#
+#  Unless required by applicable law or agreed to in writing, software
+#  distributed under the License is distributed on an "AS IS" BASIS,
+#  WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+#  See the License for the specific language governing permissions and
+#  limitations under the License.
+import json
+import random
+import re
+from copy import deepcopy
+from functools import partial
+
+import trio
+
+from api.utils import get_uuid
+from api.utils.base64_image import id2image, image2id
+from deepdoc.parser.pdf_parser import RAGFlowPdfParser
+from rag.flow.base import ProcessBase, ProcessParamBase
+from rag.flow.hierarchical_merger.schema import HierarchicalMergerFromUpstream
+from rag.nlp import concat_img
+from rag.utils.storage_factory import STORAGE_IMPL
+
+
+class HierarchicalMergerParam(ProcessParamBase):
+    def __init__(self):
+        super().__init__()
+        self.levels = []
+        self.hierarchy = None
+
+    def check(self):
+        self.check_empty(self.levels, "Hierarchical setups.")
+        self.check_empty(self.hierarchy, "Hierarchy number.")
+
+    def get_input_form(self) -> dict[str, dict]:
+        return {}
+
+
+class HierarchicalMerger(ProcessBase):
+    component_name = "HierarchicalMerger"
+
+    async def _invoke(self, **kwargs):
+        try:
+            from_upstream = HierarchicalMergerFromUpstream.model_validate(kwargs)
+        except Exception as e:
+            self.set_output("_ERROR", f"Input error: {str(e)}")
+            return
+
+        self.callback(random.randint(1, 5) / 100.0, "Start to merge hierarchically.")
+        if from_upstream.output_format in ["markdown", "text", "html"]:
+            if from_upstream.output_format == "markdown":
+                payload = from_upstream.markdown_result
+            elif from_upstream.output_format == "text":
+                payload = from_upstream.text_result
+            else:  # == "html"
+                payload = from_upstream.html_result
+
+            if not payload:
+                payload = ""
+
+            lines = [ln for ln in payload.split("\n") if ln]
+        else:
+            lines = [o.get("text", "") for o in from_upstream.json_result]
+            sections, section_images = [], []
+            for o in from_upstream.json_result or []:
+                sections.append((o.get("text", ""), o.get("position_tag", "")))
+                section_images.append(o.get("img_id"))
+
+        matches = []
+        for txt in lines:
+            good = False
+            for lvl, regs in enumerate(self._param.levels):
+                for reg in regs:
+                    if re.search(reg, txt):
+                        matches.append(lvl)
+                        good = True
+                        break
+                if good:
+                    break
+            if not good:
+                matches.append(len(self._param.levels))
+        assert len(matches) == len(lines), f"{len(matches)} vs. {len(lines)}"
+
+        root = {
+            "level": -1,
+            "index": -1,
+            "texts": [],
+            "children": []
+        }
+        for i, m in enumerate(matches):
+            if m == 0:
+                root["children"].append({
+                    "level": m,
+                    "index": i,
+                    "texts": [],
+                    "children": []
+                })
+            elif m == len(self._param.levels):
+                def dfs(b):
+                    if not b["children"]:
+                        b["texts"].append(i)
+                    else:
+                        dfs(b["children"][-1])
+                dfs(root)
+            else:
+                def dfs(b):
+                    nonlocal m, i
+                    if not b["children"] or  m == b["level"] + 1:
+                        b["children"].append({
+                            "level": m,
+                            "index": i,
+                            "texts": [],
+                            "children": []
+                        })
+                        return
+                    dfs(b["children"][-1])
+
+                dfs(root)
+
+        all_pathes = []
+        def dfs(n, path, depth):
+            nonlocal all_pathes
+            if depth < self._param.hierarchy:
+                path = deepcopy(path)
+
+            for nn in n["children"]:
+                path.extend([nn["index"], *nn["texts"]])
+                dfs(nn, path, depth+1)
+
+            if depth == self._param.hierarchy:
+                all_pathes.append(path)
+
+        for i in range(len(lines)):
+            print(i, lines[i])
+        dfs(root, [], 0)
+        print("sSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSS", json.dumps(root, ensure_ascii=False, indent=2))
+
+        if from_upstream.output_format in ["markdown", "text", "html"]:
+            cks = []
+            for path in all_pathes:
+                txt = ""
+                for i in path:
+                    txt += lines[i] + "\n"
+                cks.append(txt)
+
+            self.set_output("chunks", [{"text": c} for c in cks if c])
+        else:
+            cks = []
+            images = []
+            for path in all_pathes:
+                txt = ""
+                img = None
+                for i in path:
+                    txt += lines[i] + "\n"
+                    concat_img(img, id2image(section_images[i], partial(STORAGE_IMPL.get)))
+                cks.append(cks)
+                images.append(img)
+
+            cks = [
+                {
+                    "text": RAGFlowPdfParser.remove_tag(c),
+                    "image": img,
+                    "positions": RAGFlowPdfParser.extract_positions(c),
+                }
+                for c, img in zip(cks, images)
+            ]
+            async with trio.open_nursery() as nursery:
+                for d in cks:
+                    nursery.start_soon(image2id, d, partial(STORAGE_IMPL.put), "_image_temps", get_uuid())
+
+        self.callback(1, "Done.")
--- a/rag/flow/hierarchical_merger/schema.py
+++ b/rag/flow/hierarchical_merger/schema.py
@ -0,0 +1,37 @@
+#
+#  Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
+#
+#  Licensed under the Apache License, Version 2.0 (the "License");
+#  you may not use this file except in compliance with the License.
+#  You may obtain a copy of the License at
+#
+#      http://www.apache.org/licenses/LICENSE-2.0
+#
+#  Unless required by applicable law or agreed to in writing, software
+#  distributed under the License is distributed on an "AS IS" BASIS,
+#  WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+#  See the License for the specific language governing permissions and
+#  limitations under the License.
+from typing import Any, Literal
+
+from pydantic import BaseModel, ConfigDict, Field
+
+
+class HierarchicalMergerFromUpstream(BaseModel):
+    created_time: float | None = Field(default=None, alias="_created_time")
+    elapsed_time: float | None = Field(default=None, alias="_elapsed_time")
+
+    name: str
+    file: dict | None = Field(default=None)
+    chunks: list[dict[str, Any]] | None = Field(default=None)
+
+    output_format: Literal["json", "markdown", "text", "html"] | None = Field(default=None)
+    json_result: list[dict[str, Any]] | None = Field(default=None, alias="json")
+    markdown_result: str | None = Field(default=None, alias="markdown")
+    text_result: str | None = Field(default=None, alias="text")
+    html_result: list[str] | None = Field(default=None, alias="html")
+
+    model_config = ConfigDict(populate_by_name=True, extra="forbid")
+
+    # def to_dict(self, *, exclude_none: bool = True) -> dict:
+    #     return self.model_dump(by_alias=True, exclude_none=exclude_none)
--- a/rag/flow/parser/parser.py
+++ b/rag/flow/parser/parser.py
@ -12,18 +12,27 @@
 #  WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
 #  See the License for the specific language governing permissions and
 #  limitations under the License.
+import io
 import logging
 import random
+from functools import partial

 import trio
+import numpy as np
+from PIL import Image

 from api.db import LLMType
+from api.db.services.file2document_service import File2DocumentService
+from api.db.services.file_service import FileService
 from api.db.services.llm_service import LLMBundle
+from api.utils import get_uuid
+from api.utils.base64_image import image2id
 from deepdoc.parser import ExcelParser
 from deepdoc.parser.pdf_parser import PlainParser, RAGFlowPdfParser, VisionParser
 from rag.flow.base import ProcessBase, ProcessParamBase
 from rag.flow.parser.schema import ParserFromUpstream
 from rag.llm.cv_model import Base as VLM
+from rag.utils.storage_factory import STORAGE_IMPL


 class ParserParam(ProcessParamBase):
@ -43,17 +52,24 @@ class ParserParam(ProcessParamBase):
                "json",
            ],
            "ppt": [],
-            "image": [],
+            "image": [
+                "text"
+            ],
            "email": [],
-            "text": [],
-            "audio": [],
+            "text": [
+                "text",
+                "json"
+            ],
+            "audio": [
+                "json"
+            ],
            "video": [],
        }

        self.setups = {
            "pdf": {
                "parse_method": "deepdoc",  # deepdoc/plain_text/vlm
-                "vlm_name": "",
+                "llm_id": "",
                "lang": "Chinese",
                "suffix": [
                    "pdf",
@ -76,16 +92,46 @@ class ParserParam(ProcessParamBase):
                "output_format": "json",
            },
            "markdown": {
-                "suffix": ["md", "markdown"],
+                "suffix": ["md", "markdown", "mdx"],
                "output_format": "json",
            },
            "ppt": {},
            "image": {
                "parse_method": "ocr",
+                "llm_id": "",
+                "lang": "Chinese",
+                "suffix": ["jpg", "jpeg", "png", "gif"],
+                "output_format": "json",
+            },
+            "email": {
+                "fields": []
+            },
+            "text": {
+                "suffix": [
+                    "txt"
+                ],
+                "output_format": "json",
+            },
+            "audio": {
+                "suffix":[
+                    "da",
+                    "wave",
+                    "wav",
+                    "mp3",
+                    "aac",
+                    "flac",
+                    "ogg",
+                    "aiff",
+                    "au",
+                    "midi",
+                    "wma",
+                    "realaudio",
+                    "vqf",
+                    "oggvorbis",
+                    "ape"
+                ],
+                "output_format": "json",
            },
-            "email": {},
-            "text": {},
-            "audio": {},
            "video": {},
        }

@ -96,7 +142,7 @@ class ParserParam(ProcessParamBase):
            self.check_valid_value(pdf_parse_method.lower(), "Parse method abnormal.", ["deepdoc", "plain_text", "vlm"])

            if pdf_parse_method not in ["deepdoc", "plain_text"]:
-                self.check_empty(pdf_config.get("vlm_name"), "VLM")
+                self.check_empty(pdf_config.get("llm_id"), "VLM")

            pdf_language = pdf_config.get("lang", "")
            self.check_empty(pdf_language, "Language")
@ -117,7 +163,23 @@ class ParserParam(ProcessParamBase):
        image_config = self.setups.get("image", "")
        if image_config:
            image_parse_method = image_config.get("parse_method", "")
-            self.check_valid_value(image_parse_method.lower(), "Parse method abnormal.", ["ocr"])
+            self.check_valid_value(image_parse_method.lower(), "Parse method abnormal.", ["ocr", "vlm"])
+            if image_parse_method not in ["ocr"]:
+                self.check_empty(image_config.get("llm_id"), "VLM")
+
+            image_language = image_config.get("lang", "")
+            self.check_empty(image_language, "Language")
+
+        text_config = self.setups.get("text", "")
+        if text_config:
+            text_output_format = text_config.get("output_format", "")
+            self.check_valid_value(text_output_format, "Text output format abnormal.", self.allowed_output_format["text"])
+
+        audio_config = self.setups.get("audio", "")
+        if audio_config:
+            self.check_empty(audio_config.get("llm_id"), "VLM")
+            audio_language = audio_config.get("lang", "")
+            self.check_empty(audio_language, "Language")

    def get_input_form(self) -> dict[str, dict]:
        return {}
@ -126,10 +188,8 @@ class ParserParam(ProcessParamBase):
 class Parser(ProcessBase):
    component_name = "Parser"

-    def _pdf(self, from_upstream: ParserFromUpstream):
+    def _pdf(self, name, blob):
        self.callback(random.randint(1, 5) / 100.0, "Start to work on a PDF.")
-
-        blob = from_upstream.blob
        conf = self._param.setups["pdf"]
        self.set_output("output_format", conf["output_format"])

@ -139,8 +199,8 @@ class Parser(ProcessBase):
            lines, _ = PlainParser()(blob)
            bboxes = [{"text": t} for t, _ in lines]
        else:
-            assert conf.get("vlm_name")
-            vision_model = LLMBundle(self._canvas._tenant_id, LLMType.IMAGE2TEXT, llm_name=conf.get("vlm_name"), lang=self._param.setups["pdf"].get("lang"))
+            assert conf.get("llm_id")
+            vision_model = LLMBundle(self._canvas._tenant_id, LLMType.IMAGE2TEXT, llm_name=conf.get("llm_id"), lang=self._param.setups["pdf"].get("lang"))
            lines, _ = VisionParser(vision_model=vision_model)(blob, callback=self.callback)
            bboxes = []
            for t, poss in lines:
@ -149,6 +209,7 @@ class Parser(ProcessBase):

        if conf.get("output_format") == "json":
            self.set_output("json", bboxes)
+
        if conf.get("output_format") == "markdown":
            mkdn = ""
            for b in bboxes:
@ -160,14 +221,10 @@ class Parser(ProcessBase):
                mkdn += b.get("text", "") + "\n"
            self.set_output("markdown", mkdn)

-    def _spreadsheet(self, from_upstream: ParserFromUpstream):
+    def _spreadsheet(self, name, blob):
        self.callback(random.randint(1, 5) / 100.0, "Start to work on a Spreadsheet.")
-
-        blob = from_upstream.blob
        conf = self._param.setups["spreadsheet"]
        self.set_output("output_format", conf["output_format"])
-
-        print("spreadsheet {conf=}", flush=True)
        spreadsheet_parser = ExcelParser()
        if conf.get("output_format") == "html":
            html = spreadsheet_parser.html(blob, 1000000000)
@ -177,19 +234,13 @@ class Parser(ProcessBase):
        elif conf.get("output_format") == "markdown":
            self.set_output("markdown", spreadsheet_parser.markdown(blob))

-    def _word(self, from_upstream: ParserFromUpstream):
+    def _word(self, name, blob):
        from tika import parser as  word_parser

        self.callback(random.randint(1, 5) / 100.0, "Start to work on a Word Processor Document")
-
-        blob = from_upstream.blob
-        name = from_upstream.name
        conf = self._param.setups["word"]
        self.set_output("output_format", conf["output_format"])
-
-        print("word {conf=}", flush=True)
        doc_parsed = word_parser.from_buffer(blob)
-
        sections = []
        if doc_parsed.get("content"):
            sections = doc_parsed["content"].split("\n")
@ -202,26 +253,18 @@ class Parser(ProcessBase):
        if conf.get("output_format") == "json":
            self.set_output("json", sections)

-    def _markdown(self, from_upstream: ParserFromUpstream):
+    def _markdown(self, name, blob):
        from functools import reduce
-
        from rag.app.naive import Markdown as naive_markdown_parser
        from rag.nlp import concat_img

-        self.callback(random.randint(1, 5) / 100.0, "Start to work on a Word Processor Document")
-
-        blob = from_upstream.blob
-        name = from_upstream.name
+        self.callback(random.randint(1, 5) / 100.0, "Start to work on a markdown.")
        conf = self._param.setups["markdown"]
        self.set_output("output_format", conf["output_format"])

-        print("markdown {conf=}", flush=True)
-
        markdown_parser = naive_markdown_parser()
        sections, tables = markdown_parser(name, blob, separate_tables=False)

-        # json
-        assert conf.get("output_format") == "json", "have to be json for doc"
        if conf.get("output_format") == "json":
            json_results = []

@ -239,14 +282,86 @@ class Parser(ProcessBase):
                json_results.append(json_result)

            self.set_output("json", json_results)
+        else:
+            self.set_output("text", "\n".join([section_text for section_text, _ in sections]))

+    def _text(self, name, blob):
+        from deepdoc.parser.utils import get_text
+
+        self.callback(random.randint(1, 5) / 100.0, "Start to work on a text.")
+        conf = self._param.setups["text"]
+        self.set_output("output_format", conf["output_format"])
+
+        # parse binary to text
+        text_content = get_text(name, binary=blob)
+
+        if conf.get("output_format") == "json":
+            result = [{"text": text_content}]
+            self.set_output("json", result)
+        else:
+            result = text_content
+            self.set_output("text", result)
+
+    def _image(self, from_upstream: ParserFromUpstream):
+        from deepdoc.vision import OCR
+
+        self.callback(random.randint(1, 5) / 100.0, "Start to work on an image.")
+
+        blob = from_upstream.blob
+        conf = self._param.setups["image"]
+        self.set_output("output_format", conf["output_format"])
+
+        img = Image.open(io.BytesIO(blob)).convert("RGB")
+        lang = conf["lang"]
+
+        if conf["parse_method"] == "ocr":
+            # use ocr, recognize chars only
+            ocr = OCR()
+            bxs = ocr(np.array(img))  # return boxes and recognize result
+            txt = "\n".join([t[0] for _, t in bxs if t[0]])
+
+        else:
+            # use VLM to describe the picture
+            cv_model = LLMBundle(self._canvas.get_tenant_id(), LLMType.IMAGE2TEXT, llm_name=conf["llm_id"],lang=lang)
+            img_binary = io.BytesIO()
+            img.save(img_binary, format="JPEG")
+            img_binary.seek(0)
+            txt = cv_model.describe(img_binary.read())
+
+        self.set_output("text", txt)
+
+    def _audio(self, from_upstream: ParserFromUpstream):
+        import os
+        import tempfile
+
+        self.callback(random.randint(1, 5) / 100.0, "Start to work on an audio.")
+
+        blob = from_upstream.blob
+        name = from_upstream.name
+        conf = self._param.setups["audio"]
+        self.set_output("output_format", conf["output_format"])
+
+        lang = conf["lang"]
+        _, ext = os.path.splitext(name)
+        with tempfile.NamedTemporaryFile(suffix=ext) as tmpf:
+            tmpf.write(blob)
+            tmpf.flush()
+            tmp_path = os.path.abspath(tmpf.name)
+
+            seq2txt_mdl = LLMBundle(self._canvas.get_tenant_id(), LLMType.SPEECH2TEXT, lang=lang)
+            txt = seq2txt_mdl.transcription(tmp_path)
+
+            self.set_output("text", txt)

    async def _invoke(self, **kwargs):
        function_map = {
            "pdf": self._pdf,
            "markdown": self._markdown,
            "spreadsheet": self._spreadsheet,
-            "word": self._word
+            "word": self._word,
+            "text": self._text,
+            "image": self._image,
+            "audio": self._audio,
        }
        try:
            from_upstream = ParserFromUpstream.model_validate(kwargs)
@ -254,8 +369,20 @@ class Parser(ProcessBase):
            self.set_output("_ERROR", f"Input error: {str(e)}")
            return

+        name = from_upstream.name
+        if self._canvas._doc_id:
+            b, n = File2DocumentService.get_storage_address(doc_id=self._canvas._doc_id)
+            blob = STORAGE_IMPL.get(b, n)
+        else:
+            blob = FileService.get_blob(from_upstream.file["created_by"], from_upstream.file["id"])
+
        for p_type, conf in self._param.setups.items():
            if from_upstream.name.split(".")[-1].lower() not in conf.get("suffix", []):
                continue
-            await trio.to_thread.run_sync(function_map[p_type], from_upstream)
+            await trio.to_thread.run_sync(function_map[p_type], name, blob)
            break
+
+        outs = self.output()
+        async with trio.open_nursery() as nursery:
+            for d in outs.get("json", []):
+                nursery.start_soon(image2id, d, partial(STORAGE_IMPL.put), "_image_temps", get_uuid())
--- a/rag/flow/parser/schema.py
+++ b/rag/flow/parser/schema.py
@ -20,6 +20,5 @@ class ParserFromUpstream(BaseModel):
    elapsed_time: float | None = Field(default=None, alias="_elapsed_time")

    name: str
-    blob: bytes
-
+    file: dict | None = Field(default=None)
    model_config = ConfigDict(populate_by_name=True, extra="forbid")
--- a/rag/flow/pipeline.py
+++ b/rag/flow/pipeline.py
@ -48,7 +48,24 @@ class Pipeline(Graph):
                    obj.append({"component_name": component_name, "trace": [{"progress": progress, "message": message, "datetime": datetime.datetime.now().strftime("%H:%M:%S")}]})
            else:
                obj = [{"component_name": component_name, "trace": [{"progress": progress, "message": message, "datetime": datetime.datetime.now().strftime("%H:%M:%S")}]}]
-            REDIS_CONN.set_obj(log_key, obj, 60 * 10)
+            REDIS_CONN.set_obj(log_key, obj, 60 * 30)
+            if self._doc_id:
+                percentage = 1./len(self.components.items())
+                msg = ""
+                finished = 0.
+                for o in obj:
+                    if o['component_name'] == "END":
+                        continue
+                    msg += f"\n[{o['component_name']}]:\n"
+                    for t in o["trace"]:
+                        msg += "%s: %s\n"%(t["datetime"], t["message"])
+                        if t["progress"] < 0:
+                            finished = -1
+                            break
+                    if finished < 0:
+                        break
+                    finished += o["trace"][-1]["progress"] * percentage
+                DocumentService.update_by_id(self._doc_id, {"progress": finished, "progress_msg": msg})
        except Exception as e:
            logging.exception(e)

@ -108,5 +125,11 @@ class Pipeline(Graph):
            idx += 1
            self.path.extend(cpn_obj.get_downstream())

+        self.callback("END", 1, json.dumps(self.get_component_obj(self.path[-1]).output(), ensure_ascii=False))
+
        if self._doc_id:
-            DocumentService.update_by_id(self._doc_id, {"progress": 1 if not self.error else -1, "progress_msg": "Pipeline finished...\n" + self.error, "process_duration": time.perf_counter() - st})
+            DocumentService.update_by_id(self._doc_id,{
+                "progress": 1 if not self.error else -1,
+                "progress_msg": "Pipeline finished...\n" + self.error,
+                "process_duration": time.perf_counter() - st
+            })
--- a/rag/flow/splitter/init.py
+++ b/rag/flow/splitter/init.py
@ -0,0 +1,15 @@
+#
+#  Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
+#
+#  Licensed under the Apache License, Version 2.0 (the "License");
+#  you may not use this file except in compliance with the License.
+#  You may obtain a copy of the License at
+#
+#      http://www.apache.org/licenses/LICENSE-2.0
+#
+#  Unless required by applicable law or agreed to in writing, software
+#  distributed under the License is distributed on an "AS IS" BASIS,
+#  WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+#  See the License for the specific language governing permissions and
+#  limitations under the License.
+
--- a/rag/flow/splitter/schema.py
+++ b/rag/flow/splitter/schema.py
@ -0,0 +1,38 @@
+#
+#  Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
+#
+#  Licensed under the Apache License, Version 2.0 (the "License");
+#  you may not use this file except in compliance with the License.
+#  You may obtain a copy of the License at
+#
+#      http://www.apache.org/licenses/LICENSE-2.0
+#
+#  Unless required by applicable law or agreed to in writing, software
+#  distributed under the License is distributed on an "AS IS" BASIS,
+#  WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+#  See the License for the specific language governing permissions and
+#  limitations under the License.
+from typing import Any, Literal
+
+from pydantic import BaseModel, ConfigDict, Field
+
+
+class SplitterFromUpstream(BaseModel):
+    created_time: float | None = Field(default=None, alias="_created_time")
+    elapsed_time: float | None = Field(default=None, alias="_elapsed_time")
+
+    name: str
+    file: dict | None = Field(default=None)
+    chunks: list[dict[str, Any]] | None = Field(default=None)
+
+    output_format: Literal["json", "markdown", "text", "html"] | None = Field(default=None)
+
+    json_result: list[dict[str, Any]] | None = Field(default=None, alias="json")
+    markdown_result: str | None = Field(default=None, alias="markdown")
+    text_result: str | None = Field(default=None, alias="text")
+    html_result: list[str] | None = Field(default=None, alias="html")
+
+    model_config = ConfigDict(populate_by_name=True, extra="forbid")
+
+    # def to_dict(self, *, exclude_none: bool = True) -> dict:
+    #     return self.model_dump(by_alias=True, exclude_none=exclude_none)
--- a/rag/flow/splitter/splitter.py
+++ b/rag/flow/splitter/splitter.py
@ -0,0 +1,112 @@
+#
+#  Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
+#
+#  Licensed under the Apache License, Version 2.0 (the "License");
+#  you may not use this file except in compliance with the License.
+#  You may obtain a copy of the License at
+#
+#      http://www.apache.org/licenses/LICENSE-2.0
+#
+#  Unless required by applicable law or agreed to in writing, software
+#  distributed under the License is distributed on an "AS IS" BASIS,
+#  WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+#  See the License for the specific language governing permissions and
+#  limitations under the License.
+import json
+import random
+from functools import partial
+
+import trio
+
+from api.utils import get_uuid
+from api.utils.base64_image import id2image, image2id
+from deepdoc.parser.pdf_parser import RAGFlowPdfParser
+from rag.flow.base import ProcessBase, ProcessParamBase
+from rag.flow.splitter.schema import SplitterFromUpstream
+from rag.nlp import naive_merge, naive_merge_with_images
+from rag.utils.storage_factory import STORAGE_IMPL
+
+
+class SplitterParam(ProcessParamBase):
+    def __init__(self):
+        super().__init__()
+        self.chunk_token_size = 512
+        self.delimiters = ["\n"]
+        self.overlapped_percent = 0
+
+    def check(self):
+        self.check_empty(self.delimiters, "Delimiters.")
+        self.check_positive_integer(self.chunk_token_size, "Chunk token size.")
+        self.check_decimal_float(self.overlapped_percent, "Overlapped percentage: [0, 1)")
+
+    def get_input_form(self) -> dict[str, dict]:
+        return {}
+
+
+class Splitter(ProcessBase):
+    component_name = "Splitter"
+
+    async def _invoke(self, **kwargs):
+        try:
+            from_upstream = SplitterFromUpstream.model_validate(kwargs)
+        except Exception as e:
+            self.set_output("_ERROR", f"Input error: {str(e)}")
+            return
+
+        deli = ""
+        for d in self._param.delimiters:
+            if len(d) > 1:
+                deli += f"`{d}`"
+            else:
+                deli += d
+
+        self.callback(random.randint(1, 5) / 100.0, "Start to split into chunks.")
+        if from_upstream.output_format in ["markdown", "text", "html"]:
+            if from_upstream.output_format == "markdown":
+                payload = from_upstream.markdown_result
+            elif from_upstream.output_format == "text":
+                payload = from_upstream.text_result
+            else:  # == "html"
+                payload = from_upstream.html_result
+
+            if not payload:
+                payload = ""
+
+            cks = naive_merge(
+                payload,
+                self._param.chunk_token_size,
+                deli,
+                self._param.overlapped_percent,
+            )
+            self.set_output("chunks", [{"text": c} for c in cks])
+
+            self.callback(1, "Done.")
+            return
+
+        # json
+        sections, section_images = [], []
+        for o in from_upstream.json_result or []:
+            sections.append((o.get("text", ""), o.get("position_tag", "")))
+            section_images.append(id2image(o.get("img_id"), partial(STORAGE_IMPL.get)))
+
+        chunks, images = naive_merge_with_images(
+            sections,
+            section_images,
+            self._param.chunk_token_size,
+            deli,
+            self._param.overlapped_percent,
+        )
+        cks = [
+            {
+                "text": RAGFlowPdfParser.remove_tag(c),
+                "image": img,
+                "positions": RAGFlowPdfParser.extract_positions(c),
+            }
+            for c, img in zip(chunks, images)
+        ]
+        async with trio.open_nursery() as nursery:
+            for d in cks:
+                nursery.start_soon(image2id, d, partial(STORAGE_IMPL.put), "_image_temps", get_uuid())
+        print("SSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSSS\n", json.dumps(cks, ensure_ascii=False, indent=2))
+        self.set_output("chunks",  cks)
+        self.callback(1, "Done.")
--- a/rag/flow/tests/dsl_examples/general_pdf_all.json
+++ b/rag/flow/tests/dsl_examples/general_pdf_all.json
@ -44,20 +44,58 @@
                    "markdown"
                  ],
                  "output_format": "json"
-                }
+                },
+                "text": {
+                  "suffix": ["txt"],
+                  "output_format": "json"
+                },
+                "image": {
+                  "parse_method": "vlm",
+                  "llm_id":"glm-4.5v",
+                  "lang": "Chinese",
+                  "suffix": [
+                    "jpg",
+                    "jpeg",
+                    "png",
+                    "gif"
+                  ],
+                  "output_format": "text"
+                },
+                "audio": {
+                  "suffix": [
+                    "da",
+                    "wave",
+                    "wav",
+                    "mp3",
+                    "aac",
+                    "flac",
+                    "ogg",
+                    "aiff",
+                    "au",
+                    "midi",
+                    "wma",
+                    "realaudio",
+                    "vqf",
+                    "oggvorbis",
+                    "ape"
+                  ],
+                  "lang": "Chinese",
+                  "llm_id": "SenseVoiceSmall",
+                  "output_format": "json"
                }
              }
          }
        },
-        "downstream": ["Chunker:0"],
+        "downstream": ["Splitter:0"],
        "upstream": ["Begin"]
    },
-    "Chunker:0": {
+    "Splitter:0": {
        "obj": {
-            "component_name": "Chunker",
+            "component_name": "Splitter",
            "params": {
-              "method": "general",
-              "auto_keywords": 5
+              "chunk_token_size": 512,
+              "delimiters": ["\n"],
+              "overlapped_percent": 0
            }
        },
        "downstream": ["Tokenizer:0"],
--- a/rag/flow/tests/dsl_examples/hierarchical_merger.json
+++ b/rag/flow/tests/dsl_examples/hierarchical_merger.json
@ -0,0 +1,84 @@
+{
+  "components": {
+    "File": {
+        "obj":{
+            "component_name": "File",
+            "params": {
+            }
+        },
+        "downstream": ["Parser:0"],
+        "upstream": []
+    },
+    "Parser:0": {
+        "obj": {
+            "component_name": "Parser",
+            "params": {
+              "setups": {
+                "pdf": {
+                  "parse_method": "deepdoc",
+                  "vlm_name": "",
+                  "lang": "Chinese",
+                  "suffix": [
+                    "pdf"
+                  ],
+                  "output_format": "json"
+                },
+                "spreadsheet": {
+                  "suffix": [
+                    "xls",
+                    "xlsx",
+                    "csv"
+                  ],
+                  "output_format": "html"
+                },
+                "word": {
+                  "suffix": [
+                    "doc",
+                    "docx"
+                  ],
+                  "output_format": "json"
+                },
+                "markdown": {
+                  "suffix": [
+                    "md",
+                    "markdown"
+                  ],
+                  "output_format": "text"
+                },
+                "text": {
+                  "suffix": ["txt"],
+                  "output_format": "json"
+                }
+              }
+          }
+        },
+        "downstream": ["Splitter:0"],
+        "upstream": ["File"]
+    },
+    "Splitter:0": {
+        "obj": {
+            "component_name": "Splitter",
+            "params": {
+              "chunk_token_size": 512,
+              "delimiters": ["\r\n"],
+              "overlapped_percent": 0
+            }
+        },
+        "downstream": ["HierarchicalMerger:0"],
+        "upstream": ["Parser:0"]
+    },
+    "HierarchicalMerger:0": {
+        "obj": {
+            "component_name": "HierarchicalMerger",
+            "params": {
+              "levels": [["^#[^#]"], ["^##[^#]"], ["^###[^#]"], ["^####[^#]"]],
+              "hierarchy": 2
+            }
+        },
+        "downstream": [],
+        "upstream": ["Splitter:0"]
+    }
+  },
+  "path": []
+}
+
--- a/rag/flow/tokenizer/schema.py
+++ b/rag/flow/tokenizer/schema.py
@ -22,7 +22,7 @@ class TokenizerFromUpstream(BaseModel):
    elapsed_time: float | None = Field(default=None, alias="_elapsed_time")

    name: str = ""
-    blob: bytes
+    file: dict | None = Field(default=None)

    output_format: Literal["json", "markdown", "text", "html"] | None = Field(default=None)

--- a/rag/flow/tokenizer/tokenizer.py
+++ b/rag/flow/tokenizer/tokenizer.py
@ -37,6 +37,7 @@ class TokenizerParam(ProcessParamBase):
        super().__init__()
        self.search_method = ["full_text", "embedding"]
        self.filename_embd_weight = 0.1
+        self.fields = ["text"]

    def check(self):
        for v in self.search_method:
@ -61,10 +62,14 @@ class Tokenizer(ProcessBase):
        embedding_model = LLMBundle(self._canvas._tenant_id, LLMType.EMBEDDING, llm_name=embedding_id)
        texts = []
        for c in chunks:
-            if c.get("questions"):
-                texts.append("\n".join(c["questions"]))
-            else:
-                texts.append(re.sub(r"</?(table|td|caption|tr|th)( [^<>]{0,12})?>", " ", c["text"]))
+            txt = ""
+            for f in self._param.fields:
+                f = c.get(f)
+                if isinstance(f, str):
+                    txt += f
+                elif isinstance(f, list):
+                    txt += "\n".join(f)
+            texts.append(re.sub(r"</?(table|td|caption|tr|th)( [^<>]{0,12})?>", " ", txt))
        vts, c = embedding_model.encode([name])
        token_count += c
        tts = np.concatenate([vts[0] for _ in range(len(texts))], axis=0)
--- a/rag/llm/init.py
+++ b/rag/llm/init.py
@ -37,6 +37,18 @@ class SupportedLiteLLMProvider(StrEnum):
    TogetherAI = "TogetherAI"
    Anthropic = "Anthropic"
    Ollama = "Ollama"
+    Meituan = "Meituan"
+    CometAPI = "CometAPI"
+    SILICONFLOW = "SILICONFLOW"
+    OpenRouter = "OpenRouter"
+    StepFun = "StepFun"
+    PPIO = "PPIO"
+    PerfXCloud = "PerfXCloud"
+    Upstage = "Upstage"
+    NovitaAI = "NovitaAI"
+    Lingyi_AI = "01.AI"
+    GiteeAI = "GiteeAI"
+    AI_302 = "302.AI"


 FACTORY_DEFAULT_BASE_URL = {
@ -44,6 +56,18 @@ FACTORY_DEFAULT_BASE_URL = {
    SupportedLiteLLMProvider.Dashscope: "https://dashscope.aliyuncs.com/compatible-mode/v1",
    SupportedLiteLLMProvider.Moonshot: "https://api.moonshot.cn/v1",
    SupportedLiteLLMProvider.Ollama: "",
+    SupportedLiteLLMProvider.Meituan: "https://api.longcat.chat/openai",
+    SupportedLiteLLMProvider.CometAPI: "https://api.cometapi.com/v1",
+    SupportedLiteLLMProvider.SILICONFLOW: "https://api.siliconflow.cn/v1",
+    SupportedLiteLLMProvider.OpenRouter: "https://openrouter.ai/api/v1",
+    SupportedLiteLLMProvider.StepFun: "https://api.stepfun.com/v1",
+    SupportedLiteLLMProvider.PPIO: "https://api.ppinfra.com/v3/openai",
+    SupportedLiteLLMProvider.PerfXCloud: "https://cloud.perfxlab.cn/v1",
+    SupportedLiteLLMProvider.Upstage: "https://api.upstage.ai/v1/solar",
+    SupportedLiteLLMProvider.NovitaAI: "https://api.novita.ai/v3/openai",
+    SupportedLiteLLMProvider.Lingyi_AI: "https://api.lingyiwanwu.com/v1",
+    SupportedLiteLLMProvider.GiteeAI: "https://ai.gitee.com/v1/",
+    SupportedLiteLLMProvider.AI_302: "https://api.302.ai/v1",
 }


@ -62,6 +86,18 @@ LITELLM_PROVIDER_PREFIX = {
    SupportedLiteLLMProvider.TogetherAI: "together_ai/",
    SupportedLiteLLMProvider.Anthropic: "",  # don't need a prefix
    SupportedLiteLLMProvider.Ollama: "ollama_chat/",
+    SupportedLiteLLMProvider.Meituan: "openai/",
+    SupportedLiteLLMProvider.CometAPI: "openai/",
+    SupportedLiteLLMProvider.SILICONFLOW: "openai/",
+    SupportedLiteLLMProvider.OpenRouter: "openai/",
+    SupportedLiteLLMProvider.StepFun: "openai/",
+    SupportedLiteLLMProvider.PPIO: "openai/",
+    SupportedLiteLLMProvider.PerfXCloud: "openai/",
+    SupportedLiteLLMProvider.Upstage: "openai/",
+    SupportedLiteLLMProvider.NovitaAI: "openai/",
+    SupportedLiteLLMProvider.Lingyi_AI: "openai/",
+    SupportedLiteLLMProvider.GiteeAI: "openai/",
+    SupportedLiteLLMProvider.AI_302: "openai/",
 }

 ChatModel = globals().get("ChatModel", {})
--- a/rag/llm/chat_model.py
+++ b/rag/llm/chat_model.py
@ -895,25 +895,6 @@ class MistralChat(Base):
        yield total_tokens


-## openrouter
-class OpenRouterChat(Base):
-    _FACTORY_NAME = "OpenRouter"
-
-    def __init__(self, key, model_name, base_url="https://openrouter.ai/api/v1", **kwargs):
-        if not base_url:
-            base_url = "https://openrouter.ai/api/v1"
-        super().__init__(key, model_name, base_url, **kwargs)
-
-
-class StepFunChat(Base):
-    _FACTORY_NAME = "StepFun"
-
-    def __init__(self, key, model_name, base_url="https://api.stepfun.com/v1", **kwargs):
-        if not base_url:
-            base_url = "https://api.stepfun.com/v1"
-        super().__init__(key, model_name, base_url, **kwargs)
-
-
 class LmStudioChat(Base):
    _FACTORY_NAME = "LM-Studio"

@ -936,15 +917,6 @@ class OpenAI_APIChat(Base):
        super().__init__(key, model_name, base_url, **kwargs)


-class PPIOChat(Base):
-    _FACTORY_NAME = "PPIO"
-
-    def __init__(self, key, model_name, base_url="https://api.ppinfra.com/v3/openai", **kwargs):
-        if not base_url:
-            base_url = "https://api.ppinfra.com/v3/openai"
-        super().__init__(key, model_name, base_url, **kwargs)
-
-
 class LeptonAIChat(Base):
    _FACTORY_NAME = "LeptonAI"

@ -954,60 +926,6 @@ class LeptonAIChat(Base):
        super().__init__(key, model_name, base_url, **kwargs)


-class PerfXCloudChat(Base):
-    _FACTORY_NAME = "PerfXCloud"
-
-    def __init__(self, key, model_name, base_url="https://cloud.perfxlab.cn/v1", **kwargs):
-        if not base_url:
-            base_url = "https://cloud.perfxlab.cn/v1"
-        super().__init__(key, model_name, base_url, **kwargs)
-
-
-class UpstageChat(Base):
-    _FACTORY_NAME = "Upstage"
-
-    def __init__(self, key, model_name, base_url="https://api.upstage.ai/v1/solar", **kwargs):
-        if not base_url:
-            base_url = "https://api.upstage.ai/v1/solar"
-        super().__init__(key, model_name, base_url, **kwargs)
-
-
-class NovitaAIChat(Base):
-    _FACTORY_NAME = "NovitaAI"
-
-    def __init__(self, key, model_name, base_url="https://api.novita.ai/v3/openai", **kwargs):
-        if not base_url:
-            base_url = "https://api.novita.ai/v3/openai"
-        super().__init__(key, model_name, base_url, **kwargs)
-
-
-class SILICONFLOWChat(Base):
-    _FACTORY_NAME = "SILICONFLOW"
-
-    def __init__(self, key, model_name, base_url="https://api.siliconflow.cn/v1", **kwargs):
-        if not base_url:
-            base_url = "https://api.siliconflow.cn/v1"
-        super().__init__(key, model_name, base_url, **kwargs)
-
-
-class YiChat(Base):
-    _FACTORY_NAME = "01.AI"
-
-    def __init__(self, key, model_name, base_url="https://api.lingyiwanwu.com/v1", **kwargs):
-        if not base_url:
-            base_url = "https://api.lingyiwanwu.com/v1"
-        super().__init__(key, model_name, base_url, **kwargs)
-
-
-class GiteeChat(Base):
-    _FACTORY_NAME = "GiteeAI"
-
-    def __init__(self, key, model_name, base_url="https://ai.gitee.com/v1/", **kwargs):
-        if not base_url:
-            base_url = "https://ai.gitee.com/v1/"
-        super().__init__(key, model_name, base_url, **kwargs)
-
-
 class ReplicateChat(Base):
    _FACTORY_NAME = "Replicate"

@ -1347,26 +1265,46 @@ class GPUStackChat(Base):
        super().__init__(key, model_name, base_url, **kwargs)


-class Ai302Chat(Base):
-    _FACTORY_NAME = "302.AI"
+class TokenPonyChat(Base):
+    _FACTORY_NAME = "TokenPony"

-    def __init__(self, key, model_name, base_url="https://api.302.ai/v1", **kwargs):
+    def __init__(self, key, model_name, base_url="https://ragflow.vip-api.tokenpony.cn/v1", **kwargs):
        if not base_url:
-            base_url = "https://api.302.ai/v1"
-        super().__init__(key, model_name, base_url, **kwargs)
-
-
-class MeituanChat(Base):
-    _FACTORY_NAME = "Meituan"
-
-    def __init__(self, key, model_name, base_url="https://api.longcat.chat/openai", **kwargs):
-        if not base_url:
-            base_url = "https://api.longcat.chat/openai"
-        super().__init__(key, model_name, base_url, **kwargs)
+            base_url = "https://ragflow.vip-api.tokenpony.cn/v1"


 class LiteLLMBase(ABC):
-    _FACTORY_NAME = ["Tongyi-Qianwen", "Bedrock", "Moonshot", "xAI", "DeepInfra", "Groq", "Cohere", "Gemini", "DeepSeek", "NVIDIA", "TogetherAI", "Anthropic", "Ollama"]
+    _FACTORY_NAME = [
+        "Tongyi-Qianwen",
+        "Bedrock",
+        "Moonshot",
+        "xAI",
+        "DeepInfra",
+        "Groq",
+        "Cohere",
+        "Gemini",
+        "DeepSeek",
+        "NVIDIA",
+        "TogetherAI",
+        "Anthropic",
+        "Ollama",
+        "Meituan",
+        "CometAPI",
+        "SILICONFLOW",
+        "OpenRouter",
+        "StepFun",
+        "PPIO",
+        "PerfXCloud",
+        "Upstage",
+        "NovitaAI",
+        "01.AI",
+        "GiteeAI",
+        "302.AI",
+    ]
+
+    import litellm
+
+    litellm._turn_on_debug()

    def __init__(self, key, model_name, base_url=None, **kwargs):
        self.timeout = int(os.environ.get("LM_TIMEOUT_SECONDS", 600))
@ -1374,7 +1312,7 @@ class LiteLLMBase(ABC):
        self.prefix = LITELLM_PROVIDER_PREFIX.get(self.provider, "")
        self.model_name = f"{self.prefix}{model_name}"
        self.api_key = key
-        self.base_url = (base_url or FACTORY_DEFAULT_BASE_URL.get(self.provider, "")).rstrip('/')
+        self.base_url = (base_url or FACTORY_DEFAULT_BASE_URL.get(self.provider, "")).rstrip("/")
        # Configure retry parameters
        self.max_retries = kwargs.get("max_retries", int(os.environ.get("LLM_MAX_RETRIES", 5)))
        self.base_delay = kwargs.get("retry_interval", float(os.environ.get("LLM_BASE_DELAY", 2.0)))
--- a/rag/llm/embedding_model.py
+++ b/rag/llm/embedding_model.py
@ -86,6 +86,7 @@ class DefaultEmbedding(Base):
            with DefaultEmbedding._model_lock:
                import torch
                from FlagEmbedding import FlagModel
+
                if "CUDA_VISIBLE_DEVICES" in os.environ:
                    input_cuda_visible_devices = os.environ["CUDA_VISIBLE_DEVICES"]
                    os.environ["CUDA_VISIBLE_DEVICES"] = "0"  # handle some issues with multiple GPUs when initializing the model
@ -145,7 +146,7 @@ class OpenAIEmbed(Base):
        ress = []
        total_tokens = 0
        for i in range(0, len(texts), batch_size):
-            res = self.client.embeddings.create(input=texts[i : i + batch_size], model=self.model_name, encoding_format="float")
+            res = self.client.embeddings.create(input=texts[i : i + batch_size], model=self.model_name, encoding_format="float", extra_body={"drop_params": True})
            try:
                ress.extend([d.embedding for d in res.data])
                total_tokens += self.total_token_count(res)
@ -154,7 +155,7 @@ class OpenAIEmbed(Base):
        return np.array(ress), total_tokens

    def encode_queries(self, text):
-        res = self.client.embeddings.create(input=[truncate(text, 8191)], model=self.model_name, encoding_format="float")
+        res = self.client.embeddings.create(input=[truncate(text, 8191)], model=self.model_name, encoding_format="float",extra_body={"drop_params": True})
        return np.array(res.data[0].embedding), self.total_token_count(res)


@ -472,6 +473,7 @@ class MistralEmbed(Base):
    def encode(self, texts: list):
        import time
        import random
+
        texts = [truncate(t, 8196) for t in texts]
        batch_size = 16
        ress = []
@ -495,6 +497,7 @@ class MistralEmbed(Base):
    def encode_queries(self, text):
        import time
        import random
+
        retry_max = 5
        while retry_max > 0:
            try:
@ -659,7 +662,7 @@ class OpenAI_APIEmbed(OpenAIEmbed):
    def __init__(self, key, model_name, base_url):
        if not base_url:
            raise ValueError("url cannot be None")
-        base_url = urljoin(base_url, "v1")
+        #base_url = urljoin(base_url, "v1")
        self.client = OpenAI(api_key=key, base_url=base_url)
        self.model_name = model_name.split("___")[0]

@ -751,6 +754,10 @@ class SILICONFLOWEmbed(Base):
        token_count = 0
        for i in range(0, len(texts), batch_size):
            texts_batch = texts[i : i + batch_size]
+            if self.model_name in ["BAAI/bge-large-zh-v1.5", "BAAI/bge-large-en-v1.5"]:
+                # limit 512, 340 is almost safe
+                texts_batch = [" " if not text.strip() else truncate(text, 340) for text in texts_batch]
+            else:
                texts_batch = [" " if not text.strip() else text for text in texts_batch]

            payload = {
@ -938,6 +945,7 @@ class GiteeEmbed(SILICONFLOWEmbed):
            base_url = "https://ai.gitee.com/v1/embeddings"
        super().__init__(key, model_name, base_url)

+
 class DeepInfraEmbed(OpenAIEmbed):
    _FACTORY_NAME = "DeepInfra"

@ -954,3 +962,12 @@ class Ai302Embed(Base):
        if not base_url:
            base_url = "https://api.302.ai/v1/embeddings"
        super().__init__(key, model_name, base_url)
+
+
+class CometEmbed(OpenAIEmbed):
+    _FACTORY_NAME = "CometAPI"
+
+    def __init__(self, key, model_name, base_url="https://api.cometapi.com/v1"):
+        if not base_url:
+            base_url = "https://api.cometapi.com/v1"
+        super().__init__(key, model_name, base_url)
--- a/rag/llm/sequence2txt_model.py
+++ b/rag/llm/sequence2txt_model.py
@ -218,7 +218,7 @@ class GPUStackSeq2txt(Base):
 class GiteeSeq2txt(Base):
    _FACTORY_NAME = "GiteeAI"

-    def __init__(self, key, model_name="whisper-1", base_url="https://ai.gitee.com/v1/"):
+    def __init__(self, key, model_name="whisper-1", base_url="https://ai.gitee.com/v1/", **kwargs):
        if not base_url:
            base_url = "https://ai.gitee.com/v1/"
        self.client = OpenAI(api_key=key, base_url=base_url)
@ -234,3 +234,13 @@ class DeepInfraSeq2txt(Base):

        self.client = OpenAI(api_key=key, base_url=base_url)
        self.model_name = model_name
+        
+        
+class CometSeq2txt(Base):
+    _FACTORY_NAME = "CometAPI"
+
+    def __init__(self, key, model_name="whisper-1", base_url="https://api.cometapi.com/v1", **kwargs):
+        if not base_url:
+            base_url = "https://api.cometapi.com/v1"
+        self.client = OpenAI(api_key=key, base_url=base_url)
+        self.model_name = model_name
--- a/rag/llm/tts_model.py
+++ b/rag/llm/tts_model.py
@ -394,3 +394,11 @@ class DeepInfraTTS(OpenAITTS):
        if not base_url:
            base_url = "https://api.deepinfra.com/v1/openai"
        super().__init__(key, model_name, base_url, **kwargs)
+        
+class CometAPITTS(OpenAITTS):
+    _FACTORY_NAME = "CometAPI"
+
+    def __init__(self, key, model_name, base_url="https://api.cometapi.com/v1", **kwargs):
+        if not base_url:
+            base_url = "https://api.cometapi.com/v1"
+        super().__init__(key, model_name, base_url, **kwargs)
--- a/rag/prompts/prompts.py
+++ b/rag/prompts/prompts.py
@ -437,3 +437,216 @@ def gen_meta_filter(chat_mdl, meta_data:dict, query: str) -> list:
    except Exception:
        logging.exception(f"Loading json failure: {ans}")
    return []
+
+
+def gen_json(system_prompt:str, user_prompt:str, chat_mdl):
+    _, msg = message_fit_in(form_message(system_prompt, user_prompt), chat_mdl.max_length)
+    ans = chat_mdl.chat(msg[0]["content"], msg[1:])
+    ans = re.sub(r"(^.*</think>|```json\n|```\n*$)", "", ans, flags=re.DOTALL)
+    try:
+        return json_repair.loads(ans)
+    except Exception:
+        logging.exception(f"Loading json failure: {ans}")
+
+
+TOC_DETECTION = load_prompt("toc_detection")
+def detect_table_of_contents(page_1024:list[str], chat_mdl):
+    toc_secs = []
+    for i, sec in enumerate(page_1024[:22]):
+        ans = gen_json(PROMPT_JINJA_ENV.from_string(TOC_DETECTION).render(page_txt=sec), "Only JSON please.", chat_mdl)
+        if toc_secs and not ans["exists"]:
+            break
+        toc_secs.append(sec)
+    return toc_secs
+
+
+TOC_EXTRACTION = load_prompt("toc_extraction")
+TOC_EXTRACTION_CONTINUE = load_prompt("toc_extraction_continue")
+def extract_table_of_contents(toc_pages, chat_mdl):
+    if not toc_pages:
+        return []
+
+    return gen_json(PROMPT_JINJA_ENV.from_string(TOC_EXTRACTION).render(toc_page="\n".join(toc_pages)), "Only JSON please.", chat_mdl)
+
+
+def toc_index_extractor(toc:list[dict], content:str, chat_mdl):
+    tob_extractor_prompt = """
+    You are given a table of contents in a json format and several pages of a document, your job is to add the physical_index to the table of contents in the json format.
+
+    The provided pages contains tags like <physical_index_X> and <physical_index_X> to indicate the physical location of the page X.
+
+    The structure variable is the numeric system which represents the index of the hierarchy section in the table of contents. For example, the first section has structure index 1, the first subsection has structure index 1.1, the second subsection has structure index 1.2, etc.
+
+    The response should be in the following JSON format: 
+    [
+        {
+            "structure": <structure index, "x.x.x" or None> (string),
+            "title": <title of the section>,
+            "physical_index": "<physical_index_X>" (keep the format)
+        },
+        ...
+    ]
+
+    Only add the physical_index to the sections that are in the provided pages.
+    If the title of the section are not in the provided pages, do not add the physical_index to it.
+    Directly return the final JSON structure. Do not output anything else."""
+
+    prompt = tob_extractor_prompt + '\nTable of contents:\n' + json.dumps(toc, ensure_ascii=False, indent=2) + '\nDocument pages:\n' + content
+    return gen_json(prompt, "Only JSON please.", chat_mdl)
+
+
+TOC_INDEX = load_prompt("toc_index")
+def table_of_contents_index(toc_arr: list[dict], sections: list[str], chat_mdl):
+    if not toc_arr or not sections:
+        return []
+
+    toc_map = {}
+    for i, it in enumerate(toc_arr):
+        k1 = (it["structure"]+it["title"]).replace(" ", "")
+        k2 = it["title"].strip()
+        if k1 not in toc_map:
+            toc_map[k1] = []
+        if k2 not in toc_map:
+            toc_map[k2] = []
+        toc_map[k1].append(i)
+        toc_map[k2].append(i)
+
+    for it in toc_arr:
+        it["indices"] = []
+    for i, sec in enumerate(sections):
+        sec = sec.strip()
+        if sec.replace(" ", "") in toc_map:
+            for j in toc_map[sec.replace(" ", "")]:
+                toc_arr[j]["indices"].append(i)
+
+    all_pathes = []
+    def dfs(start, path):
+        nonlocal all_pathes
+        if start >= len(toc_arr):
+            if path:
+                all_pathes.append(path)
+            return
+        if not toc_arr[start]["indices"]:
+            dfs(start+1, path)
+            return
+        added = False
+        for j in toc_arr[start]["indices"]:
+            if path and j < path[-1][0]:
+                continue
+            _path = deepcopy(path)
+            _path.append((j, start))
+            added = True
+            dfs(start+1, _path)
+        if not added and path:
+            all_pathes.append(path)
+
+    dfs(0, [])
+    path = max(all_pathes, key=lambda x:len(x))
+    for it in toc_arr:
+        it["indices"] = []
+    for j, i in path:
+        toc_arr[i]["indices"] = [j]
+    print(json.dumps(toc_arr, ensure_ascii=False, indent=2))
+
+    i = 0
+    while i < len(toc_arr):
+        it  = toc_arr[i]
+        if it["indices"]:
+            i += 1
+            continue
+
+        if i>0 and toc_arr[i-1]["indices"]:
+            st_i = toc_arr[i-1]["indices"][-1]
+        else:
+            st_i = 0
+        e = i + 1
+        while e <len(toc_arr) and not toc_arr[e]["indices"]:
+            e += 1
+        if e >= len(toc_arr):
+            e = len(sections)
+        else:
+            e = toc_arr[e]["indices"][0]
+
+        for j in range(st_i, min(e+1, len(sections))):
+            ans = gen_json(PROMPT_JINJA_ENV.from_string(TOC_INDEX).render(
+                structure=it["structure"],
+                title=it["title"],
+                text=sections[j]), "Only JSON please.", chat_mdl)
+            if ans["exist"] == "yes":
+                it["indices"].append(j)
+                break
+
+        i += 1
+
+    return toc_arr
+
+
+def check_if_toc_transformation_is_complete(content, toc, chat_mdl):
+    prompt = """
+    You are given a raw table of contents and a  table of contents.
+    Your job is to check if the  table of contents is complete.
+
+    Reply format:
+    {{
+        "thinking": <why do you think the cleaned table of contents is complete or not>
+        "completed": "yes" or "no"
+    }}
+    Directly return the final JSON structure. Do not output anything else."""
+
+    prompt = prompt + '\n Raw Table of contents:\n' + content + '\n Cleaned Table of contents:\n' + toc
+    response = gen_json(prompt, "Only JSON please.", chat_mdl)
+    return response['completed']
+
+
+def toc_transformer(toc_pages, chat_mdl):
+    init_prompt = """
+    You are given a table of contents, You job is to transform the whole table of content into a JSON format included table_of_contents.
+
+    The `structure` is the numeric system which represents the index of the hierarchy section in the table of contents. For example, the first section has structure index 1, the first subsection has structure index 1.1, the second subsection has structure index 1.2, etc.
+    The `title` is a short phrase or a several-words term.
+    
+    The response should be in the following JSON format: 
+    [
+        {
+            "structure": <structure index, "x.x.x" or None> (string),
+            "title": <title of the section>
+        },
+        ...
+    ],
+    You should transform the full table of contents in one go.
+    Directly return the final JSON structure, do not output anything else. """
+
+    toc_content = "\n".join(toc_pages)
+    prompt = init_prompt + '\n Given table of contents\n:' + toc_content
+    def clean_toc(arr):
+        for a in arr:
+            a["title"] = re.sub(r"[.·….]{2,}", "", a["title"])
+    last_complete = gen_json(prompt, "Only JSON please.", chat_mdl)
+    if_complete = check_if_toc_transformation_is_complete(toc_content, json.dumps(last_complete, ensure_ascii=False, indent=2), chat_mdl)
+    clean_toc(last_complete)
+    if if_complete == "yes":
+        return last_complete
+
+    while not (if_complete == "yes"):
+        prompt = f"""
+        Your task is to continue the table of contents json structure, directly output the remaining part of the json structure.
+        The response should be in the following JSON format: 
+
+        The raw table of contents json structure is:
+        {toc_content}
+
+        The incomplete transformed table of contents json structure is:
+        {json.dumps(last_complete[-24:], ensure_ascii=False, indent=2)}
+
+        Please continue the json structure, directly output the remaining part of the json structure."""
+        new_complete = gen_json(prompt, "Only JSON please.", chat_mdl)
+        if not new_complete or str(last_complete).find(str(new_complete)) >= 0:
+            break
+        clean_toc(new_complete)
+        last_complete.extend(new_complete)
+        if_complete = check_if_toc_transformation_is_complete(toc_content, json.dumps(last_complete, ensure_ascii=False, indent=2), chat_mdl)
+
+    return last_complete
+
+
+
--- a/rag/prompts/toc_detection.md
+++ b/rag/prompts/toc_detection.md
@ -0,0 +1,29 @@
+You are an AI assistant designed to analyze text content and detect whether a table of contents (TOC) list exists on the given page. Follow these steps:  
+
+1. **Analyze the Input**: Carefully review the provided text content.  
+2. **Identify Key Features**: Look for common indicators of a TOC, such as:  
+   - Section titles or headings paired with page numbers.
+   - Patterns like repeated formatting (e.g., bold/italicized text, dots/dashes between titles and numbers).  
+   - Phrases like "Table of Contents," "Contents," or similar headings.  
+   - Logical grouping of topics/subtopics with sequential page references.  
+3. **Discern Negative  Features**:
+   - The text contains no numbers, or the numbers present are clearly not page references (e.g., dates, statistical figures, phone numbers, version numbers).
+   - The text consists of full, descriptive sentences and paragraphs that form a narrative, present arguments, or explain concepts, rather than succinctly listing topics.
+   - Contains citations with authors, publication years, journal titles, and page ranges (e.g., "Smith, J. (2020). Journal Title, 10(2), 45-67.").
+   - Lists keywords or terms followed by multiple page numbers, often in alphabetical order.
+   - Comprises terms followed by their definitions or explanations.
+   - Labeled with headers like "Appendix A," "Appendix B," etc.
+   - Contains expressive language thanking individuals or organizations for their support or contributions.
+4. **Evaluate Evidence**: Weigh the presence/absence of these features to determine if the content resembles a TOC.
+5. **Output Format**: Provide your response in the following JSON structure:  
+   ```json  
+   {  
+     "reasoning": "Step-by-step explanation of your analysis based on the features identified." ,
+     "exists": true/false
+   }  
+   ```  
+6. **DO NOT** output anything else except JSON structure.
+
+**Input text Content ( Text-Only Extraction ):**  
+{{ page_txt }} 
+
--- a/rag/prompts/toc_extraction.md
+++ b/rag/prompts/toc_extraction.md
@ -0,0 +1,53 @@
+You are an expert parser and data formatter. Your task is to analyze the provided table of contents (TOC) text and convert it into a valid JSON array of objects.
+
+**Instructions:**
+1.  Analyze each line of the input TOC.
+2.  For each line, extract the following three pieces of information:
+    *   `structure`: The hierarchical index/numbering (e.g., "1", "2.1", "3.2.5", "A.1"). If a line has no visible numbering or structure indicator (like a main "Chapter" title), use `null`.
+    *   `title`: The textual title of the section or chapter. This should be the main descriptive text, clean and without the page number.
+3.  Output **only** a valid JSON array. Do not include any other text, explanations, or markdown code block fences (like ```json) in your response.
+
+**JSON Format:**
+The output must be a list of objects following this exact schema:
+```json
+[
+    {
+        "structure": <structure index, "x.x.x" or None> (string）,
+        "title": <title of the section>
+    },
+    ...
+]
+```
+
+**Input Example:**
+```
+Contents
+1 Introduction to the System ... 1
+1.1 Overview .... 2
+1.2 Key Features .... 5
+2 Installation Guide ....8
+2.1 Prerequisites ........ 9
+2.2 Step-by-Step Process ........ 12
+Appendix A: Specifications ..... 45
+References ... 47
+```
+
+**Expected Output For The Example:**
+```json
+[
+    {"structure": null, "title": "Contents"},
+    {"structure": "1", "title": "Introduction to the System"},
+    {"structure": "1.1", "title": "Overview"},
+    {"structure": "1.2", "title": "Key Features"},
+    {"structure": "2", "title": "Installation Guide"},
+    {"structure": "2.1", "title": "Prerequisites"},
+    {"structure": "2.2", "title": "Step-by-Step Process"},
+    {"structure": "A", "title": "Specifications"},
+    {"structure": null, "title": "References"}
+]
+```
+
+**Now, process the following TOC input:**
+```
+{{ toc_page }}
+```
--- a/rag/prompts/toc_extraction_continue.md
+++ b/rag/prompts/toc_extraction_continue.md
@ -0,0 +1,60 @@
+You are an expert parser and data formatter, currently in the process of building a JSON array from a multi-page table of contents (TOC). Your task is to analyze the new page of content and **append** the new entries to the existing JSON array.
+
+**Instructions:**
+1.  You will be given two inputs:
+    *   `current_page_text`: The text content from the new page of the TOC.
+    *   `existing_json`: The valid JSON array you have generated from the previous pages.
+2.  Analyze each line of the `current_page_text` input.
+3.  For each new line, extract the following three pieces of information:
+    *   `structure`: The hierarchical index/numbering (e.g., "1", "2.1", "3.2.5"). Use `null` if none exists.
+    *   `title`: The clean textual title of the section or chapter.
+    *   `page`: The page number on which the section starts. Extract only the number. Use `null` if not present.
+4.  **Append these new entries** to the `existing_json` array. Do not modify, reorder, or delete any of the existing entries.
+5.  Output **only** the complete, updated JSON array. Do not include any other text, explanations, or markdown code block fences (like ```json).
+
+**JSON Format:**
+The output must be a valid JSON array following this schema:
+```json
+[
+    {
+        "structure": <string or null>,
+        "title": <string>,
+        "page": <number or null>
+    },
+    ...
+]
+```
+
+**Input Example:**
+`current_page_text`:
+```
+3.2 Advanced Configuration ........... 25
+3.3 Troubleshooting .................. 28
+4 User Management .................... 30
+```
+
+`existing_json`:
+```json
+[
+    {"structure": "1", "title": "Introduction", "page": 1},
+    {"structure": "2", "title": "Installation", "page": 5},
+    {"structure": "3", "title": "Configuration", "page": 12},
+    {"structure": "3.1", "title": "Basic Setup", "page": 15}
+]
+```
+
+**Expected Output For The Example:**
+```json
+[
+    {"structure": "3.2", "title": "Advanced Configuration", "page": 25},
+    {"structure": "3.3", "title": "Troubleshooting", "page": 28},
+    {"structure": "4", "title": "User Management", "page": 30}
+]
+```
+
+**Now, process the following inputs:**
+`current_page_text`:
+{{ toc_page }}
+
+`existing_json`:
+{{ toc_json }}
--- a/rag/prompts/toc_index.md
+++ b/rag/prompts/toc_index.md
@ -0,0 +1,20 @@
+You are an expert analyst tasked with matching text content to the title.
+
+**Instructions:**
+1. Analyze the given title with its numeric structure index and the provided text.
+2. Determine whether the title is mentioned as a section tile in the given text.
+3. Provide a concise, step-by-step reasoning for your decision.
+4. Output **only** the complete JSON object. Do not include any other text, explanations, or markdown code block fences (like ```json).
+
+**Output Format:**
+Your output must be a valid JSON object with the following keys:
+{
+"reasoning": "Step-by-step explanation of your analysis.",
+"exist": "<yes or no>",
+}
+
+** The title: **
+{{ structure }} {{ title }}
+
+** Given text: **
+{{ text }}
--- a/rag/svr/task_executor.py
+++ b/rag/svr/task_executor.py
@ -23,6 +23,7 @@ import time

 from api.utils import get_uuid
 from api.utils.api_utils import timeout
+from api.utils.base64_image import image2id
 from api.utils.log_utils import init_root_logger, get_project_base_directory
 from graphrag.general.index import run_graphrag
 from graphrag.utils import get_llm_cache, set_llm_cache, get_tags_from_cache, set_tags_to_cache
@ -37,7 +38,6 @@ import xxhash
 import copy
 import re
 from functools import partial
-from io import BytesIO
 from multiprocessing.context import TimeoutError
 from timeit import default_timer as timer
 import tracemalloc
@ -301,29 +301,7 @@ async def build_chunks(task, progress_callback):
                d["img_id"] = ""
                docs.append(d)
                return
-
-            with BytesIO() as output_buffer:
-                if isinstance(d["image"], bytes):
-                    output_buffer.write(d["image"])
-                    output_buffer.seek(0)
-                else:
-                    # If the image is in RGBA mode, convert it to RGB mode before saving it in JPEG format.
-                    if d["image"].mode in ("RGBA", "P"):
-                        converted_image = d["image"].convert("RGB")
-                        #d["image"].close()  # Close original image
-                        d["image"] = converted_image
-                    try:
-                        d["image"].save(output_buffer, format='JPEG')
-                    except OSError as e:
-                        logging.warning(
-                            "Saving image of chunk {}/{}/{} got exception, ignore: {}".format(task["location"], task["name"], d["id"], str(e)))
-
-                async with minio_limiter:
-                    await trio.to_thread.run_sync(lambda: STORAGE_IMPL.put(task["kb_id"], d["id"], output_buffer.getvalue()))
-                d["img_id"] = "{}-{}".format(task["kb_id"], d["id"])
-                if not isinstance(d["image"], bytes):
-                    d["image"].close()
-                del d["image"]  # Remove image reference
+            await image2id(d, partial(STORAGE_IMPL.put), task["kb_id"], d["id"])
            docs.append(d)
        except Exception:
            logging.exception(
--- a/sandbox/sandbox_base_image/nodejs/package-lock.json
+++ b/sandbox/sandbox_base_image/nodejs/package-lock.json
@ -14,24 +14,24 @@
    },
    "node_modules/asynckit": {
      "version": "0.4.0",
-      "resolved": "https://registry.npmmirror.com/asynckit/-/asynckit-0.4.0.tgz",
+      "resolved": "https://registry.npmjs.org/asynckit/-/asynckit-0.4.0.tgz",
      "integrity": "sha512-Oei9OH4tRh0YqU3GxhX79dM/mwVgvbZJaSNaRk+bshkj0S5cfHcgYakreBjrHwatXKbz+IoIdYLxrKim2MjW0Q==",
      "license": "MIT"
    },
    "node_modules/axios": {
-      "version": "1.9.0",
-      "resolved": "https://registry.npmmirror.com/axios/-/axios-1.9.0.tgz",
-      "integrity": "sha512-re4CqKTJaURpzbLHtIi6XpDv20/CnpXOtjRY5/CU32L8gU8ek9UIivcfvSWvmKEngmVbrUtPpdDwWDWL7DNHvg==",
+      "version": "1.12.0",
+      "resolved": "https://registry.npmjs.org/axios/-/axios-1.12.0.tgz",
+      "integrity": "sha512-oXTDccv8PcfjZmPGlWsPSwtOJCZ/b6W5jAMCNcfwJbCzDckwG0jrYJFaWH1yvivfCXjVzV/SPDEhMB3Q+DSurg==",
      "license": "MIT",
      "dependencies": {
        "follow-redirects": "^1.15.6",
-        "form-data": "^4.0.0",
+        "form-data": "^4.0.4",
        "proxy-from-env": "^1.1.0"
      }
    },
    "node_modules/call-bind-apply-helpers": {
      "version": "1.0.2",
-      "resolved": "https://registry.npmmirror.com/call-bind-apply-helpers/-/call-bind-apply-helpers-1.0.2.tgz",
+      "resolved": "https://registry.npmjs.org/call-bind-apply-helpers/-/call-bind-apply-helpers-1.0.2.tgz",
      "integrity": "sha512-Sp1ablJ0ivDkSzjcaJdxEunN5/XvksFJ2sMBFfq6x0ryhQV/2b/KwFe21cMpmHtPOSij8K99/wSfoEuTObmuMQ==",
      "license": "MIT",
      "dependencies": {
@ -44,7 +44,7 @@
    },
    "node_modules/combined-stream": {
      "version": "1.0.8",
-      "resolved": "https://registry.npmmirror.com/combined-stream/-/combined-stream-1.0.8.tgz",
+      "resolved": "https://registry.npmjs.org/combined-stream/-/combined-stream-1.0.8.tgz",
      "integrity": "sha512-FQN4MRfuJeHf7cBbBMJFXhKSDq+2kAArBlmRBvcvFE5BB1HZKXtSFASDhdlz9zOYwxh8lDdnvmMOe/+5cdoEdg==",
      "license": "MIT",
      "dependencies": {
@ -56,7 +56,7 @@
    },
    "node_modules/delayed-stream": {
      "version": "1.0.0",
-      "resolved": "https://registry.npmmirror.com/delayed-stream/-/delayed-stream-1.0.0.tgz",
+      "resolved": "https://registry.npmjs.org/delayed-stream/-/delayed-stream-1.0.0.tgz",
      "integrity": "sha512-ZySD7Nf91aLB0RxL4KGrKHBXl7Eds1DAmEdcoVawXnLD7SDhpNgtuII2aAkg7a7QS41jxPSZ17p4VdGnMHk3MQ==",
      "license": "MIT",
      "engines": {
@ -65,7 +65,7 @@
    },
    "node_modules/dunder-proto": {
      "version": "1.0.1",
-      "resolved": "https://registry.npmmirror.com/dunder-proto/-/dunder-proto-1.0.1.tgz",
+      "resolved": "https://registry.npmjs.org/dunder-proto/-/dunder-proto-1.0.1.tgz",
      "integrity": "sha512-KIN/nDJBQRcXw0MLVhZE9iQHmG68qAVIBg9CqmUYjmQIhgij9U5MFvrqkUL5FbtyyzZuOeOt0zdeRe4UY7ct+A==",
      "license": "MIT",
      "dependencies": {
@ -79,7 +79,7 @@
    },
    "node_modules/es-define-property": {
      "version": "1.0.1",
-      "resolved": "https://registry.npmmirror.com/es-define-property/-/es-define-property-1.0.1.tgz",
+      "resolved": "https://registry.npmjs.org/es-define-property/-/es-define-property-1.0.1.tgz",
      "integrity": "sha512-e3nRfgfUZ4rNGL232gUgX06QNyyez04KdjFrF+LTRoOXmrOgFKDg4BCdsjW8EnT69eqdYGmRpJwiPVYNrCaW3g==",
      "license": "MIT",
      "engines": {
@ -88,7 +88,7 @@
    },
    "node_modules/es-errors": {
      "version": "1.3.0",
-      "resolved": "https://registry.npmmirror.com/es-errors/-/es-errors-1.3.0.tgz",
+      "resolved": "https://registry.npmjs.org/es-errors/-/es-errors-1.3.0.tgz",
      "integrity": "sha512-Zf5H2Kxt2xjTvbJvP2ZWLEICxA6j+hAmMzIlypy4xcBg1vKVnx89Wy0GbS+kf5cwCVFFzdCFh2XSCFNULS6csw==",
      "license": "MIT",
      "engines": {
@ -97,7 +97,7 @@
    },
    "node_modules/es-object-atoms": {
      "version": "1.1.1",
-      "resolved": "https://registry.npmmirror.com/es-object-atoms/-/es-object-atoms-1.1.1.tgz",
+      "resolved": "https://registry.npmjs.org/es-object-atoms/-/es-object-atoms-1.1.1.tgz",
      "integrity": "sha512-FGgH2h8zKNim9ljj7dankFPcICIK9Cp5bm+c2gQSYePhpaG5+esrLODihIorn+Pe6FGJzWhXQotPv73jTaldXA==",
      "license": "MIT",
      "dependencies": {
@ -109,7 +109,7 @@
    },
    "node_modules/es-set-tostringtag": {
      "version": "2.1.0",
-      "resolved": "https://registry.npmmirror.com/es-set-tostringtag/-/es-set-tostringtag-2.1.0.tgz",
+      "resolved": "https://registry.npmjs.org/es-set-tostringtag/-/es-set-tostringtag-2.1.0.tgz",
      "integrity": "sha512-j6vWzfrGVfyXxge+O0x5sh6cvxAog0a/4Rdd2K36zCMV5eJ+/+tOAngRO8cODMNWbVRdVlmGZQL2YS3yR8bIUA==",
      "license": "MIT",
      "dependencies": {
@ -143,14 +143,15 @@
      }
    },
    "node_modules/form-data": {
-      "version": "4.0.2",
-      "resolved": "https://registry.npmmirror.com/form-data/-/form-data-4.0.2.tgz",
-      "integrity": "sha512-hGfm/slu0ZabnNt4oaRZ6uREyfCj6P4fT/n6A1rGV+Z0VdGXjfOhVUpkn6qVQONHGIFwmveGXyDs75+nr6FM8w==",
+      "version": "4.0.4",
+      "resolved": "https://registry.npmjs.org/form-data/-/form-data-4.0.4.tgz",
+      "integrity": "sha512-KrGhL9Q4zjj0kiUt5OO4Mr/A/jlI2jDYs5eHBpYHPcBEVSiipAvn2Ko2HnPe20rmcuuvMHNdZFp+4IlGTMF0Ow==",
      "license": "MIT",
      "dependencies": {
        "asynckit": "^0.4.0",
        "combined-stream": "^1.0.8",
        "es-set-tostringtag": "^2.1.0",
+        "hasown": "^2.0.2",
        "mime-types": "^2.1.12"
      },
      "engines": {
@ -159,7 +160,7 @@
    },
    "node_modules/function-bind": {
      "version": "1.1.2",
-      "resolved": "https://registry.npmmirror.com/function-bind/-/function-bind-1.1.2.tgz",
+      "resolved": "https://registry.npmjs.org/function-bind/-/function-bind-1.1.2.tgz",
      "integrity": "sha512-7XHNxH7qX9xG5mIwxkhumTox/MIRNcOgDrxWsMt2pAr23WHp6MrRlN7FBSFpCpr+oVO0F744iUgR82nJMfG2SA==",
      "license": "MIT",
      "funding": {
@ -168,7 +169,7 @@
    },
    "node_modules/get-intrinsic": {
      "version": "1.3.0",
-      "resolved": "https://registry.npmmirror.com/get-intrinsic/-/get-intrinsic-1.3.0.tgz",
+      "resolved": "https://registry.npmjs.org/get-intrinsic/-/get-intrinsic-1.3.0.tgz",
      "integrity": "sha512-9fSjSaos/fRIVIp+xSJlE6lfwhES7LNtKaCBIamHsjr2na1BiABJPo0mOjjz8GJDURarmCPGqaiVg5mfjb98CQ==",
      "license": "MIT",
      "dependencies": {
@ -192,7 +193,7 @@
    },
    "node_modules/get-proto": {
      "version": "1.0.1",
-      "resolved": "https://registry.npmmirror.com/get-proto/-/get-proto-1.0.1.tgz",
+      "resolved": "https://registry.npmjs.org/get-proto/-/get-proto-1.0.1.tgz",
      "integrity": "sha512-sTSfBjoXBp89JvIKIefqw7U2CCebsc74kiY6awiGogKtoSGbgjYE/G/+l9sF3MWFPNc9IcoOC4ODfKHfxFmp0g==",
      "license": "MIT",
      "dependencies": {
@ -205,7 +206,7 @@
    },
    "node_modules/gopd": {
      "version": "1.2.0",
-      "resolved": "https://registry.npmmirror.com/gopd/-/gopd-1.2.0.tgz",
+      "resolved": "https://registry.npmjs.org/gopd/-/gopd-1.2.0.tgz",
      "integrity": "sha512-ZUKRh6/kUFoAiTAtTYPZJ3hw9wNxx+BIBOijnlG9PnrJsCcSjs1wyyD6vJpaYtgnzDrKYRSqf3OO6Rfa93xsRg==",
      "license": "MIT",
      "engines": {
@ -217,7 +218,7 @@
    },
    "node_modules/has-symbols": {
      "version": "1.1.0",
-      "resolved": "https://registry.npmmirror.com/has-symbols/-/has-symbols-1.1.0.tgz",
+      "resolved": "https://registry.npmjs.org/has-symbols/-/has-symbols-1.1.0.tgz",
      "integrity": "sha512-1cDNdwJ2Jaohmb3sg4OmKaMBwuC48sYni5HUw2DvsC8LjGTLK9h+eb1X6RyuOHe4hT0ULCW68iomhjUoKUqlPQ==",
      "license": "MIT",
      "engines": {
@ -229,7 +230,7 @@
    },
    "node_modules/has-tostringtag": {
      "version": "1.0.2",
-      "resolved": "https://registry.npmmirror.com/has-tostringtag/-/has-tostringtag-1.0.2.tgz",
+      "resolved": "https://registry.npmjs.org/has-tostringtag/-/has-tostringtag-1.0.2.tgz",
      "integrity": "sha512-NqADB8VjPFLM2V0VvHUewwwsw0ZWBaIdgo+ieHtK3hasLz4qeCRjYcqfB6AQrBggRKppKF8L52/VqdVsO47Dlw==",
      "license": "MIT",
      "dependencies": {
@ -244,7 +245,7 @@
    },
    "node_modules/hasown": {
      "version": "2.0.2",
-      "resolved": "https://registry.npmmirror.com/hasown/-/hasown-2.0.2.tgz",
+      "resolved": "https://registry.npmjs.org/hasown/-/hasown-2.0.2.tgz",
      "integrity": "sha512-0hJU9SCPvmMzIBdZFqNPXWa6dqh7WdH0cII9y+CyS8rG3nL48Bclra9HmKhVVUHyPWNH5Y7xDwAB7bfgSjkUMQ==",
      "license": "MIT",
      "dependencies": {
@ -256,7 +257,7 @@
    },
    "node_modules/math-intrinsics": {
      "version": "1.1.0",
-      "resolved": "https://registry.npmmirror.com/math-intrinsics/-/math-intrinsics-1.1.0.tgz",
+      "resolved": "https://registry.npmjs.org/math-intrinsics/-/math-intrinsics-1.1.0.tgz",
      "integrity": "sha512-/IXtbwEk5HTPyEwyKX6hGkYXxM9nbj64B+ilVJnC/R6B0pH5G4V3b0pVbL7DBj4tkhBAppbQUlf6F6Xl9LHu1g==",
      "license": "MIT",
      "engines": {
@ -265,7 +266,7 @@
    },
    "node_modules/mime-db": {
      "version": "1.52.0",
-      "resolved": "https://registry.npmmirror.com/mime-db/-/mime-db-1.52.0.tgz",
+      "resolved": "https://registry.npmjs.org/mime-db/-/mime-db-1.52.0.tgz",
      "integrity": "sha512-sPU4uV7dYlvtWJxwwxHD0PuihVNiE7TyAbQ5SWxDCB9mUYvOgroQOwYQQOKPJ8CIbE+1ETVlOoK1UC2nU3gYvg==",
      "license": "MIT",
      "engines": {
@ -274,7 +275,7 @@
    },
    "node_modules/mime-types": {
      "version": "2.1.35",
-      "resolved": "https://registry.npmmirror.com/mime-types/-/mime-types-2.1.35.tgz",
+      "resolved": "https://registry.npmjs.org/mime-types/-/mime-types-2.1.35.tgz",
      "integrity": "sha512-ZDY+bPm5zTTF+YpCrAU9nK0UgICYPT0QtT1NZWFv4s++TNkcgVaT0g6+4R2uI4MjQjzysHB1zxuWL50hzaeXiw==",
      "license": "MIT",
      "dependencies": {
--- a/web/src/assets/svg/llm/cometapi.svg
+++ b/web/src/assets/svg/llm/cometapi.svg
--- a/web/src/assets/svg/llm/token-pony.svg
+++ b/web/src/assets/svg/llm/token-pony.svg
--- a/web/src/components/embed-dialog/index.tsx
+++ b/web/src/components/embed-dialog/index.tsx
@ -139,7 +139,7 @@ function EmbedDialog({
            </form>
          </Form>
          <div>
-            <span>Embed code</span>
+            <span>{t('embedCode', { keyPrefix: 'search' })}</span>
            <HightLightMarkdown>{text}</HightLightMarkdown>
          </div>
          <div className=" font-medium mt-4 mb-1">
--- a/web/src/constants/llm.ts
+++ b/web/src/constants/llm.ts
@ -54,7 +54,9 @@ export enum LLMFactory {
  DeepInfra = 'DeepInfra',
  Grok = 'Grok',
  XAI = 'xAI',
+  TokenPony = 'TokenPony',
  Meituan = 'Meituan',
+  CometAPI = 'CometAPI',
 }

 // Please lowercase the file name
@ -114,5 +116,7 @@ export const IconMap = {
  [LLMFactory.DeepInfra]: 'deepinfra',
  [LLMFactory.Grok]: 'grok',
  [LLMFactory.XAI]: 'xai',
+  [LLMFactory.TokenPony]: 'token-pony',
  [LLMFactory.Meituan]: 'longcat',
+  [LLMFactory.CometAPI]: 'cometapi',
 };
--- a/web/src/hooks/llm-hooks.tsx
+++ b/web/src/hooks/llm-hooks.tsx
@ -136,6 +136,7 @@ export const useSelectLlmOptionsByModelType = () => {
  };
 };

+// Merge different types of models from the same manufacturer under one manufacturer
 export const useComposeLlmOptionsByModelTypes = (
  modelTypes: LlmModelType[],
 ) => {
@ -155,7 +156,12 @@ export const useComposeLlmOptionsByModelTypes = (
    options.forEach((x) => {
      const item = pre.find((y) => y.label === x.label);
      if (item) {
-        item.options.push(...x.options);
+        x.options.forEach((y) => {
+          // A model that is both an image2text and speech2text model
+          if (!item.options.some((z) => z.value === y.value)) {
+            item.options.push(y);
+          }
+        });
      } else {
        pre.push(x);
      }
--- a/web/src/locales/zh.ts
+++ b/web/src/locales/zh.ts
@ -155,7 +155,7 @@ export default {
      similarityThreshold: '相似度阈值',
      similarityThresholdTip:
        '我们使用混合相似度得分来评估两行文本之间的距离。 它是加权关键词相似度和向量余弦相似度。 如果查询和块之间的相似度小于此阈值，则该块将被过滤掉。默认设置为 0.2，也就是说文本块的混合相似度得分至少 20 才会被召回。',
-      vectorSimilarityWeight: '相似度相似度权重',
+      vectorSimilarityWeight: '向量相似度权重',
      vectorSimilarityWeightTip:
        '我们使用混合相似性评分来评估两行文本之间的距离。它是加权关键字相似性和矢量余弦相似性或rerank得分（0〜1）。两个权重的总和为1.0。',
      keywordSimilarityWeight: '关键词相似度权重',
@ -633,6 +633,8 @@ General：实体和关系提取提示来自 GitHub - microsoft/graphrag：基于
      },
      cancel: '取消',
      chatSetting: '聊天设置',
+      avatarHidden: '隐藏头像',
+      locale: '地区',
    },
    setting: {
      profile: '概要',
--- a/web/src/pages/agent/chat/box.tsx
+++ b/web/src/pages/agent/chat/box.tsx
@ -62,7 +62,7 @@ function AgentChatBox() {

  return (
    <>
-      <section className="flex flex-1 flex-col px-5 h-[90vh]">
+      <section className="flex flex-1 flex-col px-5 min-h-0 pb-4">
        <div className="flex-1 overflow-auto" ref={messageContainerRef}>
          <div>
            {/* <Spin spinning={sendLoading}> */}
--- a/web/src/pages/agent/chat/chat-sheet.tsx
+++ b/web/src/pages/agent/chat/chat-sheet.tsx
@ -9,7 +9,7 @@ export function ChatSheet({ hideModal }: IModalProps<any>) {
  return (
    <Sheet open modal={false} onOpenChange={hideModal}>
      <SheetContent
-        className={cn('top-20 p-0')}
+        className={cn('top-20 bottom-0 p-0 flex flex-col h-auto')}
        onInteractOutside={(e) => e.preventDefault()}
      >
        <SheetTitle className="hidden"></SheetTitle>
--- a/web/src/pages/agent/form/agent-form/index.tsx
+++ b/web/src/pages/agent/form/agent-form/index.tsx
@ -145,7 +145,7 @@ function AgentForm({ node }: INextOperatorForm) {
                  <PromptEditor
                    {...field}
                    placeholder={t('flow.messagePlaceholder')}
-                    showToolbar={false}
+                    showToolbar={true}
                    extraOptions={extraOptions}
                  ></PromptEditor>
                </FormControl>
@ -166,7 +166,7 @@ function AgentForm({ node }: INextOperatorForm) {
                    <section>
                      <PromptEditor
                        {...field}
-                        showToolbar={false}
+                        showToolbar={true}
                      ></PromptEditor>
                    </section>
                  </FormControl>
--- a/web/src/pages/agent/options.ts
+++ b/web/src/pages/agent/options.ts
@ -2133,7 +2133,7 @@ export const QWeatherTimePeriodOptions = [
  '30d',
 ];

-export const ExeSQLOptions = ['mysql', 'postgresql', 'mariadb', 'mssql'].map(
+export const ExeSQLOptions = ['mysql', 'postgres', 'mariadb', 'mssql'].map(
  (x) => ({
    label: upperFirst(x),
    value: x,
--- a/web/src/pages/data-flow/options.ts
+++ b/web/src/pages/data-flow/options.ts
@ -2133,7 +2133,7 @@ export const QWeatherTimePeriodOptions = [
  '30d',
 ];

-export const ExeSQLOptions = ['mysql', 'postgresql', 'mariadb', 'mssql'].map(
+export const ExeSQLOptions = ['mysql', 'postgres', 'mariadb', 'mssql'].map(
  (x) => ({
    label: upperFirst(x),
    value: x,
--- a/web/src/pages/dataset/sidebar/index.tsx
+++ b/web/src/pages/dataset/sidebar/index.tsx
@ -9,13 +9,7 @@ import { cn, formatBytes } from '@/lib/utils';
 import { Routes } from '@/routes';
 import { formatPureDate } from '@/utils/date';
 import { isEmpty } from 'lodash';
-import {
-  Banknote,
-  Database,
-  DatabaseZap,
-  FileSearch2,
-  GitGraph,
-} from 'lucide-react';
+import { Banknote, Database, FileSearch2, GitGraph } from 'lucide-react';
 import { useMemo } from 'react';
 import { useTranslation } from 'react-i18next';
 import { useHandleMenuClick } from './hooks';
@ -34,11 +28,11 @@ export function SideBar({ refreshCount }: PropType) {

  const items = useMemo(() => {
    const list = [
-      {
-        icon: DatabaseZap,
-        label: t(`knowledgeDetails.overview`),
-        key: Routes.DataSetOverview,
-      },
+      // {
+      //   icon: DatabaseZap,
+      //   label: t(`knowledgeDetails.overview`),
+      //   key: Routes.DataSetOverview,
+      // },
      {
        icon: Database,
        label: t(`knowledgeDetails.dataset`),
--- a/web/src/pages/datasets/dataset-creating-dialog.tsx
+++ b/web/src/pages/datasets/dataset-creating-dialog.tsx
@ -19,9 +19,10 @@ import { Input } from '@/components/ui/input';
 import { useNavigatePage } from '@/hooks/logic-hooks/navigate-hooks';
 import { IModalProps } from '@/interfaces/common';
 import { zodResolver } from '@hookform/resolvers/zod';
-import { useForm, useWatch } from 'react-hook-form';
+import { useForm } from 'react-hook-form';
 import { useTranslation } from 'react-i18next';
 import { z } from 'zod';
+
 import {
  ChunkMethodItem,
  EmbeddingModelItem,
@ -89,6 +90,7 @@ export function InputForm({ onOk }: IModalProps<any>) {
    console.log('submit', data);
    onOk?.(data);
  }
+
  const parseType = useWatch({
    control: form.control,
    name: 'parseType',
@ -121,6 +123,7 @@ export function InputForm({ onOk }: IModalProps<any>) {
            </FormItem>
          )}
        />
+
        <EmbeddingModelItem line={2} isEdit={false} />
        <ParseTypeItem />
        {parseType === 1 && (
--- a/web/src/pages/datasets/dataset-dataflow-creating-dialog.tsx
+++ b/web/src/pages/datasets/dataset-dataflow-creating-dialog.tsx
@ -0,0 +1,123 @@
+import { ButtonLoading } from '@/components/ui/button';
+import {
+  Dialog,
+  DialogContent,
+  DialogFooter,
+  DialogHeader,
+  DialogTitle,
+} from '@/components/ui/dialog';
+import {
+  Form,
+  FormControl,
+  FormField,
+  FormItem,
+  FormLabel,
+  FormMessage,
+} from '@/components/ui/form';
+import { Input } from '@/components/ui/input';
+import { IModalProps } from '@/interfaces/common';
+import { zodResolver } from '@hookform/resolvers/zod';
+import { useForm, useWatch } from 'react-hook-form';
+import { useTranslation } from 'react-i18next';
+import { z } from 'zod';
+import {
+  DataExtractKnowledgeItem,
+  DataFlowItem,
+  EmbeddingModelItem,
+  ParseTypeItem,
+  TeamItem,
+} from '../dataset/dataset-setting/configuration/common-item';
+
+const FormId = 'dataset-creating-form';
+
+export function InputForm({ onOk }: IModalProps<any>) {
+  const { t } = useTranslation();
+
+  const FormSchema = z.object({
+    name: z
+      .string()
+      .min(1, {
+        message: t('knowledgeList.namePlaceholder'),
+      })
+      .trim(),
+    parseType: z.number().optional(),
+  });
+
+  const form = useForm<z.infer<typeof FormSchema>>({
+    resolver: zodResolver(FormSchema),
+    defaultValues: {
+      name: '',
+      parseType: 1,
+    },
+  });
+
+  function onSubmit(data: z.infer<typeof FormSchema>) {
+    onOk?.(data.name);
+  }
+  const parseType = useWatch({
+    control: form.control,
+    name: 'parseType',
+  });
+  return (
+    <Form {...form}>
+      <form
+        onSubmit={form.handleSubmit(onSubmit)}
+        className="space-y-6"
+        id={FormId}
+      >
+        <FormField
+          control={form.control}
+          name="name"
+          render={({ field }) => (
+            <FormItem>
+              <FormLabel>
+                <span className="text-destructive mr-1"> *</span>
+                {t('knowledgeList.name')}
+              </FormLabel>
+              <FormControl>
+                <Input
+                  placeholder={t('knowledgeList.namePlaceholder')}
+                  {...field}
+                />
+              </FormControl>
+              <FormMessage />
+            </FormItem>
+          )}
+        />
+        <EmbeddingModelItem line={2} />
+        <ParseTypeItem />
+        {parseType === 2 && (
+          <>
+            <DataFlowItem />
+            <DataExtractKnowledgeItem />
+            <TeamItem />
+          </>
+        )}
+      </form>
+    </Form>
+  );
+}
+
+export function DatasetCreatingDialog({
+  hideModal,
+  onOk,
+  loading,
+}: IModalProps<any>) {
+  const { t } = useTranslation();
+
+  return (
+    <Dialog open onOpenChange={hideModal}>
+      <DialogContent className="sm:max-w-[425px]">
+        <DialogHeader>
+          <DialogTitle>{t('knowledgeList.createKnowledgeBase')}</DialogTitle>
+        </DialogHeader>
+        <InputForm onOk={onOk}></InputForm>
+        <DialogFooter>
+          <ButtonLoading type="submit" form={FormId} loading={loading}>
+            {t('common.save')}
+          </ButtonLoading>
+        </DialogFooter>
+      </DialogContent>
+    </Dialog>
+  );
+}
--- a/web/src/pages/flow/constant.tsx
+++ b/web/src/pages/flow/constant.tsx
@ -2911,7 +2911,7 @@ export const QWeatherTimePeriodOptions = [
  '30d',
 ];

-export const ExeSQLOptions = ['mysql', 'postgresql', 'mariadb', 'mssql'].map(
+export const ExeSQLOptions = ['mysql', 'postgres', 'mariadb', 'mssql'].map(
  (x) => ({
    label: upperFirst(x),
    value: x,
--- a/web/src/pages/user-setting/setting-model/ollama-modal/index.tsx
+++ b/web/src/pages/user-setting/setting-model/ollama-modal/index.tsx
@ -37,6 +37,7 @@ const llmFactoryToUrlMap = {
    'https://huggingface.co/docs/text-embeddings-inference/quick_tour',
  [LLMFactory.GPUStack]: 'https://docs.gpustack.ai/latest/quickstart',
  [LLMFactory.VLLM]: 'https://docs.vllm.ai/en/latest/',
+  [LLMFactory.TokenPony]: 'https://docs.tokenpony.cn/#/',
 };
 type LlmFactory = keyof typeof llmFactoryToUrlMap;

--- a/web/src/pages/user-setting/setting-model/system-model-setting-modal/index.tsx
+++ b/web/src/pages/user-setting/setting-model/system-model-setting-modal/index.tsx
@ -1,7 +1,10 @@
 import { IModalManagerChildrenProps } from '@/components/modal-manager';
 import { LlmModelType } from '@/constants/knowledge';
 import { useTranslate } from '@/hooks/common-hooks';
-import { ISystemModelSettingSavingParams } from '@/hooks/llm-hooks';
+import {
+  ISystemModelSettingSavingParams,
+  useComposeLlmOptionsByModelTypes,
+} from '@/hooks/llm-hooks';
 import { Form, Modal, Select } from 'antd';
 import { useEffect } from 'react';
 import { useFetchSystemModelSettingOnMount } from '../hooks';
@ -43,6 +46,11 @@ const SystemModelSettingModal = ({

  const onFormLayoutChange = () => {};

+  const modelOptions = useComposeLlmOptionsByModelTypes([
+    LlmModelType.Chat,
+    LlmModelType.Image2text,
+  ]);
+
  return (
    <Modal
      title={t('systemModelSettings')}
@ -58,14 +66,7 @@ const SystemModelSettingModal = ({
          name="llm_id"
          tooltip={t('chatModelTip')}
        >
-          <Select
-            options={[
-              ...allOptions[LlmModelType.Chat],
-              ...allOptions[LlmModelType.Image2text],
-            ]}
-            allowClear
-            showSearch
-          />
+          <Select options={modelOptions} allowClear showSearch />
        </Form.Item>
        <Form.Item
          label={t('embeddingModel')}
--- a/web/src/utils/common-util.ts
+++ b/web/src/utils/common-util.ts
@ -44,6 +44,7 @@ const orderFactoryList = [
  LLMFactory.Ollama,
  LLMFactory.Xinference,
  LLMFactory.Ai302,
+  LLMFactory.CometAPI,
 ];

 export const sortLLmFactoryListBySpecifiedOrder = (list: IFactory[]) => {