Fix errors detected by Ruff (#3918)

### What problem does this PR solve? Fix errors detected by Ruff ### Type of change - [x] Refactoring
2026-01-31 23:55:06 +08:00 · 2024-12-08 14:21:12 +08:00
parent e267a026f3
commit 0d68a6cd1b
97 changed files with 2558 additions and 1976 deletions
--- a/deepdoc/parser/resume/entities/corporations.py
+++ b/deepdoc/parser/resume/entities/corporations.py
@ -21,13 +21,18 @@ from . import regions


 current_file_path = os.path.dirname(os.path.abspath(__file__))
-GOODS = pd.read_csv(os.path.join(current_file_path, "res/corp_baike_len.csv"), sep="\t", header=0).fillna(0)
+GOODS = pd.read_csv(
+    os.path.join(current_file_path, "res/corp_baike_len.csv"), sep="\t", header=0
+).fillna(0)
 GOODS["cid"] = GOODS["cid"].astype(str)
 GOODS = GOODS.set_index(["cid"])
-CORP_TKS = json.load(open(os.path.join(current_file_path, "res/corp.tks.freq.json"), "r"))
+CORP_TKS = json.load(
+    open(os.path.join(current_file_path, "res/corp.tks.freq.json"), "r")
+)
 GOOD_CORP = json.load(open(os.path.join(current_file_path, "res/good_corp.json"), "r"))
 CORP_TAG = json.load(open(os.path.join(current_file_path, "res/corp_tag.json"), "r"))

+
 def baike(cid, default_v=0):
    global GOODS
    try:
@ -39,27 +44,41 @@ def baike(cid, default_v=0):

 def corpNorm(nm, add_region=True):
    global CORP_TKS
-    if not nm or type(nm)!=type(""):return ""
+    if not nm or isinstance(nm, str):
+        return ""
    nm = rag_tokenizer.tradi2simp(rag_tokenizer.strQ2B(nm)).lower()
    nm = re.sub(r"&amp;", "&", nm)
    nm = re.sub(r"[\(\)（）\+'\"\t \*\\【】-]+", " ", nm)
-    nm = re.sub(r"([—-]+.*| +co\..*|corp\..*| +inc\..*| +ltd.*)", "", nm, 10000, re.IGNORECASE)
-    nm = re.sub(r"(计算机|技术|(技术|科技|网络)*有限公司|公司|有限|研发中心|中国|总部)$", "", nm, 10000, re.IGNORECASE)
-    if not nm or (len(nm)<5 and not regions.isName(nm[0:2])):return nm
+    nm = re.sub(
+        r"([—-]+.*| +co\..*|corp\..*| +inc\..*| +ltd.*)", "", nm, 10000, re.IGNORECASE
+    )
+    nm = re.sub(
+        r"(计算机|技术|(技术|科技|网络)*有限公司|公司|有限|研发中心|中国|总部)$",
+        "",
+        nm,
+        10000,
+        re.IGNORECASE,
+    )
+    if not nm or (len(nm) < 5 and not regions.isName(nm[0:2])):
+        return nm

    tks = rag_tokenizer.tokenize(nm).split()
-    reg = [t for i,t in enumerate(tks) if regions.isName(t) and (t != "中国" or i > 0)]
+    reg = [t for i, t in enumerate(tks) if regions.isName(t) and (t != "中国" or i > 0)]
    nm = ""
    for t in tks:
-        if regions.isName(t) or t in CORP_TKS:continue
-        if re.match(r"[0-9a-zA-Z\\,.]+", t) and re.match(r".*[0-9a-zA-Z\,.]+$", nm):nm += " "
+        if regions.isName(t) or t in CORP_TKS:
+            continue
+        if re.match(r"[0-9a-zA-Z\\,.]+", t) and re.match(r".*[0-9a-zA-Z\,.]+$", nm):
+            nm += " "
        nm += t

    r = re.search(r"^([^a-z0-9 \(\)&]{2,})[a-z ]{4,}$", nm.strip())
-    if r:nm = r.group(1)
+    if r:
+        nm = r.group(1)
    r = re.search(r"^([a-z ]{3,})[^a-z0-9 \(\)&]{2,}$", nm.strip())
-    if r:nm = r.group(1)
-    return nm.strip() + (("" if not reg else "(%s)"%reg[0]) if add_region else "")
+    if r:
+        nm = r.group(1)
+    return nm.strip() + (("" if not reg else "(%s)" % reg[0]) if add_region else "")


 def rmNoise(n):
@ -67,33 +86,40 @@ def rmNoise(n):
    n = re.sub(r"[,. &（）()]+", "", n)
    return n

+
 GOOD_CORP = set([corpNorm(rmNoise(c), False) for c in GOOD_CORP])
-for c,v in CORP_TAG.items():
+for c, v in CORP_TAG.items():
    cc = corpNorm(rmNoise(c), False)
    if not cc:
        logging.debug(c)
-CORP_TAG = {corpNorm(rmNoise(c), False):v for c,v in CORP_TAG.items()}
+CORP_TAG = {corpNorm(rmNoise(c), False): v for c, v in CORP_TAG.items()}
+

 def is_good(nm):
    global GOOD_CORP
-    if nm.find("外派")>=0:return False
+    if nm.find("外派") >= 0:
+        return False
    nm = rmNoise(nm)
    nm = corpNorm(nm, False)
    for n in GOOD_CORP:
        if re.match(r"[0-9a-zA-Z]+$", n):
-            if n == nm: return True
-        elif nm.find(n)>=0:return True
+            if n == nm:
+                return True
+        elif nm.find(n) >= 0:
+            return True
    return False

+
 def corp_tag(nm):
    global CORP_TAG
    nm = rmNoise(nm)
    nm = corpNorm(nm, False)
    for n in CORP_TAG.keys():
        if re.match(r"[0-9a-zA-Z., ]+$", n):
-            if n == nm: return CORP_TAG[n]
-        elif nm.find(n)>=0:
-            if len(n)<3 and len(nm)/len(n)>=2:continue
+            if n == nm:
+                return CORP_TAG[n]
+        elif nm.find(n) >= 0:
+            if len(n) < 3 and len(nm) / len(n) >= 2:
+                continue
            return CORP_TAG[n]
    return []
-
--- a/deepdoc/parser/resume/entities/degrees.py
+++ b/deepdoc/parser/resume/entities/degrees.py
@ -11,27 +11,31 @@
 #  limitations under the License.
 #

-TBL = {"94":"EMBA",
-"6":"MBA",
-"95":"MPA",
-"92":"专升本",
-"4":"专科",
-"90":"中专",
-"91":"中技",
-"86":"初中",
-"3":"博士",
-"10":"博士后",
-"1":"本科",
-"2":"硕士",
-"87":"职高",
-"89":"高中"
+TBL = {
+    "94": "EMBA",
+    "6": "MBA",
+    "95": "MPA",
+    "92": "专升本",
+    "4": "专科",
+    "90": "中专",
+    "91": "中技",
+    "86": "初中",
+    "3": "博士",
+    "10": "博士后",
+    "1": "本科",
+    "2": "硕士",
+    "87": "职高",
+    "89": "高中",
 }

-TBL_ = {v:k for k,v in TBL.items()}
+TBL_ = {v: k for k, v in TBL.items()}
+

 def get_name(id):
    return TBL.get(str(id), "")

+
 def get_id(nm):
-    if not nm:return ""
+    if not nm:
+        return ""
    return TBL_.get(nm.upper().strip(), "")
--- a/deepdoc/parser/resume/entities/industries.py
+++ b/deepdoc/parser/resume/entities/industries.py
--- a/deepdoc/parser/resume/entities/regions.py
+++ b/deepdoc/parser/resume/entities/regions.py
--- a/deepdoc/parser/resume/entities/schools.py
+++ b/deepdoc/parser/resume/entities/schools.py
@ -16,8 +16,11 @@ import json
 import re
 import copy
 import pandas as pd
+
 current_file_path = os.path.dirname(os.path.abspath(__file__))
-TBL = pd.read_csv(os.path.join(current_file_path, "res/schools.csv"), sep="\t", header=0).fillna("")
+TBL = pd.read_csv(
+    os.path.join(current_file_path, "res/schools.csv"), sep="\t", header=0
+).fillna("")
 TBL["name_en"] = TBL["name_en"].map(lambda x: x.lower().strip())
 GOOD_SCH = json.load(open(os.path.join(current_file_path, "res/good_sch.json"), "r"))
 GOOD_SCH = set([re.sub(r"[,. &（）()]+", "", c) for c in GOOD_SCH])
@ -26,14 +29,15 @@ GOOD_SCH = set([re.sub(r"[,. &（）()]+", "", c) for c in GOOD_SCH])
 def loadRank(fnm):
    global TBL
    TBL["rank"] = 1000000
-    with open(fnm, "r", encoding='utf-8') as f:
+    with open(fnm, "r", encoding="utf-8") as f:
        while True:
-            l = f.readline()
-            if not l:break
-            l = l.strip("\n").split(",")
+            line = f.readline()
+            if not line:
+                break
+            line = line.strip("\n").split(",")
            try:
-                nm,rk = l[0].strip(),int(l[1])
-                #assert len(TBL[((TBL.name_cn == nm) | (TBL.name_en == nm))]),f"<{nm}>"
+                nm, rk = line[0].strip(), int(line[1])
+                # assert len(TBL[((TBL.name_cn == nm) | (TBL.name_en == nm))]),f"<{nm}>"
                TBL.loc[((TBL.name_cn == nm) | (TBL.name_en == nm)), "rank"] = rk
            except Exception:
                pass
@ -44,27 +48,35 @@ loadRank(os.path.join(current_file_path, "res/school.rank.csv"))

 def split(txt):
    tks = []
-    for t in re.sub(r"[ \t]+", " ",txt).split():
-        if tks and re.match(r".*[a-zA-Z]$", tks[-1]) and \
-           re.match(r"[a-zA-Z]", t) and tks:
+    for t in re.sub(r"[ \t]+", " ", txt).split():
+        if (
+            tks
+            and re.match(r".*[a-zA-Z]$", tks[-1])
+            and re.match(r"[a-zA-Z]", t)
+            and tks
+        ):
            tks[-1] = tks[-1] + " " + t
-        else:tks.append(t)
+        else:
+            tks.append(t)
    return tks


 def select(nm):
    global TBL
-    if not nm:return 
-    if isinstance(nm, list):nm = str(nm[0])
+    if not nm:
+        return
+    if isinstance(nm, list):
+        nm = str(nm[0])
    nm = split(nm)[0]
    nm = str(nm).lower().strip()
    nm = re.sub(r"[(（][^()（）]+[)）]", "", nm.lower())
    nm = re.sub(r"(^the |[,.&（）();；·]+|^(英国|美国|瑞士))", "", nm)
    nm = re.sub(r"大学.*学院", "大学", nm)
    tbl = copy.deepcopy(TBL)
-    tbl["hit_alias"] = tbl["alias"].map(lambda x:nm in set(x.split("+")))
-    res = tbl[((tbl.name_cn == nm) | (tbl.name_en == nm) | (tbl.hit_alias == True))]
-    if res.empty:return
+    tbl["hit_alias"] = tbl["alias"].map(lambda x: nm in set(x.split("+")))
+    res = tbl[((tbl.name_cn == nm) | (tbl.name_en == nm) | tbl.hit_alias)]
+    if res.empty:
+        return

    return json.loads(res.to_json(orient="records"))[0]

@ -74,4 +86,3 @@ def is_good(nm):
    nm = re.sub(r"[(（][^()（）]+[)）]", "", nm.lower())
    nm = re.sub(r"[''`‘’“”,. &（）();；]+", "", nm)
    return nm in GOOD_SCH
-