fix position extraction bug (#93)

* fix position extraction bug

* remove delimiter for naive parser
This commit is contained in:
KevinHuSh
2024-03-04 17:08:35 +08:00
committed by GitHub
parent fae00827e6
commit 7bfaf0df29
11 changed files with 34 additions and 22 deletions

View File

@ -83,10 +83,10 @@ def dispatch():
pages = PdfParser.total_page_number(r["name"], MINIO.get(r["kb_id"], r["location"]))
for s,e in r["parser_config"].get("pages", [(0,100000)]):
e = min(e, pages)
for p in range(s, e, 10):
for p in range(s, e, 5):
task = new_task()
task["from_page"] = p
task["to_page"] = min(p + 10, e)
task["to_page"] = min(p + 5, e)
tsks.append(task)
else:
tsks.append(new_task())