From 3f96c9c6502243b06efd4bb34a189e72e3a5557c Mon Sep 17 00:00:00 2001
From: yearning <10538594+wangweifeng1999@user.noreply.gitee.com>
Date: 星期一, 28 九月 2026 19:24:55 +0800
Subject: [PATCH] 复制项目
---
backend/app/sql_parser.py | 183 +++++++++++++++++++++++++++++++++++++++++++++
1 files changed, 183 insertions(+), 0 deletions(-)
diff --git a/backend/app/sql_parser.py b/backend/app/sql_parser.py
new file mode 100644
index 0000000..6816f06
--- /dev/null
+++ b/backend/app/sql_parser.py
@@ -0,0 +1,183 @@
+import re
+from typing import Any
+
+
+def split_inserts(sql_text: str) -> list[str]:
+ parts = re.split(r"(?=INSERT INTO)", sql_text, flags=re.IGNORECASE)
+ return [p.strip().rstrip(";") for p in parts if p.strip().upper().startswith("INSERT")]
+
+
+def parse_insert_columns(insert_sql: str) -> list[str]:
+ m = re.search(r"INSERT INTO\s+[`\"]?[^`\"(\s]+[`\"]?\s*\(([^)]+)\)\s*VALUES", insert_sql, re.I)
+ if not m:
+ return []
+ return [c.strip().strip("`\"") for c in m.group(1).split(",")]
+
+
+def parse_values_tuple(insert_sql: str) -> list[Any]:
+ idx = insert_sql.upper().find("VALUES")
+ if idx < 0:
+ return []
+ rest = insert_sql[idx + 6 :].strip()
+ if not rest.startswith("("):
+ return []
+ return _parse_tuple(rest)
+
+
+def _parse_tuple(s: str) -> list[Any]:
+ assert s[0] == "("
+ i = 1
+ values: list[Any] = []
+ while i < len(s):
+ while i < len(s) and s[i] in " \t\n\r":
+ i += 1
+ if i >= len(s):
+ break
+ if s[i] == ")":
+ break
+ if i < len(s) - 3 and s[i : i + 4].upper() == "NULL":
+ values.append(None)
+ i += 4
+ while i < len(s) and s[i] in " \t":
+ i += 1
+ if i < len(s) and s[i] == ",":
+ i += 1
+ continue
+ if s[i] == "'":
+ i += 1
+ buf: list[str] = []
+ while i < len(s):
+ ch = s[i]
+ if ch == "\\" and i + 1 < len(s):
+ n = s[i + 1]
+ if n == "'":
+ buf.append("'")
+ i += 2
+ continue
+ if n == '"':
+ buf.append('"')
+ i += 2
+ continue
+ if n == "\\":
+ buf.append("\\")
+ i += 2
+ continue
+ buf.append(ch)
+ i += 1
+ continue
+ if ch == "'":
+ if i + 1 < len(s) and s[i + 1] == "'":
+ buf.append("'")
+ i += 2
+ continue
+ i += 1
+ break
+ buf.append(ch)
+ i += 1
+ values.append("".join(buf))
+ while i < len(s) and s[i] in " \t":
+ i += 1
+ if i < len(s) and s[i] == ",":
+ i += 1
+ continue
+ if s[i : i + 2] == "b'":
+ i += 2
+ while i < len(s) and s[i] != "'":
+ i += 1
+ i += 1
+ values.append(0)
+ while i < len(s) and s[i] in " \t":
+ i += 1
+ if i < len(s) and s[i] == ",":
+ i += 1
+ continue
+ m = re.match(r"-?\d+(?:\.\d+)?", s[i:])
+ if m:
+ num = m.group(0)
+ values.append(float(num) if "." in num else int(num))
+ i += len(num)
+ while i < len(s) and s[i] in " \t":
+ i += 1
+ if i < len(s) and s[i] == ",":
+ i += 1
+ continue
+ i += 1
+ return values
+
+
+def rows_from_sql(sql_text: str) -> list[dict[str, Any]]:
+ rows = []
+ for ins in split_inserts(sql_text):
+ cols = parse_insert_columns(ins)
+ vals = parse_values_tuple(ins)
+ if not cols or len(vals) != len(cols):
+ continue
+ rows.append(dict(zip(cols, vals)))
+ return rows
+
+
+def _match_examinee(low: str) -> bool:
+ if "examination_examinee" not in low:
+ return False
+ if "exam_answer" in low or "exam_examinee" in low:
+ return False
+ return True
+
+
+def _match_exam_examinee(low: str) -> bool:
+ return "exam_examinee" in low or "鑰冭瘯鑰冪敓" in low
+
+
+def _match_answer(low: str) -> bool:
+ return "exam_answer" in low or "鑰冪敓绛旀" in low
+
+
+def _match_paper_source(low: str) -> bool:
+ return "paper_source" in low or ("璇曞嵎" in low and "item" not in low)
+
+
+def _match_paper_item(low: str) -> bool:
+ return "paper_item" in low or "璇曢" in low
+
+
+def _match_exam(low: str) -> bool:
+ if "examinee" in low or "exam_answer" in low:
+ return False
+ if "exam_examinee" in low:
+ return False
+ return "examination_exam" in low
+
+
+KIND_MATCHERS = {
+ "examinee": _match_examinee,
+ "exam_examinee": _match_exam_examinee,
+ "answer": _match_answer,
+ "paper_source": _match_paper_source,
+ "paper_item": _match_paper_item,
+ "exam": _match_exam,
+}
+
+
+def find_file_content(files: dict[str, bytes], kind: str) -> str | None:
+ matcher = KIND_MATCHERS.get(kind)
+ if not matcher:
+ return None
+ candidates: list[tuple[str, bytes]] = []
+ for name, data in files.items():
+ low = name.lower().replace("锛�", "(").replace("锛�", ")")
+ if matcher(low):
+ candidates.append((name, data))
+ if not candidates:
+ return None
+ candidates.sort(key=lambda x: len(x[0]))
+ return candidates[0][1].decode("utf-8", errors="replace")
+
+
+def find_file_content_legacy(files: dict[str, bytes], patterns: list[str]) -> str | None:
+ """Deprecated: use kind-based find_file_content in import_service."""
+ for name, data in files.items():
+ low = name.lower()
+ for pat in patterns:
+ if pat.lower() in low:
+ return data.decode("utf-8", errors="replace")
+ return None
--
Gitblit v1.8.0