add scripts

This commit is contained in:
2026-07-10 09:49:37 +08:00
parent f323a9026c
commit 8fde7c04f4
3 changed files with 1103 additions and 0 deletions

View File

@@ -0,0 +1,825 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
学科网资料时间分类程序。
按资料文件本身和压缩包内最新子文件的时间关系,把学科网资料分为四类。
"""
import argparse
import hashlib
import json
import os
import re
import shutil
import subprocess
import zipfile
from datetime import datetime
from pathlib import Path
from typing import Dict, List, Optional, Tuple
BASE_DIR = Path(__file__).resolve().parent
ARCHIVE_EXTENSIONS = {".zip", ".rar", ".7z"}
FIVE_MINUTES_SECONDS = 5 * 60
DB_CONFIG = {
"host": os.environ.get("XKW_DB_HOST", "192.168.0.164"),
"port": int(os.environ.get("XKW_DB_PORT", "3307")),
"user": os.environ.get("XKW_DB_USER", "root"),
"password": os.environ.get("XKW_DB_PASSWORD", "myP#ssw0rd"),
"database": os.environ.get("XKW_DB_NAME", "xkw"),
"charset": "utf8mb4",
}
EVIDENCE_TABLE_NAME = os.environ.get("XKW_EVIDENCE_TABLE", "证据10总表")
SINGLE_FILE = "single_file"
ARCHIVE_WITHIN_5_MINUTES = "archive_within_5_minutes"
ARCHIVE_CHILD_LATER = "archive_child_later"
ARCHIVE_CHILD_EARLIER = "archive_child_earlier"
ARCHIVE_REQUIRES_SOURCE_TIME = "archive_requires_source_time"
UNSUPPORTED_ARCHIVE = "unsupported_archive"
EMPTY_ARCHIVE = "empty_archive"
FAILED_ARCHIVE = "failed_archive"
CLASSIFICATION_CATEGORIES = (
SINGLE_FILE,
ARCHIVE_WITHIN_5_MINUTES,
ARCHIVE_CHILD_LATER,
ARCHIVE_CHILD_EARLIER,
)
MATCH_KEYWORDS = [
"语文", "数学", "英语", "物理", "化学", "生物", "历史", "地理", "政治", "科学",
"中考", "高考", "会考", "学考", "一模", "二模", "三模", "模拟", "月考", "期中",
"期末", "联考", "调研", "测试", "试卷", "试题", "课件", "导学案", "教案",
]
def format_datetime(ts: Optional[float]) -> Optional[str]:
if ts is None:
return None
return datetime.fromtimestamp(ts).strftime("%Y-%m-%d %H:%M:%S")
def clear_directory(dir_path: Path) -> None:
if not dir_path.exists():
return
for item in dir_path.iterdir():
if item.is_dir():
shutil.rmtree(item)
else:
item.unlink()
def is_archive(path: Path) -> bool:
return path.suffix.lower() in ARCHIVE_EXTENSIONS
def build_file_record(path: Path, source_root: Optional[Path] = None) -> Dict:
stat_info = path.stat()
record = {
"file_name": path.name,
"path": str(path.resolve()),
"mtime": stat_info.st_mtime,
"datetime": format_datetime(stat_info.st_mtime),
}
if source_root:
try:
record["relative_path"] = str(path.resolve().relative_to(source_root.resolve()))
except ValueError:
record["relative_path"] = path.name
return record
def extract_zip_with_member_times(zip_path: Path, extract_dir: Path) -> List[Path]:
clear_directory(extract_dir)
extract_dir.mkdir(parents=True, exist_ok=True)
extracted_files: List[Path] = []
with zipfile.ZipFile(zip_path, "r") as zip_file:
for member in zip_file.infolist():
extracted_path = zip_file.extract(member, extract_dir)
member_path = Path(extracted_path)
if not member_path.exists():
continue
timestamp = datetime(*member.date_time[:6]).timestamp()
os.utime(member_path, (timestamp, timestamp))
if member_path.is_file():
extracted_files.append(member_path)
return extracted_files
def collect_files(base_dir: Path) -> List[Path]:
return sorted((path for path in base_dir.rglob("*") if path.is_file()), key=lambda path: str(path).lower())
def extract_external_archive(archive_path: Path, extract_dir: Path) -> List[Path]:
clear_directory(extract_dir)
extract_dir.mkdir(parents=True, exist_ok=True)
seven_zip = shutil.which("7z")
if not seven_zip:
raise RuntimeError(f"需要安装 7z 才能解压 {archive_path.suffix} 文件")
result = subprocess.run(
[seven_zip, "x", "-y", f"-o{extract_dir}", str(archive_path)],
capture_output=True,
text=True,
)
if result.returncode != 0:
error_text = (result.stderr or result.stdout or "").strip()
raise RuntimeError(error_text or f"7z 解压失败:{archive_path.name}")
return collect_files(extract_dir)
def extract_archive(archive_path: Path, extract_dir: Path) -> List[Path]:
if archive_path.suffix.lower() == ".zip":
return extract_zip_with_member_times(archive_path, extract_dir)
return extract_external_archive(archive_path, extract_dir)
def latest_child_mtime_from_files(files: List[Path]) -> Optional[float]:
child_times = [file_path.stat().st_mtime for file_path in files if file_path.is_file()]
return max(child_times) if child_times else None
def preliminary_archive_category(archive_mtime: float, child_mtime: float) -> str:
diff_seconds = child_mtime - archive_mtime
if abs(diff_seconds) <= FIVE_MINUTES_SECONDS:
return ARCHIVE_WITHIN_5_MINUTES
return ARCHIVE_REQUIRES_SOURCE_TIME
def parse_database_datetime(value) -> Tuple[Optional[str], Optional[float]]:
if value in (None, ""):
return None, None
if isinstance(value, datetime):
dt = value
return dt.strftime("%Y-%m-%d %H:%M:%S"), dt.timestamp()
text = str(value).strip()
formats = (
"%d/%m/%y %H:%M:%S",
"%d/%m/%y %H:%M",
"%d/%m/%Y %H:%M:%S",
"%d/%m/%Y %H:%M",
"%Y/%m/%d %H:%M:%S",
"%Y/%m/%d %H:%M",
"%Y-%m-%d %H:%M:%S",
"%Y-%m-%d %H:%M",
"%Y/%m/%d",
"%Y-%m-%d",
)
for fmt in formats:
try:
dt = datetime.strptime(text, fmt)
return dt.strftime("%Y-%m-%d %H:%M:%S"), dt.timestamp()
except ValueError:
continue
return text, None
def normalize_for_match(text: str) -> str:
text = str(text).lower()
text = re.sub(r"[\u0020-\u002f\u003a-\u0040\u005b-\u0060\u007b-\u007e]+", " ", text)
text = re.sub(r"\s+", " ", text).strip()
return text
def tokenize_for_match(text: str) -> Dict[str, float]:
normalized = normalize_for_match(text)
compact = re.sub(r"[^0-9a-z\u4e00-\u9fff]+", "", normalized)
tokens: Dict[str, float] = {}
def add_token(token: str, weight: float) -> None:
token = re.sub(r"^[年上下前后新旧本]+", "", token)
if token:
tokens[token] = max(tokens.get(token, 0.0), weight)
for keyword in MATCH_KEYWORDS:
if keyword in compact:
add_token(keyword, 1.8)
for number in re.findall(r"\d{4}|\d+", normalized):
add_token(number, 0.6 if len(number) == 4 else 0.3)
for alpha in re.findall(r"[a-z]+", normalized):
add_token(alpha, 0.2)
for chunk in re.findall(r"[\u4e00-\u9fff]+", compact):
for n, weight in ((2, 0.25), (3, 0.7), (4, 1.1)):
if len(chunk) < n:
continue
for index in range(len(chunk) - n + 1):
add_token(chunk[index:index + n], weight)
return tokens
def token_overlap_score(a: str, b: str) -> float:
tokens_a = tokenize_for_match(a)
tokens_b = tokenize_for_match(b)
if not tokens_a or not tokens_b:
return 0.0
overlap_weight = sum(min(weight, tokens_b.get(token, 0.0)) for token, weight in tokens_a.items() if token in tokens_b)
total_weight = sum(tokens_a.values())
return overlap_weight / total_weight if total_weight else 0.0
def parse_word_content_lines(raw_lines: List[str]) -> List[str]:
def normalize_line(line: str) -> str:
return line.replace("\ufeff", "").strip()
def is_url_line(line: str) -> bool:
return bool(re.match(r"^https?://", normalize_line(line)))
def is_index_line(line: str) -> bool:
return bool(re.match(r"^\d+[\.、]?$", normalize_line(line)))
def is_twole_label(line: str) -> bool:
return bool(re.match(r"^(二一教育|21 世纪教育|21 世纪教育网|二一世纪教育)\s*[:]?$", normalize_line(line)))
def is_zxxk_label(line: str) -> bool:
return bool(re.match(r"^学科网\s*[:]?$", normalize_line(line)))
def is_date_time(line: str) -> bool:
pattern = r"^\d{4}[-/]\d{1,2}[-/]\d{1,2}(\s+\d{1,2}:\d{1,2}(:\d{1,2})?(\s*(AM|PM|上午 | 下午))?)?$"
return bool(re.match(pattern, normalize_line(line)))
def is_skip_line(line: str) -> bool:
return is_index_line(line) or is_twole_label(line) or is_zxxk_label(line) or not normalize_line(line)
def is_url_fragment(line: str) -> bool:
return bool(re.match(r"^[A-Za-z0-9._~:/?#\[\]@!$&'()*+,;=%-]+$", normalize_line(line)))
def consume_url(lines: List[str], start: int) -> Optional[Tuple[str, int]]:
i = start
while i < len(lines) and is_skip_line(lines[i]):
i += 1
if i >= len(lines):
return None
first_part = normalize_line(lines[i])
if not first_part.startswith("http"):
return None
url_parts = [first_part]
i += 1
while i < len(lines):
line = normalize_line(lines[i])
if is_date_time(line):
break
if is_skip_line(line):
i += 1
continue
if not is_url_fragment(line):
break
url_parts.append(line)
i += 1
return "".join(url_parts), i
def parse_content_item(lines: List[str], start: int) -> Optional[Tuple[str, str, str, int]]:
i = start
while i < len(lines) and is_skip_line(lines[i]):
i += 1
title_parts = []
while i < len(lines) and len(title_parts) < 8:
line = normalize_line(lines[i])
if is_url_line(line):
break
if is_date_time(line):
return None
if is_skip_line(line):
i += 1
continue
title_parts.append(line)
i += 1
while i < len(lines) and is_skip_line(lines[i]):
i += 1
url_result = consume_url(lines, i)
if not title_parts or not url_result:
return None
title = "".join(title_parts)
url, i = url_result
date = "1970-1-1"
if i < len(lines) and is_date_time(lines[i]):
date = normalize_line(lines[i])
i += 1
return title, url, date, i
def get_site_from_url(url: str) -> Optional[str]:
normalized_url = normalize_line(url).lower()
if "21cnjy.com" in normalized_url:
return "twole"
if "zxxk.com" in normalized_url:
return "zxxk"
return None
def extract_content_items(lines: List[str]) -> List[Tuple[str, str, str, str]]:
items = []
i = 0
while i < len(lines):
item = parse_content_item(lines, i)
if not item:
i += 1
continue
title, url, date, next_i = item
site = get_site_from_url(url)
if site in ("twole", "zxxk"):
items.append((title, url, date, site))
i = next_i
return items
def build_grouped_rows(items: List[Tuple[str, str, str, str]]) -> List[str]:
twole_items = [(title, url, date) for title, url, date, site in items if site == "twole"]
zxxk_items = [(title, url, date) for title, url, date, site in items if site == "zxxk"]
count = min(len(twole_items), len(zxxk_items))
return ["\t".join([twole_items[index][0], twole_items[index][1], twole_items[index][2],
zxxk_items[index][0], zxxk_items[index][1], zxxk_items[index][2]]) for index in range(count)]
def parse_site_items(lines: List[str]) -> List[str]:
items = extract_content_items(lines)
sites = [site for _, _, _, site in items]
if len(sites) < 2 or len(set(sites)) < 2:
return []
return build_grouped_rows(items)
def parse_grouped_site_rows(lines: List[str]) -> List[str]:
twole_items = []
zxxk_items = []
current_site = None
i = 0
while i < len(lines):
line = normalize_line(lines[i])
if is_twole_label(line):
current_site = "twole"
i += 1
continue
if is_zxxk_label(line):
current_site = "zxxk"
i += 1
continue
if not line or line in ("......", "……") or line.startswith("其它内容") or line.startswith("其他内容"):
i += 1
continue
if current_site not in ("twole", "zxxk"):
i += 1
continue
title = line
url = ""
date = "1970-1-1"
url_result = consume_url(lines, i + 1)
if url_result:
url, i = url_result
if i < len(lines) and is_date_time(lines[i]):
date = normalize_line(lines[i])
i += 1
else:
i += 1
if current_site == "twole":
twole_items.append((title, url, date))
else:
zxxk_items.append((title, url, date))
count = min(len(twole_items), len(zxxk_items))
return ["\t".join([twole_items[index][0], twole_items[index][1], twole_items[index][2],
zxxk_items[index][0], zxxk_items[index][1], zxxk_items[index][2]]) for index in range(count)]
lines = [normalize_line(line) for line in raw_lines if normalize_line(line)]
has_tab_in_first_rows = any("\t" in line for line in lines[:6])
split_lines = []
for line in lines:
split_lines.extend(normalize_line(part) for part in line.split("\t") if normalize_line(part))
site_result = parse_site_items(split_lines)
if site_result:
return site_result
has_site_labels = any(is_twole_label(line) or is_zxxk_label(line) for line in split_lines)
if has_site_labels:
grouped_result = parse_grouped_site_rows(split_lines)
if grouped_result:
return grouped_result
if has_tab_in_first_rows or len(split_lines) < 6:
return lines
return lines
def read_docx_content(doc_path: Path) -> List[str]:
def clean_hyperlink(line: str) -> str:
patterns = [r"HYPERLINK\s+&quot;([^&]+)&quot;\s+(\S+)", r"HYPERLINK\s+\"([^\"]+)\"\s+(\S+)"]
for pattern in patterns:
match = re.match(pattern, line)
if match:
return match.group(2)
return line
try:
with zipfile.ZipFile(doc_path, "r") as z:
content = z.read("word/document.xml").decode("utf-8")
texts = re.findall(r"<w:t[^>]*>(.*?)</w:t>", content, re.DOTALL)
processed = []
for text in texts:
text = re.sub(r"<[^>]+>", "", text).strip()
if text:
processed.append(clean_hyperlink(text))
return parse_word_content_lines(processed)
except Exception:
return []
def read_doc_content_old(doc_path: Path) -> List[str]:
for soffice_cmd in ["/Applications/LibreOffice.app/Contents/MacOS/soffice",
"/Applications/LibreOffice.app/Contents/MacOS/libreoffice", "soffice"]:
if os.path.isfile(soffice_cmd) and subprocess.run([soffice_cmd, "--version"], capture_output=True).returncode == 0:
break
else:
return []
import tempfile
try:
with tempfile.TemporaryDirectory() as tmpdir:
result = subprocess.run(
[soffice_cmd, "--headless", "--convert-to", "txt", "--outdir", tmpdir, str(doc_path)],
capture_output=True,
timeout=60,
)
if result.returncode != 0:
return []
txt_file = Path(tmpdir) / (doc_path.stem + ".txt")
if txt_file.exists():
lines = [line.strip() for line in txt_file.read_text(encoding="utf-8", errors="ignore").splitlines() if line.strip()]
return parse_word_content_lines(lines)
except Exception:
pass
return []
def read_word_content(doc_path: Path) -> List[str]:
if doc_path.suffix.lower() == ".docx":
return read_docx_content(doc_path)
if doc_path.suffix.lower() == ".doc":
return read_doc_content_old(doc_path)
return []
def find_word_doc_for_zxxk_dir(zxxk_dir: Path) -> Optional[Path]:
parent = zxxk_dir.parent
word_files = sorted(
(item for item in parent.iterdir() if item.is_file() and item.suffix.lower() in (".docx", ".doc")),
key=lambda item: item.name.lower(),
)
return word_files[0] if word_files else None
def extract_zxxk_id(url: str) -> Optional[str]:
match = re.search(r"/(?:soft/)?(\d+)\.html(?:[?#].*)?$", url or "")
if match:
return match.group(1)
numbers = re.findall(r"\d+", url or "")
return numbers[-1] if numbers else None
def extract_zxxk_metadata_rows(word_doc: Optional[Path]) -> List[Dict]:
if not word_doc:
return []
rows = []
for row_index, content in enumerate(read_word_content(word_doc)):
parts = [part.strip() for part in content.split("\t")]
if len(parts) < 6:
continue
candidates = [(parts[0], parts[1], parts[2]), (parts[3], parts[4], parts[5])]
for title, url, source_time in candidates:
if "zxxk.com" not in url.lower():
continue
rows.append({
"title": title,
"url": url,
"source_time": source_time,
"word_source_time": source_time,
"zxxk_id": extract_zxxk_id(url),
"word_doc": word_doc.name,
"word_doc_path": str(word_doc.resolve()),
"word_row_index": row_index,
})
return rows
def match_metadata_for_file(material_path: Path, metadata_rows: List[Dict]) -> Optional[Dict]:
if not metadata_rows:
return None
file_name = material_path.name
file_stem = material_path.stem
file_norm = normalize_for_match(file_stem)
exact_matches = []
for row in metadata_rows:
title_norm = normalize_for_match(row["title"])
if title_norm and (title_norm == file_norm or title_norm in file_norm or file_norm in title_norm):
exact_matches.append(row)
if exact_matches:
return max(exact_matches, key=lambda row: len(normalize_for_match(row["title"])))
scored_rows = [
(token_overlap_score(row["title"], file_name), row)
for row in metadata_rows
]
best_score, best_row = max(scored_rows, key=lambda item: item[0])
return best_row if best_score > 0 else None
def enrich_record_with_metadata(record: Dict, material_path: Path, metadata_rows: List[Dict]) -> None:
metadata = match_metadata_for_file(material_path, metadata_rows)
record["metadata_found"] = metadata is not None
if not metadata:
record.update({
"title": None,
"url": None,
"source_time": None,
"source_time_raw": None,
"source_time_mtime": None,
"word_source_time": None,
"source_time_source": None,
"classification_time_basis": None,
"child_source_time_diff_seconds": None,
"zxxk_id": None,
"word_doc": None,
"word_doc_path": None,
"word_row_index": None,
})
return
record.update(metadata)
record["source_time_source"] = "word"
class MysqlUploadTimeLookup:
def __init__(self, config: Optional[Dict] = None, table_name: str = EVIDENCE_TABLE_NAME):
self.config = dict(config or DB_CONFIG)
self.table_name = table_name
self.conn = None
self.cache: Dict[Tuple[str, str], Optional[str]] = {}
def close(self) -> None:
if self.conn:
self.conn.close()
self.conn = None
def connect(self):
if self.conn:
return self.conn
import mysql.connector
self.conn = mysql.connector.connect(**self.config)
return self.conn
def __call__(self, metadata: Dict) -> Optional[str]:
zxxk_id = metadata.get("zxxk_id") or ""
url = metadata.get("url") or ""
cache_key = (zxxk_id, url)
if cache_key not in self.cache:
self.cache[cache_key] = self.fetch_upload_time(zxxk_id, url)
return self.cache[cache_key]
def fetch_upload_time(self, zxxk_id: str, url: str) -> Optional[str]:
if not zxxk_id and not url:
return None
conn = self.connect()
cursor = conn.cursor()
try:
if zxxk_id:
cursor.execute(
f"SELECT `学科网上传日期` FROM `{self.table_name}` "
"WHERE `xkwID` = %s AND `学科网上传日期` IS NOT NULL AND `学科网上传日期` <> '' LIMIT 1",
(zxxk_id,),
)
row = cursor.fetchone()
if row and row[0]:
return str(row[0])
if url:
cursor.execute(
f"SELECT `学科网上传日期` FROM `{self.table_name}` "
"WHERE `学科网资料链接` = %s AND `学科网上传日期` IS NOT NULL AND `学科网上传日期` <> '' LIMIT 1",
(url,),
)
row = cursor.fetchone()
if row and row[0]:
return str(row[0])
finally:
cursor.close()
return None
def apply_upload_time_lookup(record: Dict, upload_time_lookup) -> None:
if not record.get("metadata_found") or not upload_time_lookup:
return
upload_time = upload_time_lookup(record)
if not upload_time:
return
normalized_time, timestamp = parse_database_datetime(upload_time)
record["source_time_raw"] = str(upload_time)
record["source_time"] = normalized_time
record["source_time_mtime"] = timestamp
record["source_time_source"] = "mysql"
def reclassify_archive_with_source_time(record: Dict) -> None:
if record.get("file_kind") != "archive" or record.get("latest_child_mtime") is None:
return
source_time_mtime = record.get("source_time_mtime")
if source_time_mtime is not None:
record["child_source_time_diff_seconds"] = record["latest_child_mtime"] - source_time_mtime
archive_diff_seconds = record.get("child_archive_time_diff_seconds")
if archive_diff_seconds is not None and abs(archive_diff_seconds) <= FIVE_MINUTES_SECONDS:
record["category"] = ARCHIVE_WITHIN_5_MINUTES
record["classification_time_basis"] = "archive_mtime_within_5_minutes"
return
if source_time_mtime is None:
record["category"] = ARCHIVE_REQUIRES_SOURCE_TIME
record["classification_time_basis"] = "requires_mysql_upload_time"
return
diff_seconds = record["child_source_time_diff_seconds"]
record["classification_time_basis"] = "mysql_upload_time"
if diff_seconds > 6000:
record["category"] = ARCHIVE_CHILD_LATER
else:
record["category"] = ARCHIVE_CHILD_EARLIER
def classify_material(material_path: Path, work_dir: Path, source_root: Optional[Path] = None) -> Dict:
material_path = Path(material_path)
file_record = build_file_record(material_path, source_root=source_root)
result = {
**file_record,
"category": SINGLE_FILE,
"file_kind": "single_file",
"archive_mtime": None,
"archive_datetime": None,
"latest_child_mtime": None,
"latest_child_datetime": None,
"child_archive_time_diff_seconds": None,
"child_source_time_diff_seconds": None,
"classification_time_basis": None,
"child_count": 0,
}
if not is_archive(material_path):
return result
result.update({
"file_kind": "archive",
"archive_mtime": file_record["mtime"],
"archive_datetime": file_record["datetime"],
})
extract_dir = Path(work_dir) / material_path.stem
child_mtime = None
child_count = 0
try:
extracted_files = extract_archive(material_path, extract_dir)
child_count = len(extracted_files)
child_mtime = latest_child_mtime_from_files(extracted_files)
except Exception as exc:
result["category"] = FAILED_ARCHIVE
result["error"] = str(exc)
return result
finally:
if extract_dir.exists():
shutil.rmtree(extract_dir)
result["child_count"] = child_count
if child_mtime is None:
result["category"] = EMPTY_ARCHIVE
return result
diff_seconds = child_mtime - file_record["mtime"]
result.update({
"category": preliminary_archive_category(file_record["mtime"], child_mtime),
"latest_child_mtime": child_mtime,
"latest_child_datetime": format_datetime(child_mtime),
"child_archive_time_diff_seconds": diff_seconds,
"classification_time_basis": "archive_mtime_initial",
})
return result
def find_zxxk_dirs(source_dir: Path) -> List[Path]:
source_dir = Path(source_dir)
if source_dir.name == "学科网" and source_dir.is_dir():
return [source_dir]
return sorted(
(path for path in source_dir.rglob("学科网") if path.is_dir()),
key=lambda path: str(path).lower(),
)
def iter_materials(zxxk_dir: Path) -> List[Path]:
return sorted(
(item for item in zxxk_dir.iterdir() if item.is_file()),
key=lambda item: item.name.lower(),
)
def empty_result(source_dir: Path) -> Dict:
return {
"source_dir": str(Path(source_dir).resolve()),
"generated_at": datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
"total_zxxk_dirs": 0,
"total_materials": 0,
"summary": {category: 0 for category in CLASSIFICATION_CATEGORIES},
"categories": {category: [] for category in CLASSIFICATION_CATEGORIES},
"other_categories": {},
"warnings": [],
}
def add_classification(result: Dict, record: Dict) -> None:
category = record["category"]
if category in result["categories"]:
result["categories"][category].append(record)
result["summary"][category] += 1
else:
result["other_categories"].setdefault(category, []).append(record)
result["total_materials"] += 1
def classify_zxxk_materials(source_dir: Path, work_dir: Path, upload_time_lookup=None) -> Dict:
source_dir = Path(source_dir)
work_dir = Path(work_dir)
result = empty_result(source_dir)
zxxk_dirs = find_zxxk_dirs(source_dir)
result["total_zxxk_dirs"] = len(zxxk_dirs)
for zxxk_dir in zxxk_dirs:
word_doc = find_word_doc_for_zxxk_dir(zxxk_dir)
metadata_rows = extract_zxxk_metadata_rows(word_doc)
for material_path in iter_materials(zxxk_dir):
archive_key = hashlib.sha1(str(material_path.resolve()).encode("utf-8")).hexdigest()[:16]
relative_work_dir = work_dir / archive_key
record = classify_material(material_path, relative_work_dir, source_root=source_dir)
enrich_record_with_metadata(record, material_path, metadata_rows)
try:
apply_upload_time_lookup(record, upload_time_lookup)
except Exception as exc:
result["warnings"].append(f"查询学科网上传日期失败:{exc}")
upload_time_lookup = None
reclassify_archive_with_source_time(record)
if relative_work_dir.exists() and not any(relative_work_dir.iterdir()):
relative_work_dir.rmdir()
try:
record["zxxk_dir"] = str(zxxk_dir.resolve().relative_to(source_dir.resolve()))
except ValueError:
record["zxxk_dir"] = str(zxxk_dir.resolve())
add_classification(result, record)
return result
def write_json(data: Dict, output_path: Path) -> None:
output_path.parent.mkdir(parents=True, exist_ok=True)
with output_path.open("w", encoding="utf-8") as f:
json.dump(data, f, ensure_ascii=False, indent=2)
def print_summary(result: Dict, output_path: Path) -> None:
print("===== 学科网资料分类完成 =====")
print(f"学科网目录数:{result['total_zxxk_dirs']}")
print(f"资料总数:{result['total_materials']}")
print(f"1、单文件{result['summary'][SINGLE_FILE]}")
print(f"2、压缩文件子文件最后时间与压缩包时间在5分钟之内{result['summary'][ARCHIVE_WITHIN_5_MINUTES]}")
print(f"3、剩余压缩文件子文件最后时间比数据库发布时间更晚{result['summary'][ARCHIVE_CHILD_LATER]}")
print(f"4、剩余压缩文件子文件最后时间比数据库发布时间更早{result['summary'][ARCHIVE_CHILD_EARLIER]}")
if result["other_categories"]:
other_total = sum(len(items) for items in result["other_categories"].values())
print(f"其他/异常:{other_total}")
print(f"分类结果:{output_path}")
def main() -> None:
parser = argparse.ArgumentParser(description="学科网资料时间分类程序")
parser.add_argument("--source-dir", type=str, default=str(BASE_DIR / "ccold"), help="待扫描目录,默认 server/cc")
parser.add_argument("--work-dir", type=str, default=str(BASE_DIR / "tmp" / "zxxk_material_classify"), help="解压临时目录")
parser.add_argument("--output", type=str, default=str(BASE_DIR / "jsons" / "zxxk_material_time_categories.json"), help="分类 JSON 输出路径")
parser.add_argument("--skip-db-time", action="store_true", help="不从 MySQL 查询学科网上传日期")
args = parser.parse_args()
source_dir = Path(args.source_dir).expanduser().resolve()
work_dir = Path(args.work_dir).expanduser().resolve()
output_path = Path(args.output).expanduser().resolve()
upload_time_lookup = None if args.skip_db_time else MysqlUploadTimeLookup()
try:
result = classify_zxxk_materials(source_dir, work_dir, upload_time_lookup=upload_time_lookup)
finally:
if upload_time_lookup:
upload_time_lookup.close()
write_json(result, output_path)
print_summary(result, output_path)
if __name__ == "__main__":
main()

Binary file not shown.

View File

@@ -0,0 +1,278 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
将 jsons/zxxk_material_time_categories.json 导入 MySQL xkw_categories 表。
"""
import argparse
import hashlib
import json
import os
import re
from pathlib import Path
from typing import Dict, Iterable, List, Optional
import mysql.connector
DB_CONFIG = {
"host": os.environ.get("XKW_DB_HOST", "192.168.0.164"),
"port": int(os.environ.get("XKW_DB_PORT", "3307")),
"user": os.environ.get("XKW_DB_USER", "root"),
"password": os.environ.get("XKW_DB_PASSWORD", "myP#ssw0rd"),
"database": os.environ.get("XKW_DB_NAME", "xkw"),
"charset": "utf8mb4",
}
BASE_DIR = Path(__file__).resolve().parent
DEFAULT_JSON_FILE_PATH = BASE_DIR / "jsons" / "zxxk_material_time_categories.json"
DEFAULT_TABLE_NAME = "xkw_categories"
DEFAULT_BATCH_SIZE = 500
SELECTED_CATEGORIES = (
"single_file",
"archive_within_5_minutes",
"archive_child_later",
"archive_child_earlier",
)
INSERT_COLUMNS = (
"relative_path_hash",
"file_name",
"mtime",
"datetime",
"relative_path",
"category",
"file_kind",
"archive_mtime",
"archive_datetime",
"latest_child_mtime",
"latest_child_datetime",
"child_archive_time_diff_seconds",
"child_source_time_diff_seconds",
"classification_time_basis",
"child_count",
"metadata_found",
"title",
"url",
"source_time",
"word_source_time",
"zxxk_id",
"word_doc",
"word_row_index",
"source_time_raw",
"source_time_mtime",
"raw_json",
)
CREATE_TABLE_TEMPLATE = """
CREATE TABLE IF NOT EXISTS `{table_name}` (
id BIGINT AUTO_INCREMENT PRIMARY KEY COMMENT '主键',
relative_path_hash CHAR(64) NOT NULL COMMENT 'relative_path/file_name 哈希,用于幂等导入',
file_name VARCHAR(1024) COMMENT '文件名',
mtime DOUBLE COMMENT '文件 mtime',
`datetime` DATETIME COMMENT '文件时间',
relative_path TEXT COMMENT '相对路径',
category VARCHAR(64) NOT NULL COMMENT '分类',
file_kind VARCHAR(32) COMMENT '文件类型: single_file/archive',
archive_mtime DOUBLE COMMENT '压缩包 mtime',
archive_datetime DATETIME COMMENT '压缩包时间',
latest_child_mtime DOUBLE COMMENT '压缩包内最新子文件 mtime',
latest_child_datetime DATETIME COMMENT '压缩包内最新子文件时间',
child_archive_time_diff_seconds DOUBLE COMMENT '子文件时间 - 压缩包时间',
child_source_time_diff_seconds DOUBLE COMMENT '子文件时间 - 数据库发布时间',
classification_time_basis VARCHAR(64) COMMENT '分类时间依据',
child_count INT COMMENT '压缩包内子文件数量',
metadata_found TINYINT(1) COMMENT '是否匹配到 Word/资料元数据',
title TEXT COMMENT '学科网资料标题',
url TEXT COMMENT '学科网资料链接',
source_time DATETIME COMMENT '数据库中的学科网上传日期,已规范化',
word_source_time VARCHAR(100) COMMENT 'Word 中解析到的原始时间',
zxxk_id VARCHAR(100) COMMENT '学科网资料 ID',
word_doc VARCHAR(1024) COMMENT '来源 Word 文件',
word_row_index INT COMMENT 'Word 解析行号',
source_time_raw VARCHAR(100) COMMENT '数据库原始上传日期字符串',
source_time_mtime DOUBLE COMMENT '数据库发布时间 timestamp',
raw_json JSON NOT NULL COMMENT '原始分类 JSON 记录',
imported_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP COMMENT '导入时间',
updated_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP ON UPDATE CURRENT_TIMESTAMP COMMENT '更新时间',
UNIQUE KEY uniq_relative_path_hash (relative_path_hash),
KEY idx_category (category),
KEY idx_file_kind (file_kind),
KEY idx_zxxk_id (zxxk_id),
KEY idx_source_time (source_time),
KEY idx_classification_time_basis (classification_time_basis)
) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COMMENT='学科网资料分类结果';
"""
def mysql_identifier(name: str) -> str:
if not re.match(r"^[A-Za-z0-9_]+$", name):
raise ValueError(f"非法表名:{name}")
return name
def normalize_datetime(value) -> Optional[str]:
if not value:
return None
if isinstance(value, str) and re.match(r"^\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2}$", value):
return value
return None
def optional_float(value) -> Optional[float]:
if value in (None, ""):
return None
return float(value)
def optional_int(value) -> Optional[int]:
if value in (None, ""):
return None
return int(value)
def bool_to_int(value) -> Optional[int]:
if value is None:
return None
return 1 if bool(value) else 0
def json_dumps(value) -> str:
return json.dumps(value or {}, ensure_ascii=False)
def build_relative_path_hash(record: Dict) -> str:
key = record.get("relative_path") or record.get("path") or record.get("file_name") or ""
return hashlib.sha256(str(key).encode("utf-8")).hexdigest()
def normalize_record(category: str, source_record: Dict) -> Dict:
return {
"relative_path_hash": build_relative_path_hash(source_record),
"file_name": source_record.get("file_name"),
"mtime": optional_float(source_record.get("mtime")),
"datetime": normalize_datetime(source_record.get("datetime")),
"relative_path": source_record.get("relative_path"),
"category": category,
"file_kind": source_record.get("file_kind"),
"archive_mtime": optional_float(source_record.get("archive_mtime")),
"archive_datetime": normalize_datetime(source_record.get("archive_datetime")),
"latest_child_mtime": optional_float(source_record.get("latest_child_mtime")),
"latest_child_datetime": normalize_datetime(source_record.get("latest_child_datetime")),
"child_archive_time_diff_seconds": optional_float(source_record.get("child_archive_time_diff_seconds")),
"child_source_time_diff_seconds": optional_float(source_record.get("child_source_time_diff_seconds")),
"classification_time_basis": source_record.get("classification_time_basis"),
"child_count": optional_int(source_record.get("child_count")),
"metadata_found": bool_to_int(source_record.get("metadata_found")),
"title": source_record.get("title"),
"url": source_record.get("url"),
"source_time": normalize_datetime(source_record.get("source_time")),
"word_source_time": source_record.get("word_source_time"),
"zxxk_id": source_record.get("zxxk_id"),
"word_doc": source_record.get("word_doc"),
"word_row_index": optional_int(source_record.get("word_row_index")),
"source_time_raw": source_record.get("source_time_raw"),
"source_time_mtime": optional_float(source_record.get("source_time_mtime")),
"raw_json": json_dumps(source_record),
}
def normalize_records_for_import(data: Dict) -> List[Dict]:
categories = data.get("categories") or {}
records: List[Dict] = []
for category in SELECTED_CATEGORIES:
for source_record in categories.get(category, []):
records.append(normalize_record(category, source_record))
return records
def load_json_data(file_path: Path) -> Dict:
with file_path.open("r", encoding="utf-8") as f:
data = json.load(f)
if not isinstance(data, dict):
raise ValueError("分类 JSON 顶层必须是对象")
return data
def ensure_table(conn, table_name: str) -> None:
safe_table = mysql_identifier(table_name)
cursor = conn.cursor()
try:
cursor.execute(CREATE_TABLE_TEMPLATE.format(table_name=safe_table))
finally:
cursor.close()
def build_insert_sql(table_name: str) -> str:
safe_table = mysql_identifier(table_name)
columns = ", ".join(f"`{column}`" for column in INSERT_COLUMNS)
placeholders = ", ".join(["%s"] * len(INSERT_COLUMNS))
updates = ", ".join(
f"`{column}` = VALUES(`{column}`)"
for column in INSERT_COLUMNS
if column != "relative_path_hash"
)
return f"INSERT INTO `{safe_table}` ({columns}) VALUES ({placeholders}) ON DUPLICATE KEY UPDATE {updates}"
def iter_batches(items: List, batch_size: int):
if batch_size <= 0:
raise ValueError("batch_size 必须大于 0")
for start in range(0, len(items), batch_size):
yield items[start : start + batch_size]
def import_records(
records: Iterable[Dict],
table_name: str = DEFAULT_TABLE_NAME,
db_config: Optional[Dict] = None,
batch_size: int = DEFAULT_BATCH_SIZE,
) -> int:
rows = list(records)
conn = mysql.connector.connect(**(db_config or DB_CONFIG))
try:
ensure_table(conn, table_name)
if not rows:
conn.commit()
return 0
insert_sql = build_insert_sql(table_name)
values = [tuple(record.get(column) for column in INSERT_COLUMNS) for record in rows]
cursor = conn.cursor()
try:
for batch in iter_batches(values, batch_size):
cursor.executemany(insert_sql, batch)
finally:
cursor.close()
conn.commit()
return len(rows)
finally:
conn.close()
def main() -> None:
parser = argparse.ArgumentParser(description="导入学科网资料分类 JSON 到 MySQL")
parser.add_argument("--json-file", type=str, default=str(DEFAULT_JSON_FILE_PATH), help="分类 JSON 文件路径")
parser.add_argument("--table", type=str, default=DEFAULT_TABLE_NAME, help="目标表名,默认 xkw_categories")
parser.add_argument("--batch-size", type=int, default=DEFAULT_BATCH_SIZE, help=f"每批写入条数,默认 {DEFAULT_BATCH_SIZE}")
parser.add_argument("--dry-run", action="store_true", help="只解析并打印记录数,不写入数据库")
args = parser.parse_args()
json_file = Path(args.json_file).expanduser().resolve()
data = load_json_data(json_file)
records = normalize_records_for_import(data)
if args.dry_run:
print(f"解析记录数:{len(records)}")
return
imported_count = import_records(records, table_name=args.table, batch_size=args.batch_size)
print(f"已导入/更新 {imported_count} 条记录到 {args.table}")
if __name__ == "__main__":
main()