Files
hiddencode_project/server/classify_zxxk_materials.py
2026-07-10 09:49:37 +08:00

826 lines
30 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
学科网资料时间分类程序。
按资料文件本身和压缩包内最新子文件的时间关系,把学科网资料分为四类。
"""
import argparse
import hashlib
import json
import os
import re
import shutil
import subprocess
import zipfile
from datetime import datetime
from pathlib import Path
from typing import Dict, List, Optional, Tuple
BASE_DIR = Path(__file__).resolve().parent
ARCHIVE_EXTENSIONS = {".zip", ".rar", ".7z"}
FIVE_MINUTES_SECONDS = 5 * 60
DB_CONFIG = {
"host": os.environ.get("XKW_DB_HOST", "192.168.0.164"),
"port": int(os.environ.get("XKW_DB_PORT", "3307")),
"user": os.environ.get("XKW_DB_USER", "root"),
"password": os.environ.get("XKW_DB_PASSWORD", "myP#ssw0rd"),
"database": os.environ.get("XKW_DB_NAME", "xkw"),
"charset": "utf8mb4",
}
EVIDENCE_TABLE_NAME = os.environ.get("XKW_EVIDENCE_TABLE", "证据10总表")
SINGLE_FILE = "single_file"
ARCHIVE_WITHIN_5_MINUTES = "archive_within_5_minutes"
ARCHIVE_CHILD_LATER = "archive_child_later"
ARCHIVE_CHILD_EARLIER = "archive_child_earlier"
ARCHIVE_REQUIRES_SOURCE_TIME = "archive_requires_source_time"
UNSUPPORTED_ARCHIVE = "unsupported_archive"
EMPTY_ARCHIVE = "empty_archive"
FAILED_ARCHIVE = "failed_archive"
CLASSIFICATION_CATEGORIES = (
SINGLE_FILE,
ARCHIVE_WITHIN_5_MINUTES,
ARCHIVE_CHILD_LATER,
ARCHIVE_CHILD_EARLIER,
)
MATCH_KEYWORDS = [
"语文", "数学", "英语", "物理", "化学", "生物", "历史", "地理", "政治", "科学",
"中考", "高考", "会考", "学考", "一模", "二模", "三模", "模拟", "月考", "期中",
"期末", "联考", "调研", "测试", "试卷", "试题", "课件", "导学案", "教案",
]
def format_datetime(ts: Optional[float]) -> Optional[str]:
if ts is None:
return None
return datetime.fromtimestamp(ts).strftime("%Y-%m-%d %H:%M:%S")
def clear_directory(dir_path: Path) -> None:
if not dir_path.exists():
return
for item in dir_path.iterdir():
if item.is_dir():
shutil.rmtree(item)
else:
item.unlink()
def is_archive(path: Path) -> bool:
return path.suffix.lower() in ARCHIVE_EXTENSIONS
def build_file_record(path: Path, source_root: Optional[Path] = None) -> Dict:
stat_info = path.stat()
record = {
"file_name": path.name,
"path": str(path.resolve()),
"mtime": stat_info.st_mtime,
"datetime": format_datetime(stat_info.st_mtime),
}
if source_root:
try:
record["relative_path"] = str(path.resolve().relative_to(source_root.resolve()))
except ValueError:
record["relative_path"] = path.name
return record
def extract_zip_with_member_times(zip_path: Path, extract_dir: Path) -> List[Path]:
clear_directory(extract_dir)
extract_dir.mkdir(parents=True, exist_ok=True)
extracted_files: List[Path] = []
with zipfile.ZipFile(zip_path, "r") as zip_file:
for member in zip_file.infolist():
extracted_path = zip_file.extract(member, extract_dir)
member_path = Path(extracted_path)
if not member_path.exists():
continue
timestamp = datetime(*member.date_time[:6]).timestamp()
os.utime(member_path, (timestamp, timestamp))
if member_path.is_file():
extracted_files.append(member_path)
return extracted_files
def collect_files(base_dir: Path) -> List[Path]:
return sorted((path for path in base_dir.rglob("*") if path.is_file()), key=lambda path: str(path).lower())
def extract_external_archive(archive_path: Path, extract_dir: Path) -> List[Path]:
clear_directory(extract_dir)
extract_dir.mkdir(parents=True, exist_ok=True)
seven_zip = shutil.which("7z")
if not seven_zip:
raise RuntimeError(f"需要安装 7z 才能解压 {archive_path.suffix} 文件")
result = subprocess.run(
[seven_zip, "x", "-y", f"-o{extract_dir}", str(archive_path)],
capture_output=True,
text=True,
)
if result.returncode != 0:
error_text = (result.stderr or result.stdout or "").strip()
raise RuntimeError(error_text or f"7z 解压失败:{archive_path.name}")
return collect_files(extract_dir)
def extract_archive(archive_path: Path, extract_dir: Path) -> List[Path]:
if archive_path.suffix.lower() == ".zip":
return extract_zip_with_member_times(archive_path, extract_dir)
return extract_external_archive(archive_path, extract_dir)
def latest_child_mtime_from_files(files: List[Path]) -> Optional[float]:
child_times = [file_path.stat().st_mtime for file_path in files if file_path.is_file()]
return max(child_times) if child_times else None
def preliminary_archive_category(archive_mtime: float, child_mtime: float) -> str:
diff_seconds = child_mtime - archive_mtime
if abs(diff_seconds) <= FIVE_MINUTES_SECONDS:
return ARCHIVE_WITHIN_5_MINUTES
return ARCHIVE_REQUIRES_SOURCE_TIME
def parse_database_datetime(value) -> Tuple[Optional[str], Optional[float]]:
if value in (None, ""):
return None, None
if isinstance(value, datetime):
dt = value
return dt.strftime("%Y-%m-%d %H:%M:%S"), dt.timestamp()
text = str(value).strip()
formats = (
"%d/%m/%y %H:%M:%S",
"%d/%m/%y %H:%M",
"%d/%m/%Y %H:%M:%S",
"%d/%m/%Y %H:%M",
"%Y/%m/%d %H:%M:%S",
"%Y/%m/%d %H:%M",
"%Y-%m-%d %H:%M:%S",
"%Y-%m-%d %H:%M",
"%Y/%m/%d",
"%Y-%m-%d",
)
for fmt in formats:
try:
dt = datetime.strptime(text, fmt)
return dt.strftime("%Y-%m-%d %H:%M:%S"), dt.timestamp()
except ValueError:
continue
return text, None
def normalize_for_match(text: str) -> str:
text = str(text).lower()
text = re.sub(r"[\u0020-\u002f\u003a-\u0040\u005b-\u0060\u007b-\u007e]+", " ", text)
text = re.sub(r"\s+", " ", text).strip()
return text
def tokenize_for_match(text: str) -> Dict[str, float]:
normalized = normalize_for_match(text)
compact = re.sub(r"[^0-9a-z\u4e00-\u9fff]+", "", normalized)
tokens: Dict[str, float] = {}
def add_token(token: str, weight: float) -> None:
token = re.sub(r"^[年上下前后新旧本]+", "", token)
if token:
tokens[token] = max(tokens.get(token, 0.0), weight)
for keyword in MATCH_KEYWORDS:
if keyword in compact:
add_token(keyword, 1.8)
for number in re.findall(r"\d{4}|\d+", normalized):
add_token(number, 0.6 if len(number) == 4 else 0.3)
for alpha in re.findall(r"[a-z]+", normalized):
add_token(alpha, 0.2)
for chunk in re.findall(r"[\u4e00-\u9fff]+", compact):
for n, weight in ((2, 0.25), (3, 0.7), (4, 1.1)):
if len(chunk) < n:
continue
for index in range(len(chunk) - n + 1):
add_token(chunk[index:index + n], weight)
return tokens
def token_overlap_score(a: str, b: str) -> float:
tokens_a = tokenize_for_match(a)
tokens_b = tokenize_for_match(b)
if not tokens_a or not tokens_b:
return 0.0
overlap_weight = sum(min(weight, tokens_b.get(token, 0.0)) for token, weight in tokens_a.items() if token in tokens_b)
total_weight = sum(tokens_a.values())
return overlap_weight / total_weight if total_weight else 0.0
def parse_word_content_lines(raw_lines: List[str]) -> List[str]:
def normalize_line(line: str) -> str:
return line.replace("\ufeff", "").strip()
def is_url_line(line: str) -> bool:
return bool(re.match(r"^https?://", normalize_line(line)))
def is_index_line(line: str) -> bool:
return bool(re.match(r"^\d+[\.、]?$", normalize_line(line)))
def is_twole_label(line: str) -> bool:
return bool(re.match(r"^(二一教育|21 世纪教育|21 世纪教育网|二一世纪教育)\s*[:]?$", normalize_line(line)))
def is_zxxk_label(line: str) -> bool:
return bool(re.match(r"^学科网\s*[:]?$", normalize_line(line)))
def is_date_time(line: str) -> bool:
pattern = r"^\d{4}[-/]\d{1,2}[-/]\d{1,2}(\s+\d{1,2}:\d{1,2}(:\d{1,2})?(\s*(AM|PM|上午 | 下午))?)?$"
return bool(re.match(pattern, normalize_line(line)))
def is_skip_line(line: str) -> bool:
return is_index_line(line) or is_twole_label(line) or is_zxxk_label(line) or not normalize_line(line)
def is_url_fragment(line: str) -> bool:
return bool(re.match(r"^[A-Za-z0-9._~:/?#\[\]@!$&'()*+,;=%-]+$", normalize_line(line)))
def consume_url(lines: List[str], start: int) -> Optional[Tuple[str, int]]:
i = start
while i < len(lines) and is_skip_line(lines[i]):
i += 1
if i >= len(lines):
return None
first_part = normalize_line(lines[i])
if not first_part.startswith("http"):
return None
url_parts = [first_part]
i += 1
while i < len(lines):
line = normalize_line(lines[i])
if is_date_time(line):
break
if is_skip_line(line):
i += 1
continue
if not is_url_fragment(line):
break
url_parts.append(line)
i += 1
return "".join(url_parts), i
def parse_content_item(lines: List[str], start: int) -> Optional[Tuple[str, str, str, int]]:
i = start
while i < len(lines) and is_skip_line(lines[i]):
i += 1
title_parts = []
while i < len(lines) and len(title_parts) < 8:
line = normalize_line(lines[i])
if is_url_line(line):
break
if is_date_time(line):
return None
if is_skip_line(line):
i += 1
continue
title_parts.append(line)
i += 1
while i < len(lines) and is_skip_line(lines[i]):
i += 1
url_result = consume_url(lines, i)
if not title_parts or not url_result:
return None
title = "".join(title_parts)
url, i = url_result
date = "1970-1-1"
if i < len(lines) and is_date_time(lines[i]):
date = normalize_line(lines[i])
i += 1
return title, url, date, i
def get_site_from_url(url: str) -> Optional[str]:
normalized_url = normalize_line(url).lower()
if "21cnjy.com" in normalized_url:
return "twole"
if "zxxk.com" in normalized_url:
return "zxxk"
return None
def extract_content_items(lines: List[str]) -> List[Tuple[str, str, str, str]]:
items = []
i = 0
while i < len(lines):
item = parse_content_item(lines, i)
if not item:
i += 1
continue
title, url, date, next_i = item
site = get_site_from_url(url)
if site in ("twole", "zxxk"):
items.append((title, url, date, site))
i = next_i
return items
def build_grouped_rows(items: List[Tuple[str, str, str, str]]) -> List[str]:
twole_items = [(title, url, date) for title, url, date, site in items if site == "twole"]
zxxk_items = [(title, url, date) for title, url, date, site in items if site == "zxxk"]
count = min(len(twole_items), len(zxxk_items))
return ["\t".join([twole_items[index][0], twole_items[index][1], twole_items[index][2],
zxxk_items[index][0], zxxk_items[index][1], zxxk_items[index][2]]) for index in range(count)]
def parse_site_items(lines: List[str]) -> List[str]:
items = extract_content_items(lines)
sites = [site for _, _, _, site in items]
if len(sites) < 2 or len(set(sites)) < 2:
return []
return build_grouped_rows(items)
def parse_grouped_site_rows(lines: List[str]) -> List[str]:
twole_items = []
zxxk_items = []
current_site = None
i = 0
while i < len(lines):
line = normalize_line(lines[i])
if is_twole_label(line):
current_site = "twole"
i += 1
continue
if is_zxxk_label(line):
current_site = "zxxk"
i += 1
continue
if not line or line in ("......", "……") or line.startswith("其它内容") or line.startswith("其他内容"):
i += 1
continue
if current_site not in ("twole", "zxxk"):
i += 1
continue
title = line
url = ""
date = "1970-1-1"
url_result = consume_url(lines, i + 1)
if url_result:
url, i = url_result
if i < len(lines) and is_date_time(lines[i]):
date = normalize_line(lines[i])
i += 1
else:
i += 1
if current_site == "twole":
twole_items.append((title, url, date))
else:
zxxk_items.append((title, url, date))
count = min(len(twole_items), len(zxxk_items))
return ["\t".join([twole_items[index][0], twole_items[index][1], twole_items[index][2],
zxxk_items[index][0], zxxk_items[index][1], zxxk_items[index][2]]) for index in range(count)]
lines = [normalize_line(line) for line in raw_lines if normalize_line(line)]
has_tab_in_first_rows = any("\t" in line for line in lines[:6])
split_lines = []
for line in lines:
split_lines.extend(normalize_line(part) for part in line.split("\t") if normalize_line(part))
site_result = parse_site_items(split_lines)
if site_result:
return site_result
has_site_labels = any(is_twole_label(line) or is_zxxk_label(line) for line in split_lines)
if has_site_labels:
grouped_result = parse_grouped_site_rows(split_lines)
if grouped_result:
return grouped_result
if has_tab_in_first_rows or len(split_lines) < 6:
return lines
return lines
def read_docx_content(doc_path: Path) -> List[str]:
def clean_hyperlink(line: str) -> str:
patterns = [r"HYPERLINK\s+&quot;([^&]+)&quot;\s+(\S+)", r"HYPERLINK\s+\"([^\"]+)\"\s+(\S+)"]
for pattern in patterns:
match = re.match(pattern, line)
if match:
return match.group(2)
return line
try:
with zipfile.ZipFile(doc_path, "r") as z:
content = z.read("word/document.xml").decode("utf-8")
texts = re.findall(r"<w:t[^>]*>(.*?)</w:t>", content, re.DOTALL)
processed = []
for text in texts:
text = re.sub(r"<[^>]+>", "", text).strip()
if text:
processed.append(clean_hyperlink(text))
return parse_word_content_lines(processed)
except Exception:
return []
def read_doc_content_old(doc_path: Path) -> List[str]:
for soffice_cmd in ["/Applications/LibreOffice.app/Contents/MacOS/soffice",
"/Applications/LibreOffice.app/Contents/MacOS/libreoffice", "soffice"]:
if os.path.isfile(soffice_cmd) and subprocess.run([soffice_cmd, "--version"], capture_output=True).returncode == 0:
break
else:
return []
import tempfile
try:
with tempfile.TemporaryDirectory() as tmpdir:
result = subprocess.run(
[soffice_cmd, "--headless", "--convert-to", "txt", "--outdir", tmpdir, str(doc_path)],
capture_output=True,
timeout=60,
)
if result.returncode != 0:
return []
txt_file = Path(tmpdir) / (doc_path.stem + ".txt")
if txt_file.exists():
lines = [line.strip() for line in txt_file.read_text(encoding="utf-8", errors="ignore").splitlines() if line.strip()]
return parse_word_content_lines(lines)
except Exception:
pass
return []
def read_word_content(doc_path: Path) -> List[str]:
if doc_path.suffix.lower() == ".docx":
return read_docx_content(doc_path)
if doc_path.suffix.lower() == ".doc":
return read_doc_content_old(doc_path)
return []
def find_word_doc_for_zxxk_dir(zxxk_dir: Path) -> Optional[Path]:
parent = zxxk_dir.parent
word_files = sorted(
(item for item in parent.iterdir() if item.is_file() and item.suffix.lower() in (".docx", ".doc")),
key=lambda item: item.name.lower(),
)
return word_files[0] if word_files else None
def extract_zxxk_id(url: str) -> Optional[str]:
match = re.search(r"/(?:soft/)?(\d+)\.html(?:[?#].*)?$", url or "")
if match:
return match.group(1)
numbers = re.findall(r"\d+", url or "")
return numbers[-1] if numbers else None
def extract_zxxk_metadata_rows(word_doc: Optional[Path]) -> List[Dict]:
if not word_doc:
return []
rows = []
for row_index, content in enumerate(read_word_content(word_doc)):
parts = [part.strip() for part in content.split("\t")]
if len(parts) < 6:
continue
candidates = [(parts[0], parts[1], parts[2]), (parts[3], parts[4], parts[5])]
for title, url, source_time in candidates:
if "zxxk.com" not in url.lower():
continue
rows.append({
"title": title,
"url": url,
"source_time": source_time,
"word_source_time": source_time,
"zxxk_id": extract_zxxk_id(url),
"word_doc": word_doc.name,
"word_doc_path": str(word_doc.resolve()),
"word_row_index": row_index,
})
return rows
def match_metadata_for_file(material_path: Path, metadata_rows: List[Dict]) -> Optional[Dict]:
if not metadata_rows:
return None
file_name = material_path.name
file_stem = material_path.stem
file_norm = normalize_for_match(file_stem)
exact_matches = []
for row in metadata_rows:
title_norm = normalize_for_match(row["title"])
if title_norm and (title_norm == file_norm or title_norm in file_norm or file_norm in title_norm):
exact_matches.append(row)
if exact_matches:
return max(exact_matches, key=lambda row: len(normalize_for_match(row["title"])))
scored_rows = [
(token_overlap_score(row["title"], file_name), row)
for row in metadata_rows
]
best_score, best_row = max(scored_rows, key=lambda item: item[0])
return best_row if best_score > 0 else None
def enrich_record_with_metadata(record: Dict, material_path: Path, metadata_rows: List[Dict]) -> None:
metadata = match_metadata_for_file(material_path, metadata_rows)
record["metadata_found"] = metadata is not None
if not metadata:
record.update({
"title": None,
"url": None,
"source_time": None,
"source_time_raw": None,
"source_time_mtime": None,
"word_source_time": None,
"source_time_source": None,
"classification_time_basis": None,
"child_source_time_diff_seconds": None,
"zxxk_id": None,
"word_doc": None,
"word_doc_path": None,
"word_row_index": None,
})
return
record.update(metadata)
record["source_time_source"] = "word"
class MysqlUploadTimeLookup:
def __init__(self, config: Optional[Dict] = None, table_name: str = EVIDENCE_TABLE_NAME):
self.config = dict(config or DB_CONFIG)
self.table_name = table_name
self.conn = None
self.cache: Dict[Tuple[str, str], Optional[str]] = {}
def close(self) -> None:
if self.conn:
self.conn.close()
self.conn = None
def connect(self):
if self.conn:
return self.conn
import mysql.connector
self.conn = mysql.connector.connect(**self.config)
return self.conn
def __call__(self, metadata: Dict) -> Optional[str]:
zxxk_id = metadata.get("zxxk_id") or ""
url = metadata.get("url") or ""
cache_key = (zxxk_id, url)
if cache_key not in self.cache:
self.cache[cache_key] = self.fetch_upload_time(zxxk_id, url)
return self.cache[cache_key]
def fetch_upload_time(self, zxxk_id: str, url: str) -> Optional[str]:
if not zxxk_id and not url:
return None
conn = self.connect()
cursor = conn.cursor()
try:
if zxxk_id:
cursor.execute(
f"SELECT `学科网上传日期` FROM `{self.table_name}` "
"WHERE `xkwID` = %s AND `学科网上传日期` IS NOT NULL AND `学科网上传日期` <> '' LIMIT 1",
(zxxk_id,),
)
row = cursor.fetchone()
if row and row[0]:
return str(row[0])
if url:
cursor.execute(
f"SELECT `学科网上传日期` FROM `{self.table_name}` "
"WHERE `学科网资料链接` = %s AND `学科网上传日期` IS NOT NULL AND `学科网上传日期` <> '' LIMIT 1",
(url,),
)
row = cursor.fetchone()
if row and row[0]:
return str(row[0])
finally:
cursor.close()
return None
def apply_upload_time_lookup(record: Dict, upload_time_lookup) -> None:
if not record.get("metadata_found") or not upload_time_lookup:
return
upload_time = upload_time_lookup(record)
if not upload_time:
return
normalized_time, timestamp = parse_database_datetime(upload_time)
record["source_time_raw"] = str(upload_time)
record["source_time"] = normalized_time
record["source_time_mtime"] = timestamp
record["source_time_source"] = "mysql"
def reclassify_archive_with_source_time(record: Dict) -> None:
if record.get("file_kind") != "archive" or record.get("latest_child_mtime") is None:
return
source_time_mtime = record.get("source_time_mtime")
if source_time_mtime is not None:
record["child_source_time_diff_seconds"] = record["latest_child_mtime"] - source_time_mtime
archive_diff_seconds = record.get("child_archive_time_diff_seconds")
if archive_diff_seconds is not None and abs(archive_diff_seconds) <= FIVE_MINUTES_SECONDS:
record["category"] = ARCHIVE_WITHIN_5_MINUTES
record["classification_time_basis"] = "archive_mtime_within_5_minutes"
return
if source_time_mtime is None:
record["category"] = ARCHIVE_REQUIRES_SOURCE_TIME
record["classification_time_basis"] = "requires_mysql_upload_time"
return
diff_seconds = record["child_source_time_diff_seconds"]
record["classification_time_basis"] = "mysql_upload_time"
if diff_seconds > 6000:
record["category"] = ARCHIVE_CHILD_LATER
else:
record["category"] = ARCHIVE_CHILD_EARLIER
def classify_material(material_path: Path, work_dir: Path, source_root: Optional[Path] = None) -> Dict:
material_path = Path(material_path)
file_record = build_file_record(material_path, source_root=source_root)
result = {
**file_record,
"category": SINGLE_FILE,
"file_kind": "single_file",
"archive_mtime": None,
"archive_datetime": None,
"latest_child_mtime": None,
"latest_child_datetime": None,
"child_archive_time_diff_seconds": None,
"child_source_time_diff_seconds": None,
"classification_time_basis": None,
"child_count": 0,
}
if not is_archive(material_path):
return result
result.update({
"file_kind": "archive",
"archive_mtime": file_record["mtime"],
"archive_datetime": file_record["datetime"],
})
extract_dir = Path(work_dir) / material_path.stem
child_mtime = None
child_count = 0
try:
extracted_files = extract_archive(material_path, extract_dir)
child_count = len(extracted_files)
child_mtime = latest_child_mtime_from_files(extracted_files)
except Exception as exc:
result["category"] = FAILED_ARCHIVE
result["error"] = str(exc)
return result
finally:
if extract_dir.exists():
shutil.rmtree(extract_dir)
result["child_count"] = child_count
if child_mtime is None:
result["category"] = EMPTY_ARCHIVE
return result
diff_seconds = child_mtime - file_record["mtime"]
result.update({
"category": preliminary_archive_category(file_record["mtime"], child_mtime),
"latest_child_mtime": child_mtime,
"latest_child_datetime": format_datetime(child_mtime),
"child_archive_time_diff_seconds": diff_seconds,
"classification_time_basis": "archive_mtime_initial",
})
return result
def find_zxxk_dirs(source_dir: Path) -> List[Path]:
source_dir = Path(source_dir)
if source_dir.name == "学科网" and source_dir.is_dir():
return [source_dir]
return sorted(
(path for path in source_dir.rglob("学科网") if path.is_dir()),
key=lambda path: str(path).lower(),
)
def iter_materials(zxxk_dir: Path) -> List[Path]:
return sorted(
(item for item in zxxk_dir.iterdir() if item.is_file()),
key=lambda item: item.name.lower(),
)
def empty_result(source_dir: Path) -> Dict:
return {
"source_dir": str(Path(source_dir).resolve()),
"generated_at": datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
"total_zxxk_dirs": 0,
"total_materials": 0,
"summary": {category: 0 for category in CLASSIFICATION_CATEGORIES},
"categories": {category: [] for category in CLASSIFICATION_CATEGORIES},
"other_categories": {},
"warnings": [],
}
def add_classification(result: Dict, record: Dict) -> None:
category = record["category"]
if category in result["categories"]:
result["categories"][category].append(record)
result["summary"][category] += 1
else:
result["other_categories"].setdefault(category, []).append(record)
result["total_materials"] += 1
def classify_zxxk_materials(source_dir: Path, work_dir: Path, upload_time_lookup=None) -> Dict:
source_dir = Path(source_dir)
work_dir = Path(work_dir)
result = empty_result(source_dir)
zxxk_dirs = find_zxxk_dirs(source_dir)
result["total_zxxk_dirs"] = len(zxxk_dirs)
for zxxk_dir in zxxk_dirs:
word_doc = find_word_doc_for_zxxk_dir(zxxk_dir)
metadata_rows = extract_zxxk_metadata_rows(word_doc)
for material_path in iter_materials(zxxk_dir):
archive_key = hashlib.sha1(str(material_path.resolve()).encode("utf-8")).hexdigest()[:16]
relative_work_dir = work_dir / archive_key
record = classify_material(material_path, relative_work_dir, source_root=source_dir)
enrich_record_with_metadata(record, material_path, metadata_rows)
try:
apply_upload_time_lookup(record, upload_time_lookup)
except Exception as exc:
result["warnings"].append(f"查询学科网上传日期失败:{exc}")
upload_time_lookup = None
reclassify_archive_with_source_time(record)
if relative_work_dir.exists() and not any(relative_work_dir.iterdir()):
relative_work_dir.rmdir()
try:
record["zxxk_dir"] = str(zxxk_dir.resolve().relative_to(source_dir.resolve()))
except ValueError:
record["zxxk_dir"] = str(zxxk_dir.resolve())
add_classification(result, record)
return result
def write_json(data: Dict, output_path: Path) -> None:
output_path.parent.mkdir(parents=True, exist_ok=True)
with output_path.open("w", encoding="utf-8") as f:
json.dump(data, f, ensure_ascii=False, indent=2)
def print_summary(result: Dict, output_path: Path) -> None:
print("===== 学科网资料分类完成 =====")
print(f"学科网目录数:{result['total_zxxk_dirs']}")
print(f"资料总数:{result['total_materials']}")
print(f"1、单文件{result['summary'][SINGLE_FILE]}")
print(f"2、压缩文件子文件最后时间与压缩包时间在5分钟之内{result['summary'][ARCHIVE_WITHIN_5_MINUTES]}")
print(f"3、剩余压缩文件子文件最后时间比数据库发布时间更晚{result['summary'][ARCHIVE_CHILD_LATER]}")
print(f"4、剩余压缩文件子文件最后时间比数据库发布时间更早{result['summary'][ARCHIVE_CHILD_EARLIER]}")
if result["other_categories"]:
other_total = sum(len(items) for items in result["other_categories"].values())
print(f"其他/异常:{other_total}")
print(f"分类结果:{output_path}")
def main() -> None:
parser = argparse.ArgumentParser(description="学科网资料时间分类程序")
parser.add_argument("--source-dir", type=str, default=str(BASE_DIR / "ccold"), help="待扫描目录,默认 server/cc")
parser.add_argument("--work-dir", type=str, default=str(BASE_DIR / "tmp" / "zxxk_material_classify"), help="解压临时目录")
parser.add_argument("--output", type=str, default=str(BASE_DIR / "jsons" / "zxxk_material_time_categories.json"), help="分类 JSON 输出路径")
parser.add_argument("--skip-db-time", action="store_true", help="不从 MySQL 查询学科网上传日期")
args = parser.parse_args()
source_dir = Path(args.source_dir).expanduser().resolve()
work_dir = Path(args.work_dir).expanduser().resolve()
output_path = Path(args.output).expanduser().resolve()
upload_time_lookup = None if args.skip_db_time else MysqlUploadTimeLookup()
try:
result = classify_zxxk_materials(source_dir, work_dir, upload_time_lookup=upload_time_lookup)
finally:
if upload_time_lookup:
upload_time_lookup.close()
write_json(result, output_path)
print_summary(result, output_path)
if __name__ == "__main__":
main()