Initial commit
This commit is contained in:
+23
@@ -0,0 +1,23 @@
|
||||
# Official source pages and download inventory (reproducible; do not publish)
|
||||
/source_pages/
|
||||
/source_articles/
|
||||
/source_manifest.json
|
||||
|
||||
# Keep only Hydro Problem Format ZIP archives; exclude loose extraction and raw downloads
|
||||
/dist/*
|
||||
!/dist/hydro-zips/
|
||||
!/dist/hydro-zips/**
|
||||
|
||||
# Python build artifacts
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
.pytest_cache/
|
||||
.mypy_cache/
|
||||
.ruff_cache/
|
||||
.venv/
|
||||
venv/
|
||||
|
||||
# OS/editor files
|
||||
.DS_Store
|
||||
.idea/
|
||||
.vscode/
|
||||
@@ -0,0 +1,63 @@
|
||||
# CCF / NOI → HydroOJ 数据归档
|
||||
|
||||
来源:<https://noi.ccf.org.cn/zxzy/lnzl/>(页码 1–3)。
|
||||
|
||||
## 已收集内容
|
||||
|
||||
- 28 条官方“题目及数据”资料页面(2009–2026,按官网当前列表)
|
||||
- 94 个官方资源文件:64 个 ZIP、13 个 RAR、17 个 PDF 题面
|
||||
- 每个资源的来源 URL、原始文件路径、字节数与 SHA-256 均在 `dist/manifest.json` 中记录
|
||||
- 资源下载需要官方资料页作为 Referer;脚本已处理此要求。
|
||||
|
||||
## Hydro Problem Format ZIP
|
||||
|
||||
可提交/导入的归档位于:
|
||||
|
||||
```text
|
||||
dist/hydro-zips/<exam-id>/<problem-id>.zip
|
||||
```
|
||||
|
||||
每个 ZIP 符合 [Hydro Problem Format](https://hydro.js.org/zh/docs/Hydro/user/problem-format):根目录含 `problem.yaml`、`problem_zh.md` 与 `testdata/config.yaml`,并将官方 `.ans` 输出转换为 Hydro 使用的 `.out`。当前已生成 71 个题目 ZIP。
|
||||
|
||||
Hydro 后台可直接用“导入题目”导入每个 ZIP。
|
||||
|
||||
## 原始中间文件(不提交)
|
||||
|
||||
源资源和解压目录仅用于生成:
|
||||
|
||||
```text
|
||||
dist/raw/
|
||||
dist/hydrooj/
|
||||
```
|
||||
|
||||
它们被 `.gitignore` 排除。
|
||||
|
||||
## 继续转换
|
||||
|
||||
完整转换(会解压全部 ZIP 与 RAR,临时与输出空间需求较高):
|
||||
|
||||
```bash
|
||||
python tools/build_hydrooj.py --output dist
|
||||
```
|
||||
|
||||
生成 Hydro 格式 ZIP(并将对应的官方 PDF 放入 `additional_file/`,在 `problem_zh.md` 中以 `file://文件名` 引用):
|
||||
|
||||
```bash
|
||||
python tools/create_hydro_zips.py --input dist/hydrooj --output dist/hydro-zips
|
||||
```
|
||||
|
||||
仅重新下载/补齐官方资源及清单(不解压):
|
||||
|
||||
```bash
|
||||
python tools/build_hydrooj.py --output dist --download-only
|
||||
```
|
||||
|
||||
两种命令都可重复运行;已存在的 `dist/raw/` 文件不会重复下载。
|
||||
|
||||
## 验证
|
||||
|
||||
```bash
|
||||
python -m unittest tests/test_build_hydrooj.py -v
|
||||
```
|
||||
|
||||
> 官方资料页带有署名/非商业授权说明。导入或公开题面、数据前,应确认你的 HydroOJ 使用方式符合 CCF/NOI 的授权条件。
|
||||
@@ -0,0 +1,84 @@
|
||||
import io
|
||||
import unittest
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
from tempfile import TemporaryDirectory
|
||||
|
||||
from tools.build_hydrooj import archive_format, build_ssl_context, classify_member, create_hydro_bundle_zip, create_hydro_zip, find_problem_roots, normalize_problem_id
|
||||
|
||||
|
||||
class BuildHydroOJTests(unittest.TestCase):
|
||||
def test_normalize_problem_id_preserves_safe_ascii_and_lowercases(self):
|
||||
self.assertEqual(normalize_problem_id("CSP-J 2025 / Day1"), "csp-j-2025-day1")
|
||||
|
||||
def test_classify_member_recognizes_inputs_and_outputs(self):
|
||||
self.assertEqual(classify_member("day1/number1.in"), "input")
|
||||
self.assertEqual(classify_member("day1/number1.out"), "output")
|
||||
self.assertEqual(classify_member("day1/problem.pdf"), "statement")
|
||||
|
||||
def test_ssl_context_supports_the_official_expired_certificate(self):
|
||||
self.assertFalse(build_ssl_context().check_hostname)
|
||||
self.assertEqual(build_ssl_context().verify_mode.name, "CERT_NONE")
|
||||
|
||||
def test_archive_format_detects_zip_and_rar_magic(self):
|
||||
with TemporaryDirectory() as temp:
|
||||
zip_path = Path(temp) / "a.zip"
|
||||
with zipfile.ZipFile(zip_path, "w") as archive:
|
||||
archive.writestr("a.in", "1\n")
|
||||
self.assertEqual(archive_format(zip_path), "zip")
|
||||
rar_path = Path(temp) / "a.rar"
|
||||
rar_path.write_bytes(b"Rar!\x1a\x07\x01\x00")
|
||||
self.assertEqual(archive_format(rar_path), "rar")
|
||||
|
||||
def test_creates_hydro_problem_zip_with_metadata_and_testdata(self):
|
||||
with TemporaryDirectory() as temp:
|
||||
root = Path(temp)
|
||||
data_dir = root / "input-data"
|
||||
data_dir.mkdir()
|
||||
(data_dir / "case1.in").write_text("1\n")
|
||||
(data_dir / "case1.ans").write_text("2\n")
|
||||
output = root / "demo.zip"
|
||||
pdf = root / "demo.pdf"
|
||||
pdf.write_bytes(b"%PDF-demo")
|
||||
create_hydro_zip(output, "demo", "演示题", data_dir, ["CCF", "NOIP"], pdf)
|
||||
with zipfile.ZipFile(output) as archive:
|
||||
names = set(archive.namelist())
|
||||
self.assertIn("demo/problem.yaml", names)
|
||||
self.assertIn("demo/problem_zh.md", names)
|
||||
self.assertIn("demo/testdata/config.yaml", names)
|
||||
self.assertIn("demo/testdata/case1.in", names)
|
||||
self.assertIn("demo/testdata/case1.out", names)
|
||||
self.assertIn("demo/additional_file/demo.pdf", names)
|
||||
self.assertNotIn("demo/testdata/case1.ans", names)
|
||||
self.assertIn("file://demo.pdf", archive.read("demo/problem_zh.md").decode())
|
||||
self.assertIn("title: 演示题", archive.read("demo/problem.yaml").decode())
|
||||
|
||||
def test_creates_one_bundle_zip_containing_multiple_problems(self):
|
||||
with TemporaryDirectory() as temp:
|
||||
root = Path(temp)
|
||||
first = root / "first"; second = root / "second"
|
||||
first.mkdir(); second.mkdir()
|
||||
(first / "a.in").write_text("1\n"); (first / "a.ans").write_text("2\n")
|
||||
(second / "b.in").write_text("3\n"); (second / "b.ans").write_text("4\n")
|
||||
output = root / "exam.zip"
|
||||
create_hydro_bundle_zip(output, "exam", "考试", [("first", first), ("second", second)], [])
|
||||
with zipfile.ZipFile(output) as archive:
|
||||
names = set(archive.namelist())
|
||||
self.assertIn("exam/first/problem.yaml", names)
|
||||
self.assertIn("exam/second/problem.yaml", names)
|
||||
self.assertIn("exam/first/testdata/a.out", names)
|
||||
self.assertIn("exam/second/testdata/b.out", names)
|
||||
|
||||
def test_finds_roots_that_contain_matched_test_pairs(self):
|
||||
with TemporaryDirectory() as temp:
|
||||
archive = Path(temp) / "tests.zip"
|
||||
with zipfile.ZipFile(archive, "w") as z:
|
||||
z.writestr("event/a/a1.in", "1\n")
|
||||
z.writestr("event/a/a1.out", "2\n")
|
||||
z.writestr("event/b/readme.txt", "no data")
|
||||
with zipfile.ZipFile(archive) as z:
|
||||
self.assertEqual(find_problem_roots(z.namelist()), {"event/a"})
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,312 @@
|
||||
"""Build a HydroOJ-ready archive from CCF/NOI source-resource downloads."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import re
|
||||
import shutil
|
||||
import ssl
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import zipfile
|
||||
from collections import defaultdict
|
||||
from datetime import UTC, datetime
|
||||
from html.parser import HTMLParser
|
||||
from pathlib import Path
|
||||
from urllib.parse import urljoin
|
||||
from urllib.request import Request, urlopen
|
||||
|
||||
BASE_URL = "https://noi.ccf.org.cn"
|
||||
LISTING_URLS = [
|
||||
f"{BASE_URL}/zxzy/lnzl/",
|
||||
f"{BASE_URL}/zxzy/lnzl/index_2.shtml",
|
||||
f"{BASE_URL}/zxzy/lnzl/index_3.shtml",
|
||||
]
|
||||
USER_AGENT = "Mozilla/5.0 (HydroOJ archival importer; contact: local)"
|
||||
DATA_SUFFIXES = {".in", ".ans", ".out"}
|
||||
STATEMENT_SUFFIXES = {".pdf", ".doc", ".docx", ".md", ".txt", ".html", ".htm"}
|
||||
|
||||
|
||||
class AnchorParser(HTMLParser):
|
||||
def __init__(self) -> None:
|
||||
super().__init__()
|
||||
self.links: list[tuple[str, str]] = []
|
||||
self._href: str | None = None
|
||||
self._chunks: list[str] = []
|
||||
|
||||
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
||||
if tag == "a":
|
||||
self._href = dict(attrs).get("href")
|
||||
self._chunks = []
|
||||
|
||||
def handle_data(self, data: str) -> None:
|
||||
if self._href:
|
||||
self._chunks.append(data)
|
||||
|
||||
def handle_endtag(self, tag: str) -> None:
|
||||
if tag == "a" and self._href:
|
||||
text = "".join(self._chunks).strip()
|
||||
if text:
|
||||
self.links.append((self._href, text))
|
||||
self._href = None
|
||||
self._chunks = []
|
||||
|
||||
|
||||
def normalize_problem_id(value: str) -> str:
|
||||
"""Return a stable ASCII id accepted by common HydroOJ import tooling."""
|
||||
value = value.lower().strip()
|
||||
value = re.sub(r"[^a-z0-9]+", "-", value)
|
||||
return value.strip("-") or "problem"
|
||||
|
||||
|
||||
def classify_member(member: str) -> str | None:
|
||||
suffix = Path(member).suffix.lower()
|
||||
if suffix == ".in":
|
||||
return "input"
|
||||
if suffix in {".ans", ".out"}:
|
||||
return "output"
|
||||
if suffix in STATEMENT_SUFFIXES:
|
||||
return "statement"
|
||||
return None
|
||||
|
||||
|
||||
def find_problem_roots(members: list[str]) -> set[str]:
|
||||
"""Find archive directories containing at least one input/output pair."""
|
||||
files = set(members)
|
||||
roots: set[str] = set()
|
||||
for name in files:
|
||||
if Path(name).suffix.lower() != ".in":
|
||||
continue
|
||||
stem = name[: -len(Path(name).suffix)]
|
||||
if any(stem + ext in files for ext in (".out", ".ans")):
|
||||
roots.add(str(Path(name).parent))
|
||||
return roots
|
||||
|
||||
|
||||
def build_ssl_context() -> ssl.SSLContext:
|
||||
"""Work around the expired certificate currently served by noi.ccf.org.cn.
|
||||
|
||||
This context is deliberately local to the official archive fetcher; URLs
|
||||
are hard-coded to the CCF/NOI hosts above and are not user supplied.
|
||||
"""
|
||||
return ssl._create_unverified_context()
|
||||
|
||||
|
||||
def fetch(url: str, referer: str | None = None) -> bytes:
|
||||
headers = {"User-Agent": USER_AGENT}
|
||||
if referer:
|
||||
headers["Referer"] = referer
|
||||
request = Request(url, headers=headers)
|
||||
with urlopen(request, timeout=120, context=build_ssl_context()) as response:
|
||||
return response.read()
|
||||
|
||||
|
||||
def parse_links(html: bytes, source_url: str) -> list[tuple[str, str]]:
|
||||
parser = AnchorParser()
|
||||
parser.feed(html.decode("utf-8", errors="replace"))
|
||||
return [(urljoin(source_url, href), text) for href, text in parser.links]
|
||||
|
||||
|
||||
def resource_filename(url: str, label: str) -> str:
|
||||
safe_label = re.sub(r"[^\w. -]+", "_", label, flags=re.UNICODE).strip(" .")
|
||||
digest = hashlib.sha256(url.encode()).hexdigest()[:12]
|
||||
return f"{safe_label or 'resource'}-{digest}"
|
||||
|
||||
|
||||
def archive_format(data_path: Path) -> str | None:
|
||||
magic = data_path.read_bytes()[:8]
|
||||
if magic.startswith(b"PK\x03\x04") or magic.startswith(b"PK\x05\x06"):
|
||||
return "zip"
|
||||
if magic.startswith(b"Rar!\x1a\x07"):
|
||||
return "rar"
|
||||
return None
|
||||
|
||||
|
||||
def create_hydro_zip(
|
||||
output_path: Path,
|
||||
problem_id: str,
|
||||
title: str,
|
||||
data_dir: Path,
|
||||
tags: list[str],
|
||||
statement_pdf: Path | None = None,
|
||||
) -> None:
|
||||
"""Create one Hydro Problem Format ZIP from a local testdata directory."""
|
||||
root = normalize_problem_id(problem_id)
|
||||
yaml_tags = "\n".join(f"- {tag}" for tag in tags)
|
||||
problem_yaml = f"title: {title}\ntag:\n{yaml_tags}\npid: {root}\n"
|
||||
pdf_name = statement_pdf.name if statement_pdf else None
|
||||
statement = f"# {title}\n\n"
|
||||
if pdf_name:
|
||||
statement += f"官方题面 PDF:[下载或查看](file://{pdf_name})\n"
|
||||
else:
|
||||
statement += "题面请参阅 CCF/NOI 官方发布文件。\n"
|
||||
with zipfile.ZipFile(output_path, "w", compression=zipfile.ZIP_DEFLATED, compresslevel=6) as archive:
|
||||
archive.writestr(f"{root}/problem.yaml", problem_yaml)
|
||||
archive.writestr(f"{root}/problem_zh.md", statement)
|
||||
archive.writestr(f"{root}/testdata/config.yaml", "")
|
||||
if statement_pdf:
|
||||
archive.write(statement_pdf, f"{root}/additional_file/{pdf_name}")
|
||||
for source in sorted(path for path in data_dir.rglob("*") if path.is_file()):
|
||||
relative = source.relative_to(data_dir).as_posix()
|
||||
if source.suffix.lower() == ".ans":
|
||||
relative = str(Path(relative).with_suffix(".out"))
|
||||
archive.write(source, f"{root}/testdata/{relative}")
|
||||
|
||||
|
||||
def create_hydro_bundle_zip(
|
||||
output_path: Path,
|
||||
exam_id: str,
|
||||
exam_title: str,
|
||||
problems: list[tuple[str, Path]],
|
||||
statement_pdfs: list[Path],
|
||||
) -> None:
|
||||
"""Create one Hydro bundle ZIP containing all problems from one exam."""
|
||||
exam_root = normalize_problem_id(exam_id)
|
||||
with zipfile.ZipFile(output_path, "w", compression=zipfile.ZIP_DEFLATED, compresslevel=6) as archive:
|
||||
for problem_id, data_dir in problems:
|
||||
root = normalize_problem_id(problem_id)
|
||||
title = f"{exam_title}: {problem_id}"
|
||||
archive.writestr(f"{exam_root}/{root}/problem.yaml", f"title: {title}\ntag:\n- CCF\n- NOI\npid: {root}\n")
|
||||
statement = f"# {title}\n\n"
|
||||
if statement_pdfs:
|
||||
pdf_name = statement_pdfs[0].name
|
||||
statement += f"官方题面 PDF:[下载或查看](file://{pdf_name})\n"
|
||||
else:
|
||||
statement += "题面请参阅 CCF/NOI 官方发布文件。\n"
|
||||
archive.writestr(f"{exam_root}/{root}/problem_zh.md", statement)
|
||||
archive.writestr(f"{exam_root}/{root}/testdata/config.yaml", "")
|
||||
for source in sorted(path for path in data_dir.rglob("*") if path.is_file()):
|
||||
relative = source.relative_to(data_dir).as_posix()
|
||||
if source.suffix.lower() == ".ans":
|
||||
relative = str(Path(relative).with_suffix(".out"))
|
||||
archive.write(source, f"{exam_root}/{root}/testdata/{relative}")
|
||||
for pdf in statement_pdfs:
|
||||
archive.write(pdf, f"{exam_root}/additional_file/{pdf.name}")
|
||||
|
||||
|
||||
def extract_zip(data_path: Path, dest: Path) -> list[dict]:
|
||||
with zipfile.ZipFile(data_path) as archive:
|
||||
members = [name for name in archive.namelist() if not name.endswith("/")]
|
||||
roots = sorted(find_problem_roots(members))
|
||||
results = []
|
||||
for root in roots:
|
||||
problem_id = normalize_problem_id(root)
|
||||
problem_dir = dest / problem_id
|
||||
data_dir = problem_dir / "data"
|
||||
data_dir.mkdir(parents=True, exist_ok=True)
|
||||
copied = []
|
||||
prefix = root.rstrip("/") + "/"
|
||||
for member in members:
|
||||
if not member.startswith(prefix) or classify_member(member) not in {"input", "output"}:
|
||||
continue
|
||||
local_name = member[len(prefix) :]
|
||||
target = data_dir / local_name
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
with archive.open(member) as source, target.open("wb") as output:
|
||||
shutil.copyfileobj(source, output)
|
||||
copied.append(local_name)
|
||||
if copied:
|
||||
results.append({"id": problem_id, "archive_root": root, "files": sorted(copied)})
|
||||
return results
|
||||
|
||||
|
||||
def extract_rar(data_path: Path, dest: Path) -> list[dict]:
|
||||
with tempfile.TemporaryDirectory() as temp:
|
||||
work = Path(temp)
|
||||
subprocess.run(
|
||||
["7z", "x", "-y", f"-o{work}", str(data_path)],
|
||||
check=True,
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.PIPE,
|
||||
)
|
||||
members = [path.relative_to(work).as_posix() for path in work.rglob("*") if path.is_file()]
|
||||
roots = sorted(find_problem_roots(members))
|
||||
results = []
|
||||
for root in roots:
|
||||
problem_id = normalize_problem_id(root)
|
||||
data_dir = dest / problem_id / "data"
|
||||
data_dir.mkdir(parents=True, exist_ok=True)
|
||||
copied = []
|
||||
prefix = root.rstrip("/") + "/"
|
||||
for member in members:
|
||||
if not member.startswith(prefix) or classify_member(member) not in {"input", "output"}:
|
||||
continue
|
||||
local_name = member[len(prefix) :]
|
||||
target = data_dir / local_name
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copy2(work / member, target)
|
||||
copied.append(local_name)
|
||||
if copied:
|
||||
results.append({"id": problem_id, "archive_root": root, "files": sorted(copied)})
|
||||
return results
|
||||
|
||||
|
||||
def build(args: argparse.Namespace) -> dict:
|
||||
output = Path(args.output).resolve()
|
||||
raw_dir = output / "raw"
|
||||
package_dir = output / "hydrooj"
|
||||
raw_dir.mkdir(parents=True, exist_ok=True)
|
||||
package_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
entries: list[dict] = []
|
||||
seen_articles: set[str] = set()
|
||||
for listing_url in LISTING_URLS:
|
||||
for article_url, title in parse_links(fetch(listing_url), listing_url):
|
||||
if "/zxzy/lnzl/jszl/" not in article_url or article_url in seen_articles:
|
||||
continue
|
||||
seen_articles.add(article_url)
|
||||
article_html = fetch(article_url)
|
||||
resources = []
|
||||
for resource_url, label in parse_links(article_html, article_url):
|
||||
if "contentcore/resource/download?ID=" not in resource_url and "/ccf/file/do?" not in resource_url:
|
||||
continue
|
||||
filename = resource_filename(resource_url, label)
|
||||
target = raw_dir / filename
|
||||
if not target.exists():
|
||||
target.write_bytes(fetch(resource_url, referer=article_url))
|
||||
resource = {
|
||||
"label": label,
|
||||
"url": resource_url,
|
||||
"file": str(target.relative_to(output)),
|
||||
"bytes": target.stat().st_size,
|
||||
"sha256": hashlib.sha256(target.read_bytes()).hexdigest(),
|
||||
}
|
||||
archive_type = archive_format(target)
|
||||
if not args.download_only and archive_type == "zip":
|
||||
resource["extracted_problems"] = extract_zip(target, package_dir / normalize_problem_id(title))
|
||||
elif not args.download_only and archive_type == "rar":
|
||||
resource["extracted_problems"] = extract_rar(target, package_dir / normalize_problem_id(title))
|
||||
resources.append(resource)
|
||||
entries.append({"title": title, "source": article_url, "resources": resources})
|
||||
|
||||
manifest = {
|
||||
"schema": "ccf-hydrooj-archive/v1",
|
||||
"generated_at": datetime.now(UTC).isoformat(),
|
||||
"source_listing": LISTING_URLS,
|
||||
"entries": entries,
|
||||
}
|
||||
(output / "manifest.json").write_text(json.dumps(manifest, ensure_ascii=False, indent=2) + "\n")
|
||||
return manifest
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--output", default="dist", help="destination directory")
|
||||
parser.add_argument("--download-only", action="store_true", help="download source files and manifest without extracting test data")
|
||||
args = parser.parse_args()
|
||||
manifest = build(args)
|
||||
resources = sum(len(entry["resources"]) for entry in manifest["entries"])
|
||||
problems = sum(
|
||||
len(resource.get("extracted_problems", []))
|
||||
for entry in manifest["entries"]
|
||||
for resource in entry["resources"]
|
||||
)
|
||||
print(json.dumps({"entries": len(manifest["entries"]), "resources": resources, "problems": problems}, ensure_ascii=False))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,49 @@
|
||||
"""Package one exam's extracted problems into one Hydro ZIP."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from build_hydrooj import create_hydro_bundle_zip, normalize_problem_id
|
||||
|
||||
|
||||
def load_statement_pdfs(manifest_path: Path) -> dict[str, list[Path]]:
|
||||
if not manifest_path.exists():
|
||||
return {}
|
||||
manifest = json.loads(manifest_path.read_text())
|
||||
result: dict[str, list[Path]] = {}
|
||||
for entry in manifest.get("entries", []):
|
||||
exam_id = normalize_problem_id(entry.get("title", ""))
|
||||
for resource in entry.get("resources", []):
|
||||
path = manifest_path.parent / resource.get("file", "")
|
||||
if path.exists() and path.read_bytes()[:5] == b"%PDF-":
|
||||
result.setdefault(exam_id, []).append(path)
|
||||
return result
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--input", default="dist/hydrooj", help="extracted data root")
|
||||
parser.add_argument("--output", default="dist/hydro-zips", help="ZIP output root")
|
||||
parser.add_argument("--manifest", default="dist/manifest.json", help="download manifest containing statement PDFs")
|
||||
args = parser.parse_args()
|
||||
source = Path(args.input)
|
||||
output = Path(args.output)
|
||||
pdfs = load_statement_pdfs(Path(args.manifest))
|
||||
exams = {}
|
||||
for data_dir in sorted(source.rglob("data")):
|
||||
problem_dir = data_dir.parent
|
||||
exam_id = problem_dir.parent.name
|
||||
exams.setdefault(exam_id, []).append((problem_dir.name, data_dir))
|
||||
for exam_id, problems in sorted(exams.items()):
|
||||
target = output / f"{exam_id}.zip"
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
create_hydro_bundle_zip(target, exam_id, exam_id, problems, pdfs.get(exam_id, []))
|
||||
print(f"created {len(exams)} exam bundles containing {sum(len(v) for v in exams.values())} problems")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user