This commit is contained in:
biss committed 2026-07-18 09:13:04 +08:00
1 parent 271f0c4efd
commit 8bbc9da1e5
62 files changed
+2048 -9

No files matched your search

+116
View File
@@ -0,0 +1,116 @@
import json
import math
from pathlib import Path
from PIL import Image, ImageDraw, ImageFont, ImageOps
ROOT = Path(__file__).resolve().parents[1]
BASE = ROOT / ".tmp" / "gzhh-images"
RAW = BASE / "raw"
MANIFEST = BASE / "manifest.json"
AUDIT = BASE / "audit.json"
SHEETS = BASE / "contact-sheets"
CELL_WIDTH = 280
CELL_HEIGHT = 220
COLS = 5
ROWS = 4
PER_SHEET = COLS * ROWS
def font(size: int):
paths = [
Path("C:/Windows/Fonts/msyh.ttc"),
Path("C:/Windows/Fonts/simhei.ttf"),
]
for path in paths:
if path.exists():
return ImageFont.truetype(str(path), size)
return ImageFont.load_default()
small_font = font(14)
label_font = font(16)
title_font = font(22)
manifest = json.loads(MANIFEST.read_text(encoding="utf-8"))
unique_entries = []
seen_files = set()
for entry in manifest:
file_name = entry.get("file")
if entry.get("status") != "ok" or not file_name or file_name in seen_files:
continue
seen_files.add(file_name)
image_path = RAW / file_name
record = {
"id": len(unique_entries) + 1,
"file": file_name,
"path": image_path.relative_to(ROOT).as_posix(),
"sources": entry.get("sources", []),
"alts": entry.get("alts", []),
"bytes": entry.get("bytes", 0),
"hash": entry.get("hash", ""),
}
try:
with Image.open(image_path) as image:
record["width"], record["height"] = image.size
record["format"] = image.format
preview = image.convert("RGB")
preview.thumbnail((128, 128))
record["entropy"] = round(preview.convert("L").entropy(), 3)
record["candidate"] = (
record["width"] * record["height"] >= 200_000
and min(record["width"], record["height"]) >= 280
and 0.32 <= record["width"] / record["height"] <= 3.2
and record["entropy"] >= 3.2
)
except Exception as error:
record["error"] = str(error)
record["candidate"] = False
unique_entries.append(record)
AUDIT.write_text(json.dumps(unique_entries, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
SHEETS.mkdir(parents=True, exist_ok=True)
candidates = [item for item in unique_entries if item.get("candidate")]
for sheet_index in range(math.ceil(len(candidates) / PER_SHEET)):
page_items = candidates[sheet_index * PER_SHEET : (sheet_index + 1) * PER_SHEET]
sheet = Image.new("RGB", (CELL_WIDTH * COLS, 48 + CELL_HEIGHT * ROWS), "#f7f3e8")
draw = ImageDraw.Draw(sheet)
draw.text(
(18, 10),
f"公众号图片候选 {sheet_index + 1}/{math.ceil(len(candidates) / PER_SHEET)} · {len(candidates)} 张",
fill="#1f2b2a",
font=title_font,
)
for item_index, item in enumerate(page_items):
col = item_index % COLS
row = item_index // COLS
x = col * CELL_WIDTH
y = 48 + row * CELL_HEIGHT
draw.rectangle((x + 5, y + 5, x + CELL_WIDTH - 5, y + CELL_HEIGHT - 5), fill="#ffffff", outline="#d6d7cf")
try:
with Image.open(ROOT / item["path"]) as source:
preview = ImageOps.contain(source.convert("RGB"), (CELL_WIDTH - 24, 154))
px = x + (CELL_WIDTH - preview.width) // 2
py = y + 12 + (154 - preview.height) // 2
sheet.paste(preview, (px, py))
except Exception:
pass
source_name = Path(item["sources"][0]).stem if item["sources"] else "未知来源"
source_name = source_name[:20]
draw.text((x + 12, y + 170), f"#{item['id']:03d} {item['width']}×{item['height']}", fill="#376d5a", font=label_font)
draw.text((x + 12, y + 192), source_name, fill="#596563", font=small_font)
sheet.save(SHEETS / f"sheet-{sheet_index + 1:02d}.jpg", quality=90)
summary = {
"unique": len(unique_entries),
"candidates": len(candidates),
"rejected_by_shape_or_size": len(unique_entries) - len(candidates),
"sheets": math.ceil(len(candidates) / PER_SHEET),
}
print(json.dumps(summary, ensure_ascii=False))
+108
View File
@@ -0,0 +1,108 @@
import { createHash } from "node:crypto";
import { mkdir, readdir, readFile, writeFile } from "node:fs/promises";
import path from "node:path";
import { fileURLToPath } from "node:url";
const projectRoot = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..");
const sourceDirectory = path.join(projectRoot, "gzhh");
const outputDirectory = path.join(projectRoot, ".tmp", "gzhh-images", "raw");
const manifestPath = path.join(projectRoot, ".tmp", "gzhh-images", "manifest.json");
const concurrency = 8;
const maxBytes = 20 * 1024 * 1024;
const extensionByType = new Map([
["image/jpeg", ".jpg"],
["image/png", ".png"],
["image/webp", ".webp"],
["image/gif", ".gif"],
["image/svg+xml", ".svg"]
]);
const markdownFiles = (await readdir(sourceDirectory)).filter((name) => name.endsWith(".md"));
const imageMap = new Map();
for (const source of markdownFiles) {
const markdown = await readFile(path.join(sourceDirectory, source), "utf8");
for (const match of markdown.matchAll(/!\[([^\]]*)\]\((https?:\/\/[^\s\)]+)\)/g)) {
const alt = match[1].trim();
const url = match[2].replaceAll("&amp;", "&").replaceAll("\\_", "_");
let parsed;
try {
parsed = new URL(url);
} catch {
continue;
}
if (parsed.hostname !== "mmbiz.qpic.cn") continue;
const existing = imageMap.get(parsed.href) ?? { url: parsed.href, sources: [], alts: [] };
if (!existing.sources.includes(source)) existing.sources.push(source);
if (alt && !existing.alts.includes(alt)) existing.alts.push(alt);
imageMap.set(parsed.href, existing);
}
}
await mkdir(outputDirectory, { recursive: true });
const entries = [...imageMap.values()];
const results = new Array(entries.length);
const contentFiles = new Map();
let cursor = 0;
let completed = 0;
const download = async (entry, index) => {
try {
const response = await fetch(entry.url, {
headers: {
"user-agent": "Mozilla/5.0 (compatible; ClassArchive/1.0)",
referer: "https://mp.weixin.qq.com/"
},
signal: AbortSignal.timeout(45_000)
});
if (!response.ok) throw new Error(`HTTP ${response.status}`);
const contentType = response.headers.get("content-type")?.split(";")[0].toLowerCase() ?? "";
if (!contentType.startsWith("image/")) throw new Error(`Unexpected content type: ${contentType || "unknown"}`);
const declaredLength = Number(response.headers.get("content-length") ?? 0);
if (declaredLength > maxBytes) throw new Error(`Image exceeds ${maxBytes} bytes`);
const buffer = Buffer.from(await response.arrayBuffer());
if (buffer.byteLength > maxBytes) throw new Error(`Image exceeds ${maxBytes} bytes`);
const hash = createHash("sha256").update(buffer).digest("hex");
const extension = extensionByType.get(contentType) ?? ".img";
let file = contentFiles.get(hash);
let duplicate = true;
if (!file) {
duplicate = false;
file = `${String(index + 1).padStart(4, "0")}-${hash.slice(0, 12)}${extension}`;
await writeFile(path.join(outputDirectory, file), buffer);
contentFiles.set(hash, file);
}
return { ...entry, status: "ok", file, contentType, bytes: buffer.byteLength, hash, duplicate };
} catch (error) {
return { ...entry, status: "error", error: error instanceof Error ? error.message : String(error) };
}
};
const worker = async () => {
while (true) {
const index = cursor++;
if (index >= entries.length) return;
results[index] = await download(entries[index], index);
completed += 1;
if (completed % 50 === 0 || completed === entries.length) {
console.log(`Downloaded ${completed}/${entries.length}`);
}
}
};
await Promise.all(Array.from({ length: concurrency }, () => worker()));
await writeFile(manifestPath, `${JSON.stringify(results, null, 2)}\n`, "utf8");
const successful = results.filter((item) => item.status === "ok");
console.log(
JSON.stringify({
references: entries.length,
successful: successful.length,
failed: results.length - successful.length,
uniqueContent: successful.filter((item) => !item.duplicate).length,
bytes: successful.filter((item) => !item.duplicate).reduce((sum, item) => sum + item.bytes, 0),
manifest: path.relative(projectRoot, manifestPath)
})
);
+56
View File
@@ -0,0 +1,56 @@
import json
from pathlib import Path
from PIL import Image, ImageOps
ROOT = Path(__file__).resolve().parents[1]
BASE = ROOT / ".tmp" / "gzhh-images"
AUDIT = BASE / "audit.json"
OUTPUT = ROOT / "public" / "assets" / "wechat-archive"
SELECTED_IDS = [
81, 84, 85, 88, 89,
117, 121, 129, 132,
145, 146, 150, 151, 152, 153,
159, 160, 161, 162,
176, 178, 180, 181,
193, 196, 197,
368, 370, 373, 375,
382, 387, 388, 391,
396, 405, 408, 411,
446, 455, 459,
470, 480, 490, 499,
]
records = {item["id"]: item for item in json.loads(AUDIT.read_text(encoding="utf-8"))}
OUTPUT.mkdir(parents=True, exist_ok=True)
exported = []
for image_id in SELECTED_IDS:
record = records[image_id]
source_path = ROOT / record["path"]
output_name = f"wechat-{image_id:03d}.webp"
output_path = OUTPUT / output_name
with Image.open(source_path) as image:
frame = ImageOps.exif_transpose(image).convert("RGB")
frame.thumbnail((1800, 1800), Image.Resampling.LANCZOS)
frame.save(output_path, "WEBP", quality=84, method=6)
exported.append({
"id": image_id,
"file": output_name,
"source": record.get("sources", [""])[0],
"width": frame.width,
"height": frame.height,
"bytes": output_path.stat().st_size,
})
(BASE / "selected-manifest.json").write_text(
json.dumps(exported, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
print(json.dumps({
"selected": len(exported),
"bytes": sum(item["bytes"] for item in exported),
"output": str(OUTPUT.relative_to(ROOT)),
}, ensure_ascii=False))
Loaded 3 of 62 files, more files were not shown because too many files have changed in this diff. Show more