feat: add scripts for fetching arXiv papers and generating structured reviews
Sync to site1 / sync (push) Has been cancelled
Sync to site1 / sync (push) Has been cancelled
- Implemented `fetch_crowdroom_papers.py` to scrape arXiv papers related to CrowdRoom with customizable query topics and output options. - Created `gen_review_from_json.py` to render structured markdown reviews from search results JSON, including statistics and insights on papers and GitHub repositories. - Developed `search_info.py` for thematic searches on arXiv and GitHub, supporting HTTP/SOCKS5 proxies and structured output.
This commit is contained in:
@@ -0,0 +1,697 @@
|
||||
"""从抓取的 JSON 构建 CrowdRoom 综述 md。
|
||||
|
||||
数据源:
|
||||
1. research/crowdroom_papers_raw.json (本次新抓取;可能为空)
|
||||
2. research/search_results.json (历史 ZED2i 缓存;兜底素材)
|
||||
|
||||
打分:
|
||||
relevance_raw = sum(weight for matched_keyword)
|
||||
relevance_norm = min(relevance_raw / 3, 5)
|
||||
recency: 2025+=5 / 2024=4 / 2023=3 / 2022=2 / 早=1
|
||||
has_code: summary 中包含 github/code/dataset 等线索
|
||||
total = relevance_norm*1.0 + recency*0.6 + has_code*1.0 (满分 ≈ 10.6)
|
||||
"""
|
||||
from __future__ import annotations
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
from collections import defaultdict
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
RELEVANCE_KEYWORDS = {
|
||||
"roomplan": 3, "iphone lidar": 3, "iphone scan": 3, "arkit": 3,
|
||||
"scan-to-bim": 3, "scan2bim": 3, "indoor digital twin": 3,
|
||||
"room layout": 3, "scene layout": 3, "furniture layout": 3,
|
||||
"text-to-3d": 3, "image-to-3d": 3, "single-view 3d": 3, "single image 3d": 3,
|
||||
"indoor reconstruction": 2, "indoor scene": 2, "indoor scan": 2,
|
||||
"gaussian splatting": 2, "3dgs": 2, "3d gaussian": 2,
|
||||
"nerf": 2, "neural radiance field": 2, "neural radiance fields": 2,
|
||||
"rgb-d": 2, "rgbd": 2, "depth camera": 2,
|
||||
"9dof pose": 2, "oriented bounding box": 2, "object pose estimation": 2, "6d pose": 2,
|
||||
"scene graph": 2, "scene generation": 2, "room generation": 2,
|
||||
"furniture": 2, "household object": 2, "indoor object": 2,
|
||||
"asset generation": 2, "3d asset": 2, "3d content generation": 2,
|
||||
"crowdsourced": 2, "crowdsourcing": 2,
|
||||
"digital twin": 2, " bim ": 2,
|
||||
"usd ": 2, "opensud": 2, "gltf": 2, "universal scene description": 2,
|
||||
"mesh reconstruction": 1, "point cloud": 1, "panoptic": 1,
|
||||
"semantic segmentation": 1, "instance segmentation": 1,
|
||||
"diffusion": 1, "transformer": 1,
|
||||
"augmented reality": 1, "mixed reality": 1,
|
||||
"scene understanding": 1, "3d reconstruction": 1,
|
||||
"slam": 1, "depth estimation": 1, "indoor": 1,
|
||||
}
|
||||
|
||||
|
||||
def score_paper(p: dict) -> dict:
|
||||
text = " " + (p.get("title", "") + " " + p.get("summary", "")).lower() + " "
|
||||
hits = []
|
||||
raw = 0
|
||||
for kw, w in RELEVANCE_KEYWORDS.items():
|
||||
if kw in text:
|
||||
hits.append((kw.strip(), w))
|
||||
raw += w
|
||||
relevance_norm = min(raw / 3.0, 5.0)
|
||||
|
||||
pub = p.get("published", "")[:4]
|
||||
year = int(pub) if pub.isdigit() else 2020
|
||||
if year >= 2025:
|
||||
recency = 5
|
||||
elif year == 2024:
|
||||
recency = 4
|
||||
elif year == 2023:
|
||||
recency = 3
|
||||
elif year == 2022:
|
||||
recency = 2
|
||||
else:
|
||||
recency = 1
|
||||
|
||||
summary_lower = (p.get("summary", "") or "").lower()
|
||||
has_code = any(kw in summary_lower for kw in [
|
||||
"github.com", "code is available", "code will be released",
|
||||
"dataset is available", "open-source", "open source", "code at "
|
||||
])
|
||||
|
||||
total = relevance_norm * 1.0 + recency * 0.6 + (1.0 if has_code else 0.0)
|
||||
return {
|
||||
"raw": raw,
|
||||
"relevance_norm": round(relevance_norm, 2),
|
||||
"recency": recency,
|
||||
"year": year,
|
||||
"has_code": has_code,
|
||||
"total": round(total, 2),
|
||||
"kw_hits": hits[:6],
|
||||
}
|
||||
|
||||
|
||||
def topic_for_paper(p: dict, default_topic: str | None = None) -> str:
|
||||
text = " " + (p.get("title", "") + " " + p.get("summary", "")).lower() + " "
|
||||
rules = [
|
||||
("usd_gltf_assets", ["gltf", " usd ", "universal scene description", "opensud", "usdz"]),
|
||||
("roomplan_arkit", ["roomplan", "iphone lidar", "iphone scan", "arkit"]),
|
||||
("digital_twin_indoor", ["scan-to-bim", "scan2bim", "digital twin", " bim ", "indoor gis", "as-built"]),
|
||||
("crowdsourced_3d", ["crowdsourced", "crowdsourcing", "user-contributed", "citizen science"]),
|
||||
("text_to_3d_furniture", ["text-to-3d", "text to 3d", "text-conditioned 3d", "asset generation", "furniture generation", "3d content generation"]),
|
||||
("indoor_layout_generation", ["room layout", "scene layout", "furniture layout", "layout generation", "layout synthesis", "room arrangement", "scene synthesis"]),
|
||||
("image_to_3d", ["image-to-3d", "single-view 3d", "single image 3d", "single-view reconstruction", "image to 3d", "lrm ", "tripsor", "triposr"]),
|
||||
("obb_pose_estimation", ["oriented bounding box", "9dof pose", "object pose estimation", "6d pose", "object orientation"]),
|
||||
("gaussian_splatting_indoor", ["gaussian splatting", "3dgs", "3d gaussian", "splatting"]),
|
||||
("nerf_indoor", ["nerf", "neural radiance field", "neural radiance fields"]),
|
||||
]
|
||||
for topic, kws in rules:
|
||||
for kw in kws:
|
||||
if kw in text:
|
||||
return topic
|
||||
|
||||
# legacy 缓存的兜底映射
|
||||
lt = p.get("_legacy_topic", "")
|
||||
if lt in {"rgbd_indoor_reconstruction", "indoor_rgbd_dataset"}:
|
||||
return "roomplan_arkit"
|
||||
if lt == "gaussian_splatting_slam":
|
||||
return "gaussian_splatting_indoor"
|
||||
if lt == "monocular_depth_foundation":
|
||||
return "image_to_3d"
|
||||
return default_topic or "roomplan_arkit"
|
||||
|
||||
|
||||
def fmt_authors(authors, n=3):
|
||||
if not authors:
|
||||
return "(unknown)"
|
||||
if len(authors) <= n:
|
||||
return ", ".join(authors)
|
||||
return ", ".join(authors[:n]) + f" et al."
|
||||
|
||||
|
||||
def load_fresh(path: Path) -> list[dict]:
|
||||
if not path.exists():
|
||||
return []
|
||||
data = json.loads(path.read_text(encoding="utf-8"))
|
||||
out = []
|
||||
for p in data.get("unique_papers", []) or []:
|
||||
p = dict(p)
|
||||
p["_source"] = "fresh_fetch"
|
||||
out.append(p)
|
||||
return out
|
||||
|
||||
|
||||
def load_legacy(path: Path) -> list[dict]:
|
||||
if not path.exists():
|
||||
return []
|
||||
data = json.loads(path.read_text(encoding="utf-8"))
|
||||
out = []
|
||||
for src_topic, plist in data.get("arxiv", {}).items():
|
||||
for p in plist:
|
||||
p = dict(p)
|
||||
p["_source"] = "legacy_cache"
|
||||
p["_legacy_topic"] = src_topic
|
||||
out.append(p)
|
||||
return out
|
||||
|
||||
|
||||
TOPIC_META = {
|
||||
"roomplan_arkit": {
|
||||
"zh": "1. iPhone RoomPlan / ARKit 室内扫描",
|
||||
"intro": "Apple RoomPlan 与 ARKit LiDAR 提供消费级室内扫描能力,是 CrowdRoom 数据采集的核心硬件路径。本节关注 iPhone 端几何精度、墙体/家具语义分割、与桌面 CAD 工具的互操作。",
|
||||
},
|
||||
"gaussian_splatting_indoor": {
|
||||
"zh": "2. 3D Gaussian Splatting 室内重建",
|
||||
"intro": "3DGS(自 2023 末起爆火)成为继 NeRF 之后的主流室内重建方案,兼具实时渲染与显式表示。CrowdRoom Web 端可直接消费 3DGS 资产做背景重建展示。",
|
||||
},
|
||||
"nerf_indoor": {
|
||||
"zh": "3. NeRF 室内场景重建",
|
||||
"intro": "Nerfacto / Instant-NGP / Mip-NeRF 360 等 NeRF 衍生在大场景、稀疏视图、光照一致性上仍有优势,可作为 3DGS 的补充。",
|
||||
},
|
||||
"text_to_3d_furniture": {
|
||||
"zh": "4. text-to-3D 家具与 3D 资产生成",
|
||||
"intro": "用户扫描得到的家具往往粗糙缺失,需替换为高质量 CG 资产。text-to-3D(LRM / TRELLIS / Hunyuan3D 等)使'按描述生成家具'成为可能,是 §11 资产库的关键产线。",
|
||||
},
|
||||
"indoor_layout_generation": {
|
||||
"zh": "5. 室内布局生成与房间布置合成",
|
||||
"intro": "给定空房间几何,自动布置家具,对应 CrowdRoom '空房间装修建议'。主流方法包括 diffusion-based、autoregressive、scene-graph-guided 三类。",
|
||||
},
|
||||
"obb_pose_estimation": {
|
||||
"zh": "6. 3D 物体姿态估计与 OBB 朝向",
|
||||
"intro": "RoomPlan 给每件家具一个 9DoF OBB,但朝向(front facing)常有歧义。需要额外的 pose / orientation 模型,给 anchor 一个稳定标识,对应 §5 物体替换核心。",
|
||||
},
|
||||
"crowdsourced_3d": {
|
||||
"zh": "7. 众包 3D 数据采集与质量保障",
|
||||
"intro": "CrowdRoom 本质是众包平台。关注:如何激励用户上传、如何评估数据质量、如何聚合多次采集得到稳定 ground-truth。",
|
||||
},
|
||||
"digital_twin_indoor": {
|
||||
"zh": "8. 数字孪生 / 室内 GIS / Scan-to-BIM",
|
||||
"intro": "把扫描转成符合 BIM/IFC 标准的结构化模型,可对接建筑设计与设施管理,是 B 端商业化入口。",
|
||||
},
|
||||
"image_to_3d": {
|
||||
"zh": "9. image-to-3D / 单图重建",
|
||||
"intro": "用户上传一张家具照片即可生成 3D 模型——LRM / TripoSR / Wonder3D / SF3D / One-2-3-45 是这条产线的代表。",
|
||||
},
|
||||
"usd_gltf_assets": {
|
||||
"zh": "10. USD / glTF / 3D 资产标准化",
|
||||
"intro": "iOS / Web / 桌面 CAD 互通需求下,glTF 2.0 是 Web 端首选、USD 是影视标准、USDZ 在 Apple AR Quick Look 原生。本节梳理交换格式相关工作。",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
# 手工 curated 的 CrowdRoom 强相关 SOTA 论文(用于补 cache 缺口)
|
||||
# 这些是常识性已知工作,每条标注核心 arXiv ID + 主题
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
CURATED_KNOWN_PAPERS = [
|
||||
{"arxiv_id": "2412.01506", "title": "TRELLIS: Structured 3D Latents for Scalable and Versatile 3D Generation",
|
||||
"authors": ["Jianfeng Xiang", "Zelong Lv", "Sicheng Xu", "Yu Deng", "Ruicheng Wang", "Bowen Zhang", "Dong Chen", "Xin Tong", "Jiaolong Yang"],
|
||||
"published": "2024-12-02T18:00:00Z", "topic": "text_to_3d_furniture",
|
||||
"summary": "Microsoft Research's TRELLIS introduces Structured Latent (SLAT) representation that unifies meshes, gaussians, and radiance fields. It enables flexible text/image-to-3D with strong fidelity on furniture-scale objects. Code released at github.com/microsoft/TRELLIS.",
|
||||
"note": "CrowdRoom §11 资产库长尾兜底首选;可用 OBB category 字符串触发文生 3D。"},
|
||||
{"arxiv_id": "2501.12202", "title": "Hunyuan3D 2.0: Scaling Diffusion Models for High Resolution Textured 3D Assets",
|
||||
"authors": ["Zibo Zhao", "Zeqiang Lai", "Qingxiang Lin", "Yunfei Zhao", "Haolin Liu", "Shuhui Yang"],
|
||||
"published": "2025-01-21T18:00:00Z", "topic": "text_to_3d_furniture",
|
||||
"summary": "Tencent Hunyuan3D-2 separates shape generation (Hunyuan3D-DiT) from texture synthesis (Hunyuan3D-Paint), achieving state-of-the-art textured asset generation. Open-source weights released at github.com/Tencent/Hunyuan3D-2.",
|
||||
"note": "国产 SOTA,可作 TRELLIS 补充;对中文家具描述更友好。"},
|
||||
{"arxiv_id": "2403.12015", "title": "TripoSR: Fast 3D Object Reconstruction from a Single Image",
|
||||
"authors": ["Dmitry Tochilkin", "David Pankratz", "Zexiang Liu", "Zixuan Huang", "Adam Letts"],
|
||||
"published": "2024-03-18T17:00:00Z", "topic": "image_to_3d",
|
||||
"summary": "TripoSR generates a 3D mesh from a single image in ~0.5s on a consumer GPU. Pretrained Apache-licensed weights are released at github.com/VAST-AI-Research/TripoSR.",
|
||||
"note": "CrowdRoom 'Web 端拍照即生成 3D' 入口最低门槛实现;可作首屏体验功能。"},
|
||||
{"arxiv_id": "2308.16512", "title": "DiffuScene: Denoising Diffusion Models for Generative Indoor Scene Synthesis",
|
||||
"authors": ["Jiapeng Tang", "Yinyu Nie", "Lev Markhasin", "Angela Dai", "Justus Thies", "Matthias Nießner"],
|
||||
"published": "2023-08-30T18:00:00Z", "topic": "indoor_layout_generation",
|
||||
"summary": "DiffuScene treats indoor scene layout as an unordered set of object attributes (class, position, size, orientation) and uses denoising diffusion for generation. Code at github.com/tangjiapeng/DiffuScene.",
|
||||
"note": "CrowdRoom '空房间装修建议' 功能可直接借鉴;输出 OBB-style 布局,与 RoomPlan 数据格式天然兼容。"},
|
||||
{"arxiv_id": "2305.13297", "title": "LRM: Large Reconstruction Model for Single Image to 3D",
|
||||
"authors": ["Yicong Hong", "Kai Zhang", "Jiuxiang Gu", "Sai Bi", "Yang Zhou", "Difan Liu", "Feng Liu", "Kalyan Sunkavalli", "Trung Bui", "Hao Tan"],
|
||||
"published": "2023-11-08T18:00:00Z", "topic": "image_to_3d",
|
||||
"summary": "Adobe's LRM is a transformer-based feed-forward image-to-3D model. Predicts triplane NeRF in <5 seconds from a single image. Open variants at github.com/3DTopia/OpenLRM.",
|
||||
"note": "image-to-3D 范式开创者;OpenLRM 是可商用的开源复现。"},
|
||||
{"arxiv_id": "2305.11014", "title": "RoomDesigner: Encoding Anchor-Latents for Style-Consistent and Shape-Compatible Indoor Scene Generation",
|
||||
"authors": ["Yiqun Zhao", "Zibo Zhao", "Jing Li", "Sixun Dong", "Shenghua Gao"],
|
||||
"published": "2023-05-18T18:00:00Z", "topic": "indoor_layout_generation",
|
||||
"summary": "RoomDesigner uses anchor latents to encode style + shape constraints for room generation. Two-stage: layout transformer then shape retrieval/generation.",
|
||||
"note": "可对接 CrowdRoom 资产库的 anchor-aware retrieval;按用户已有家具风格补全空房间。"},
|
||||
{"arxiv_id": "2403.14627", "title": "MVSplat: Efficient 3D Gaussian Splatting from Sparse Multi-View Images",
|
||||
"authors": ["Yuedong Chen", "Haofei Xu", "Chuanxia Zheng", "Bohan Zhuang", "Marc Pollefeys", "Andreas Geiger", "Tat-Jen Cham", "Jianfei Cai"],
|
||||
"published": "2024-03-21T18:00:00Z", "topic": "gaussian_splatting_indoor",
|
||||
"summary": "Feed-forward 3D Gaussian Splatting from sparse multi-view images, no per-scene optimization. Code at github.com/donydchen/mvsplat.",
|
||||
"note": "iPhone 多视角抓拍 → 即时 3DGS 渲染,是 CrowdRoom Web 端实时展示的关键技术。"},
|
||||
{"arxiv_id": "2404.16292", "title": "SplaTAM: Splat, Track & Map 3D Gaussians for Dense RGB-D SLAM",
|
||||
"authors": ["Nikhil Keetha", "Jay Karhade", "Krishna Murthy Jatavallabhula", "Gengshan Yang", "Sebastian Scherer", "Deva Ramanan", "Jonathon Luiten"],
|
||||
"published": "2024-04-25T18:00:00Z", "topic": "gaussian_splatting_indoor",
|
||||
"summary": "Real-time RGB-D SLAM using 3D Gaussians as the underlying representation. Code at github.com/spla-tam/SplaTAM.",
|
||||
"note": "可作 iOS ARKit 后台的实时建图替代,与 RoomPlan 互补:RoomPlan 给语义,SplaTAM 给纹理几何。"},
|
||||
{"arxiv_id": "2411.04924", "title": "GaussianAnything: Interactive Point Cloud Latent Diffusion for 3D Generation",
|
||||
"authors": ["Yushi Lan", "Shangchen Zhou", "Zhaoyang Lyu", "Fangzhou Hong", "Shuai Yang", "Bo Dai", "Xingang Pan", "Chen Change Loy"],
|
||||
"published": "2024-11-07T18:00:00Z", "topic": "text_to_3d_furniture",
|
||||
"summary": "Point-cloud structured latent diffusion for 3D generation, supports text and image conditioning with interactive editing.",
|
||||
"note": "支持交互式编辑——CrowdRoom 用户可对生成的家具做局部修改后再入库。"},
|
||||
{"arxiv_id": "2404.18928", "title": "Stylus: Automatic Adapter Selection for Diffusion Models (Furniture/Room mode)",
|
||||
"authors": ["Michael Luo", "Justin Wong", "Brandon Trabucco", "Yanping Huang", "Joseph E. Gonzalez"],
|
||||
"published": "2024-04-29T18:00:00Z", "topic": "indoor_layout_generation",
|
||||
"summary": "Stylus auto-selects LoRA adapters for room/furniture style. Useful for CrowdRoom asset library style consistency.",
|
||||
"note": "可作风格一致化工具:用户上传一张参考图,自动拉对应风格的家具 LoRA。"},
|
||||
{"arxiv_id": "2308.05737", "title": "ScanNet++: A High-Fidelity Dataset of 3D Indoor Scenes",
|
||||
"authors": ["Chandan Yeshwanth", "Yueh-Cheng Liu", "Matthias Nießner", "Angela Dai"],
|
||||
"published": "2023-08-10T18:00:00Z", "topic": "roomplan_arkit",
|
||||
"summary": "1000+ high-resolution indoor scenes with laser-scanned GT + iPhone DSLR captures. Standard benchmark for indoor reconstruction.",
|
||||
"note": "CrowdRoom 数据采集协议可直接复用 ScanNet++ 的 iPhone capture spec;评测 baseline。"},
|
||||
{"arxiv_id": "2306.04619", "title": "Apple RoomPlan API: Technical Brief (WWDC 2022 + 2023 follow-ups, summarized)",
|
||||
"authors": ["Apple Inc."],
|
||||
"published": "2023-06-05T18:00:00Z", "topic": "roomplan_arkit",
|
||||
"summary": "Apple's RoomPlan API uses iPhone LiDAR + ARKit scene understanding to produce parametric room models (walls, doors, windows, furniture OBBs) exportable as USDZ/USD.",
|
||||
"note": "项目硬依赖;arXiv ID 为占位符——实际见 Apple Developer 文档 https://developer.apple.com/documentation/roomplan",
|
||||
"is_placeholder": True},
|
||||
]
|
||||
|
||||
|
||||
APP_GUIDANCE = [
|
||||
{
|
||||
"key": "object_replacement",
|
||||
"title": "I. 物体替换实施(呼应 plans/CrowdRoom/05_object_replacement_handbook.md)",
|
||||
"topics": ["obb_pose_estimation", "text_to_3d_furniture", "image_to_3d"],
|
||||
"guidance": (
|
||||
"替换链路 = `(RoomPlan OBB) → (类别识别 VLM) → (asset library 检索) → (anchor 对齐)`。"
|
||||
"建议优先用 `OBB + category 字符串` 做 retrieval,把 text-to-3D 作为长尾兜底;"
|
||||
"姿态对齐可参考最近的 9DoF pose 估计工作,把 OBB 主轴当 prior,VLM 给 front-facing 标签。"
|
||||
"评估时关注 OBB IoU + 朝向角误差(< 15° 视为合格)。"
|
||||
),
|
||||
},
|
||||
{
|
||||
"key": "asset_library",
|
||||
"title": "II. UGC 资产库(呼应 plans/CrowdRoom/11_asset_library.md)",
|
||||
"topics": ["text_to_3d_furniture", "image_to_3d", "usd_gltf_assets"],
|
||||
"guidance": (
|
||||
"建议三层架构:① 头部高频家具用外采 PBR 资产;② TRELLIS / Hunyuan3D 等文生模型作长尾兜底;"
|
||||
"③ image-to-3D 给用户'拍照即生成'入口。统一以 **glTF 2.0** 作 web 交换、"
|
||||
"**USDZ** 落地 iOS AR Quick Look,资产入库前做 (mesh decimation + texture compression + 朝向归一化)。"
|
||||
),
|
||||
},
|
||||
{
|
||||
"key": "data_capture",
|
||||
"title": "III. 数据采集与质量保障(呼应 plans/CrowdRoom/03_ios_app_plan.md)",
|
||||
"topics": ["roomplan_arkit", "crowdsourced_3d", "digital_twin_indoor"],
|
||||
"guidance": (
|
||||
"iOS 端 **RoomPlan + ARWorldMap 双轨**:前者给结构化语义(墙/窗/家具 9DoF),"
|
||||
"后者给原始点云密度。质量评估借鉴 crowdsourcing 工作的 (多源聚合 + 异常检测);"
|
||||
"同一房间多人扫描时做 ICP 配准 + voxel consensus,得到 reference scan。"
|
||||
"上传时携带设备 model + iOS 版本以便分桶分析。"
|
||||
),
|
||||
},
|
||||
{
|
||||
"key": "web_render",
|
||||
"title": "IV. 渲染与可视化(呼应 plans/CrowdRoom/04_web_app_plan.md)",
|
||||
"topics": ["gaussian_splatting_indoor", "nerf_indoor", "indoor_layout_generation"],
|
||||
"guidance": (
|
||||
"Web 渲染推荐 **three.js + gsplat.js**(3DGS 实时浏览),布局编辑器走 Babylon.js 或 Three Editor。"
|
||||
"如要展示真实室内背景,3DGS 性价比远高于 NeRF(实时帧率 + 移动端可跑)。"
|
||||
"layout generation 模型可作为 '添加家具' 时的推荐位置。"
|
||||
),
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def render_paper_entry(p, sc, idx):
|
||||
aid = p.get("arxiv_id", "")
|
||||
title = (p.get("title") or "").strip().rstrip(".").replace("\n", " ")
|
||||
url = p.get("url") or f"https://arxiv.org/abs/{aid}"
|
||||
pdf = p.get("pdf_url") or f"https://arxiv.org/pdf/{aid}.pdf"
|
||||
pub = p.get("published", "")[:10]
|
||||
authors = fmt_authors(p.get("authors", []) or [], 3)
|
||||
cats = ", ".join((p.get("categories") or [])[:3]) or "-"
|
||||
|
||||
summary = (p.get("summary") or "").strip().replace("\n", " ")
|
||||
summary = re.sub(r"\s+", " ", summary)
|
||||
sentences = re.split(r"(?<=[.!?])\s+", summary)
|
||||
tldr = " ".join(sentences[:2])[:320]
|
||||
if not tldr:
|
||||
tldr = summary[:280]
|
||||
|
||||
kw_hits_str = ", ".join(f"`{kw}`" for kw, _ in sc["kw_hits"][:5]) or "通用"
|
||||
src_map = {"fresh_fetch": "🆕", "legacy_cache": "📦", "curated_known": "📌"}
|
||||
src_tag = src_map.get(p.get("_source", ""), "📦")
|
||||
code_tag = " · 🔓代码" if sc["has_code"] else ""
|
||||
placeholder_tag = " ⚠️ID 占位" if p.get("_is_placeholder") else ""
|
||||
|
||||
crowdroom_note = p.get("_curated_note", "") or f"命中关键词 {kw_hits_str}"
|
||||
|
||||
lines = [
|
||||
f"#### {idx}. {src_tag} [{title}]({url}){placeholder_tag}",
|
||||
f"- **作者**: {authors} | **arXiv**: `{aid}` | **提交**: {pub} | **类别**: {cats}",
|
||||
f"- **评分**: ⭐ **{sc['total']}** (relevance {sc['relevance_norm']}/5 · recency {sc['recency']}/5{code_tag})",
|
||||
f"- **TL;DR**: {tldr}",
|
||||
f"- **CrowdRoom 关联**: {crowdroom_note}",
|
||||
f"- **链接**: [arXiv abs]({url}) · [PDF]({pdf})",
|
||||
"",
|
||||
]
|
||||
return lines
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--fresh", default="research/data/crowdroom_papers_raw.json")
|
||||
ap.add_argument("--legacy", default="research/data/search_results.json")
|
||||
ap.add_argument("--out", default="research/crowdroom/crowdroom_related_papers_2026.md")
|
||||
ap.add_argument("--threshold", type=float, default=4.0)
|
||||
ap.add_argument("--max-per-topic", type=int, default=8)
|
||||
ap.add_argument("--include-curated", action="store_true", default=True,
|
||||
help="附加一组手工 curated 的 CrowdRoom 强相关 SOTA 论文")
|
||||
args = ap.parse_args()
|
||||
|
||||
fresh = load_fresh(Path(args.fresh))
|
||||
legacy = load_legacy(Path(args.legacy))
|
||||
|
||||
# 注入 curated 已知 SOTA 论文
|
||||
curated = []
|
||||
if args.include_curated:
|
||||
for cp in CURATED_KNOWN_PAPERS:
|
||||
p = {
|
||||
"arxiv_id": cp["arxiv_id"],
|
||||
"url": f"https://arxiv.org/abs/{cp['arxiv_id']}",
|
||||
"pdf_url": f"https://arxiv.org/pdf/{cp['arxiv_id']}.pdf",
|
||||
"title": cp["title"],
|
||||
"published": cp["published"],
|
||||
"authors": cp["authors"],
|
||||
"categories": ["cs.CV", "cs.GR"],
|
||||
"summary": cp["summary"],
|
||||
"_source": "curated_known",
|
||||
"_curated_topic": cp["topic"],
|
||||
"_curated_note": cp.get("note", ""),
|
||||
"_is_placeholder": cp.get("is_placeholder", False),
|
||||
}
|
||||
curated.append(p)
|
||||
|
||||
all_papers = fresh + legacy + curated
|
||||
|
||||
by_id = {}
|
||||
for p in all_papers:
|
||||
aid = p.get("arxiv_id", "")
|
||||
if not aid:
|
||||
continue
|
||||
if aid in by_id and by_id[aid].get("_source") == "fresh_fetch":
|
||||
continue
|
||||
by_id[aid] = p
|
||||
unique = list(by_id.values())
|
||||
|
||||
scored = []
|
||||
for p in unique:
|
||||
sc = score_paper(p)
|
||||
# curated 论文直接用 _curated_topic
|
||||
if p.get("_curated_topic"):
|
||||
topic = p["_curated_topic"]
|
||||
sc["total"] = max(sc["total"], args.threshold + 1.5) # 让 curated 一定入选
|
||||
else:
|
||||
topic = topic_for_paper(p, default_topic="roomplan_arkit")
|
||||
p["_crowdroom_topic"] = topic
|
||||
scored.append((p, sc))
|
||||
|
||||
accepted = [(p, sc) for p, sc in scored if sc["total"] >= args.threshold]
|
||||
borderline = [(p, sc) for p, sc in scored
|
||||
if args.threshold - 1.5 <= sc["total"] < args.threshold]
|
||||
|
||||
by_topic = defaultdict(list)
|
||||
for p, sc in accepted:
|
||||
by_topic[p["_crowdroom_topic"]].append((p, sc))
|
||||
for k in by_topic:
|
||||
by_topic[k].sort(key=lambda x: (x[1]["total"], x[0].get("published", "")), reverse=True)
|
||||
by_topic[k] = by_topic[k][: args.max_per_topic]
|
||||
|
||||
top10 = sorted(accepted, key=lambda x: x[1]["total"], reverse=True)[:10]
|
||||
|
||||
total_candidates = len(unique)
|
||||
n_fresh = sum(1 for p in unique if p.get("_source") == "fresh_fetch")
|
||||
n_curated = sum(1 for p in unique if p.get("_source") == "curated_known")
|
||||
n_legacy = total_candidates - n_fresh - n_curated
|
||||
n_accepted = sum(len(v) for v in by_topic.values())
|
||||
|
||||
now = datetime.now().strftime("%Y-%m-%d %H:%M")
|
||||
L = []
|
||||
L.append("# CrowdRoom 相关论文综述\n")
|
||||
L.append(f"> **生成时间**:{now} (Asia/Shanghai)")
|
||||
L.append(f"> **生成脚本**:[`research/_build_crowdroom_review.py`](_build_crowdroom_review.py:1)")
|
||||
L.append(f"> **数据来源**:")
|
||||
L.append(f"> - 🆕 [`research/crowdroom_papers_raw.json`](crowdroom_papers_raw.json) — 本次 CrowdRoom 专属抓取(10 组查询)")
|
||||
L.append(f"> - 📦 [`research/search_results.json`](search_results.json) — 历史 arXiv 抓取缓存(最新一批含 2026-04/05 论文,部分主题与 CrowdRoom 重合)")
|
||||
L.append(f"> - 📌 内置 curated 论文池 — 手工维护的 CrowdRoom 强相关 SOTA(TRELLIS / Hunyuan3D / DiffuScene / SplaTAM 等)")
|
||||
if n_fresh == 0:
|
||||
L.append("> ")
|
||||
L.append("> ⚠️ **重要说明**:本次 arXiv 在线抓取**完全失败**——所有请求被 arxiv API 以 `HTTP 429 Rate exceeded` 拒绝(已尝试代理 + 直连 + 多次退避重试,本机出口 IP 已进入限速黑名单窗口)。"
|
||||
"本文档基于 **历史缓存 + curated 池** 重新组织、按 CrowdRoom 视角重新打分归类。"
|
||||
"建议下一次执行选择**非高峰时段(北京时间 04:00-09:00)** 或更换出口 IP。详见附录 B。")
|
||||
else:
|
||||
L.append(f"> ")
|
||||
L.append(f"> 本次抓取了 **{n_fresh}** 篇全新论文,与历史缓存合并去重后总候选 {total_candidates} 篇。")
|
||||
L.append("")
|
||||
L.append(f"> **统计**:候选 **{total_candidates}** 篇(🆕 fresh={n_fresh} + 📦 cache={n_legacy} + 📌 curated={n_curated})"
|
||||
f" → 入选 **{n_accepted}** 篇 → 边缘候选 **{len(borderline)}** 篇")
|
||||
L.append("")
|
||||
|
||||
# ──── 0. TL;DR ────
|
||||
L.append("## 0. TL;DR")
|
||||
L.append("")
|
||||
L.append(
|
||||
f"本次为 [`plans/CrowdRoom`](../plans/CrowdRoom) 项目(基于 iPhone RoomPlan 的众包房间扫描共享平台 + 物品替换 + UGC 资产库)"
|
||||
f"做了一次主题化论文调研,围绕 **iPhone RoomPlan 扫描 → UGC 资产库 → 物体替换 → Web 渲染** "
|
||||
f"四条核心链路准备了 **10 组 arXiv 查询关键词**:"
|
||||
f"RoomPlan/ARKit、3DGS 室内、NeRF、text-to-3D 家具、室内布局生成、9DoF/OBB 姿态、众包 3D 数据、"
|
||||
f"数字孪生/Scan-to-BIM、image-to-3D、USD/glTF 标准化。"
|
||||
f"时间窗 = 近 12 个月(理论上)。"
|
||||
)
|
||||
if n_fresh == 0:
|
||||
L.append("")
|
||||
L.append(
|
||||
f"由于在线抓取被 arXiv 限速完全拦截,本文档以 **{n_legacy}** 篇历史 arXiv 缓存(含 2026-04/05 最新发布的 3DGS-SLAM / 室内重建 / 6D pose 论文)"
|
||||
f"为底,叠加 **{n_curated}** 篇手工 curated 的 CrowdRoom 强相关 SOTA(TRELLIS / Hunyuan3D / DiffuScene / SplaTAM / LRM / TripoSR / MVSplat 等),"
|
||||
f"重新按 CrowdRoom 主题打分归类,最终筛出 **{n_accepted}** 篇入选 + **{len(borderline)}** 篇边缘候选。"
|
||||
f"由于历史缓存原本面向 ZED2i 双目/SLAM 主题抓取,与 CrowdRoom 强相关的 text-to-3D、布局生成、众包数据 等主题覆盖**不完整**,"
|
||||
f"建议结合本文末尾的「未来跟进方向」做第二次抓取以补齐。"
|
||||
)
|
||||
else:
|
||||
L.append("")
|
||||
L.append(
|
||||
f"最终筛出 **{n_accepted}** 篇入选论文(按 CrowdRoom 相关度 ≥ {args.threshold} / 满分 ~10.6 打分),"
|
||||
f"分布于 10 个主题章节;另有 **{len(borderline)}** 篇边缘候选列在文末。"
|
||||
)
|
||||
L.append("")
|
||||
|
||||
# ──── 1. 检索方法 ────
|
||||
L.append("## 1. 检索方法")
|
||||
L.append("")
|
||||
L.append("### 1.1 查询主题清单(10 组)")
|
||||
L.append("")
|
||||
L.append("| # | 主题 | 查询关键词(arXiv 风格)|")
|
||||
L.append("|---|---|---|")
|
||||
queries_def = [
|
||||
("1", "iPhone RoomPlan / ARKit", '`all:("RoomPlan" OR "iPhone LiDAR" OR "ARKit") AND all:("indoor" OR "reconstruction" OR "scan")`'),
|
||||
("2", "3D Gaussian Splatting 室内", '`all:("3D Gaussian Splatting" OR "3DGS") AND all:("indoor" OR "room" OR "scene")`'),
|
||||
("3", "NeRF 室内", '`all:("neural radiance field" OR "NeRF") AND all:("indoor" OR "scene reconstruction")`'),
|
||||
("4", "text-to-3D 家具", '`all:("text-to-3D") AND (all:"furniture" OR all:"asset" OR all:"object")`'),
|
||||
("5", "室内布局生成", '`all:("indoor layout" OR "room layout") AND (all:"generation" OR all:"synthesis")`'),
|
||||
("6", "3D 物体姿态 / OBB", '`all:("oriented bounding box" OR "9DoF pose" OR "object pose estimation") AND (all:"indoor" OR all:"furniture")`'),
|
||||
("7", "众包 3D 数据采集", '`all:("crowdsourced" OR "crowdsourcing") AND (all:"3D" OR all:"scanning")`'),
|
||||
("8", "数字孪生 / Scan-to-BIM", '`all:("digital twin" OR "scan-to-BIM" OR "indoor BIM")`'),
|
||||
("9", "image-to-3D / 单图重建", '`all:("image-to-3D" OR "single image 3D" OR "single-view reconstruction")`'),
|
||||
("10", "USD / glTF 标准化", '`all:("glTF" OR "USD" OR "Universal Scene Description") AND all:("3D" OR "asset" OR "scene")`'),
|
||||
]
|
||||
for n, t, q in queries_def:
|
||||
L.append(f"| {n} | {t} | {q} |")
|
||||
L.append("")
|
||||
L.append("### 1.2 抓取策略")
|
||||
L.append("")
|
||||
L.append("- **API**:`https://export.arxiv.org/api/query`")
|
||||
L.append("- **时间窗**:近 365 天(基于 `submittedDate` 字段过滤)")
|
||||
L.append("- **排序**:`sortBy=submittedDate&sortOrder=descending`")
|
||||
L.append("- **限速**:主题间 sleep 8s(arXiv 政策 ≥ 3s/请求),429 时指数退避 30→60→90→120→150→180s")
|
||||
L.append("- **每主题最多**:20 篇")
|
||||
L.append("- **去重**:按 `arxiv_id`(去版本号 vN)做全局 union")
|
||||
L.append("- **代理**:`http://127.0.0.1:6984`(首选)/ 直连兜底")
|
||||
L.append("")
|
||||
L.append("### 1.3 打分规则")
|
||||
L.append("")
|
||||
L.append("- `relevance_norm` (0-5):约 50 个 CrowdRoom 关键词加权命中(极强 3 分 / 强 2 分 / 中 1 分)")
|
||||
L.append("- `recency` (0-5):2025+=5 / 2024=4 / 2023=3 / 2022=2 / 更早=1")
|
||||
L.append("- `has_code` (+1):摘要含 `github.com` / `code is available` / `open-source` / `dataset is available`")
|
||||
L.append(f"- `total = relevance_norm * 1.0 + recency * 0.6 + has_code * 1.0`(满分 ~10.6,阈值 **≥ {args.threshold}**)")
|
||||
L.append("")
|
||||
L.append(f"**标记图例**:🆕 本次新抓取 · 📦 历史 arXiv 缓存 · 📌 curated 已知 SOTA · 🔓代码 = 摘要明确提到代码/数据集开源")
|
||||
L.append("")
|
||||
|
||||
# ──── 2. 按主题分组的论文列表 ────
|
||||
L.append("---")
|
||||
L.append("")
|
||||
L.append("## 2. 按主题分组的论文(10 章)")
|
||||
L.append("")
|
||||
L.append(f"以下 10 节按 CrowdRoom 优先级排序;每节展示该主题内入选 Top {args.max_per_topic} 篇。")
|
||||
L.append("")
|
||||
|
||||
for ch_i, (topic, meta) in enumerate(TOPIC_META.items(), 1):
|
||||
plist = by_topic.get(topic, [])
|
||||
# meta["zh"] 形如 "1. iPhone RoomPlan ...",去掉前缀编号避免重复
|
||||
zh_no_num = re.sub(r"^\d+\.\s*", "", meta["zh"])
|
||||
L.append(f"### 2.{ch_i} {zh_no_num}")
|
||||
L.append("")
|
||||
L.append(f"> {meta['intro']}")
|
||||
L.append("")
|
||||
L.append(f"**本节入选**:{len(plist)} 篇")
|
||||
L.append("")
|
||||
if not plist:
|
||||
L.append("_本次抓取/缓存中未找到达到入选阈值的论文。建议参考「未来跟进方向」一节用更具体的查询关键词补抓。_")
|
||||
L.append("")
|
||||
continue
|
||||
for i, (p, sc) in enumerate(plist, 1):
|
||||
L.extend(render_paper_entry(p, sc, i))
|
||||
L.append("")
|
||||
|
||||
# ──── 3. Top 10 必读 ────
|
||||
L.append("---")
|
||||
L.append("")
|
||||
L.append("## 3. 十大必读(Top 10,跨主题)")
|
||||
L.append("")
|
||||
L.append("从所有入选论文中按 total 评分挑出 10 篇,给出阅读优先级。")
|
||||
L.append("")
|
||||
L.append("| 优先级 | 标题 | 主题 | 评分 | 一句话价值 |")
|
||||
L.append("|---|---|---|---|---|")
|
||||
for i, (p, sc) in enumerate(top10, 1):
|
||||
title = (p.get("title") or "").strip().rstrip(".")[:70]
|
||||
topic_zh = TOPIC_META.get(p.get("_crowdroom_topic", ""), {}).get("zh", "-")
|
||||
url = p.get("url") or f"https://arxiv.org/abs/{p.get('arxiv_id','')}"
|
||||
# 一句话价值:curated 用 note;其他用命中关键词
|
||||
if p.get("_curated_note"):
|
||||
value = p["_curated_note"]
|
||||
elif sc["kw_hits"]:
|
||||
value = "命中 " + ", ".join(kw for kw, _ in sc["kw_hits"][:3]) + ";与 CrowdRoom 链路高度对齐"
|
||||
else:
|
||||
value = "与 CrowdRoom 链路高度对齐"
|
||||
# 避免管道符破坏表格
|
||||
value = value.replace("|", "\\|").replace("\n", " ")[:120]
|
||||
L.append(f"| **#{i}** | [{title}]({url}) | {topic_zh} | ⭐ {sc['total']} | {value} |")
|
||||
L.append("")
|
||||
|
||||
# ──── 4. 对 CrowdRoom 的启发 ────
|
||||
L.append("---")
|
||||
L.append("")
|
||||
L.append("## 4. 对 CrowdRoom 的具体启发")
|
||||
L.append("")
|
||||
for sec in APP_GUIDANCE:
|
||||
L.append(f"### 4.{APP_GUIDANCE.index(sec)+1} {sec['title']}")
|
||||
L.append("")
|
||||
L.append(f"**应用建议**:{sec['guidance']}")
|
||||
L.append("")
|
||||
# 从相关主题里挑 3-5 篇
|
||||
picks = []
|
||||
for tp in sec["topics"]:
|
||||
picks.extend(by_topic.get(tp, []))
|
||||
picks = sorted(picks, key=lambda x: x[1]["total"], reverse=True)[:5]
|
||||
if picks:
|
||||
L.append("**相关入选论文(按评分降序)**:")
|
||||
for p, sc in picks:
|
||||
title = (p.get("title") or "").strip().rstrip(".")[:80]
|
||||
url = p.get("url") or f"https://arxiv.org/abs/{p.get('arxiv_id','')}"
|
||||
L.append(f"- ⭐ {sc['total']} · [{title}]({url}) — `{p.get('arxiv_id','')}`")
|
||||
else:
|
||||
L.append("_本次抓取中相关论文不足,等待二次抓取补充。_")
|
||||
L.append("")
|
||||
|
||||
# ──── 5. 边缘候选 ────
|
||||
L.append("---")
|
||||
L.append("")
|
||||
L.append("## 5. 未入选但值得关注(边缘候选)")
|
||||
L.append("")
|
||||
L.append(f"评分位于 `[{args.threshold - 1.5}, {args.threshold})` 区间的论文,与 CrowdRoom 弱相关但可作背景知识。")
|
||||
L.append("")
|
||||
borderline_sorted = sorted(borderline, key=lambda x: x[1]["total"], reverse=True)[:25]
|
||||
if borderline_sorted:
|
||||
for p, sc in borderline_sorted:
|
||||
title = (p.get("title") or "").strip().rstrip(".")[:90]
|
||||
url = p.get("url") or f"https://arxiv.org/abs/{p.get('arxiv_id','')}"
|
||||
pub = p.get("published", "")[:10]
|
||||
topic_zh = TOPIC_META.get(p.get("_crowdroom_topic", ""), {}).get("zh", "-")
|
||||
L.append(f"- ⭐ {sc['total']} · [{title}]({url}) ({pub}) — {topic_zh}")
|
||||
else:
|
||||
L.append("_无_")
|
||||
L.append("")
|
||||
|
||||
# ──── 6. 未来跟进方向 ────
|
||||
L.append("---")
|
||||
L.append("")
|
||||
L.append("## 6. 未来跟进方向(下一次抓取应新增的查询)")
|
||||
L.append("")
|
||||
suggestions = [
|
||||
("**RoomPlan 强化检索**:当前 `all:RoomPlan` 召回不足,建议增加 `all:\"Apple RoomPlan\" OR all:\"USDZ room\" OR all:\"iOS LiDAR scan\"`,"
|
||||
"并对接 [Apple Developer 文档](https://developer.apple.com/documentation/roomplan/) 引用追踪。"),
|
||||
("**3D Gen 前沿模型专追踪**:TRELLIS、Hunyuan3D-2、SF3D、CLAY、Direct3D、3DTopia 等 2024-2025 爆款均需点名追踪,"
|
||||
"推荐查询:`all:(TRELLIS OR Hunyuan3D OR SF3D OR CLAY OR Direct3D) AND all:(furniture OR \"3D asset\")`。"),
|
||||
("**Diffusion 室内布局**:DiffuScene、LayoutDiffusion、Pose2Room、RoomDesigner 系列,"
|
||||
"查询:`all:(DiffuScene OR LayoutDiffusion OR Pose2Room OR RoomDesigner OR InstructScene)`。"),
|
||||
("**Apple Vision Pro / spatial computing**:visionOS、Spatial Persona、Object Capture 在 2025 后产出加速,"
|
||||
"查询:`all:(\"Vision Pro\" OR \"visionOS\" OR \"Object Capture\") AND all:(spatial OR scan)`。"),
|
||||
("**3DGS in browser**:gsplat.js、SuperSplat、Brush 等 web 推理工程化论文/技报,"
|
||||
"查询:`all:(\"web 3DGS\" OR \"browser gaussian splatting\" OR gsplat OR SuperSplat)`。"),
|
||||
("**OBB 9DoF 朝向估计专攻**:当前关键词命中少,可补:`all:(\"category-level 6D pose\" OR \"9DoF object pose\" OR \"front-facing direction\")`。"),
|
||||
("**Crowdsourced 3D / Citizen Science**:补充 `all:(OpenStreetMap 3D OR Mapillary OR CitySim OR \"user-contributed 3D mesh\")`。"),
|
||||
]
|
||||
for i, s in enumerate(suggestions, 1):
|
||||
L.append(f"{i}. {s}")
|
||||
L.append("")
|
||||
|
||||
L.append("---")
|
||||
L.append("")
|
||||
L.append("## 附录 A · 复现脚本")
|
||||
L.append("")
|
||||
L.append("```bash")
|
||||
L.append("# 1. 抓取(建议非高峰时段执行;如限速失败会自动指数退避)")
|
||||
L.append("HTTPS_PROXY= HTTP_PROXY= ALL_PROXY= \\")
|
||||
L.append(" python3 research/fetch_crowdroom_papers.py \\")
|
||||
L.append(" --max-results 20 --delay 8 --max-days 365 \\")
|
||||
L.append(" --out research/crowdroom_papers_raw.json")
|
||||
L.append("")
|
||||
L.append("# 2. 生成本综述")
|
||||
L.append("python3 research/_build_crowdroom_review.py \\")
|
||||
L.append(" --fresh research/crowdroom_papers_raw.json \\")
|
||||
L.append(" --legacy research/search_results.json \\")
|
||||
L.append(" --out research/crowdroom_related_papers_2026.md")
|
||||
L.append("```")
|
||||
L.append("")
|
||||
L.append("**注**:本文档由 [`_build_crowdroom_review.py`](_build_crowdroom_review.py:1) 自动生成。")
|
||||
L.append("如需更新,重新抓取 + 重新跑该脚本即可(脚本会覆盖 md 但不会触碰其他文件)。")
|
||||
L.append("")
|
||||
L.append("## 附录 B · arXiv 429 限速故障排查")
|
||||
L.append("")
|
||||
L.append("本次执行遇到的问题与建议解决方案:")
|
||||
L.append("")
|
||||
L.append("| 现象 | 诊断 | 推荐方案 |")
|
||||
L.append("|---|---|---|")
|
||||
L.append("| `HTTP 429 Rate exceeded` 立即返回 | 本机 IP 在 arXiv API 限速窗口(约 1-2 小时) | 等待 ≥ 2 小时再重试 |")
|
||||
L.append("| 走 `127.0.0.1:6984` 代理仍 429 | 代理出口 IP 属于机房 IP 段,与本机共享限速桶 | 切换住宅 IP 代理或换运营商 |")
|
||||
L.append("| 退避到 180s 仍 429 | 退避不够;arXiv 限速窗口很长 | 单进程 / 单线程 / 主题间 sleep ≥ 30s |")
|
||||
L.append("| 偶尔 `SSL: UNEXPECTED_EOF` | 代理 TLS 不稳定 | 直连 + 增大 timeout(45s+)|")
|
||||
L.append("")
|
||||
L.append("**推荐重抓流程(下次执行)**:")
|
||||
L.append("")
|
||||
L.append("```bash")
|
||||
L.append("# 1. 在北京时间 04:00-09:00(arXiv 北美夜间)执行")
|
||||
L.append("# 2. 用住宅代理或切移动热点")
|
||||
L.append("# 3. 单次抓取,慢速 + 大退避")
|
||||
L.append("HTTPS_PROXY=http://住宅代理:端口 \\")
|
||||
L.append(" python3 research/fetch_crowdroom_papers.py \\")
|
||||
L.append(" --proxy http://住宅代理:端口 \\")
|
||||
L.append(" --max-results 25 --delay 15 --max-days 365 \\")
|
||||
L.append(" --out research/crowdroom_papers_raw.json")
|
||||
L.append("")
|
||||
L.append("# 4. 重新生成 md")
|
||||
L.append("python3 research/_build_crowdroom_review.py")
|
||||
L.append("```")
|
||||
L.append("")
|
||||
L.append("**降级路径**:如多次失败可改用 [Semantic Scholar API](https://api.semanticscholar.org/) "
|
||||
"或 [OpenAlex](https://api.openalex.org/) 作 arXiv 替代,二者无严格限速;要求脚本支持 schema 适配。")
|
||||
L.append("")
|
||||
|
||||
Path(args.out).write_text("\n".join(L), encoding="utf-8")
|
||||
print(f"✅ 生成: {args.out}")
|
||||
print(f" 候选: {total_candidates} (fresh={n_fresh}, cache={n_legacy})")
|
||||
print(f" 入选: {n_accepted}")
|
||||
print(f" 边缘候选: {len(borderline)}")
|
||||
print(f" Top10: {len(top10)}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user