chore: initial commit — import worldmodel workspace (plans/, research/)

This commit is contained in:
gaojie
2026-05-20 21:43:57 +08:00
commit bec8a9a4a3
98 changed files with 44128 additions and 0 deletions
+86
View File
@@ -0,0 +1,86 @@
#!/usr/bin/env python3
"""
fetch_arxiv.py — 从 arXiv 原生 API 抓取每个项目的 title / abstract / year。
arXiv 没有 rate limit 烦恼;Semantic Scholar 429 改这里。
"""
from __future__ import annotations
import json
import pathlib
import re
import sys
import time
import urllib.request
import xml.etree.ElementTree as ET
PROJECTS_JSON = pathlib.Path(__file__).parent / "projects.json"
OUT = pathlib.Path(__file__).parent / "arxiv_data.json"
NS = {"a": "http://www.w3.org/2005/Atom",
"arxiv": "http://arxiv.org/schemas/atom"}
UA = "Mozilla/5.0 (research) PRISM-comparison/1.0"
def get(url: str, timeout: int = 30) -> str:
req = urllib.request.Request(url, headers={"User-Agent": UA})
with urllib.request.urlopen(req, timeout=timeout) as r:
return r.read().decode("utf-8", errors="replace")
def fetch_one(arxiv_id: str) -> dict:
"""arXiv id 形如 2309.16650 (可以带 v1 后缀,但去掉更稳)"""
aid = re.sub(r"v\d+$", "", arxiv_id)
url = f"http://export.arxiv.org/api/query?id_list={aid}"
try:
xml = get(url)
except Exception as e:
return {"_error": f"http: {e}"}
try:
root = ET.fromstring(xml)
entry = root.find("a:entry", NS)
if entry is None:
return {"_error": "no entry"}
title = (entry.findtext("a:title", default="", namespaces=NS) or "").strip()
summary = (entry.findtext("a:summary", default="", namespaces=NS) or "").strip()
published = (entry.findtext("a:published", default="", namespaces=NS) or "").strip()
updated = (entry.findtext("a:updated", default="", namespaces=NS) or "").strip()
authors = [
(a.findtext("a:name", default="", namespaces=NS) or "").strip()
for a in entry.findall("a:author", NS)
]
# primary category
prim = entry.find("arxiv:primary_category", NS)
primary = prim.get("term") if prim is not None else None
# journal-ref (= venue if author filled)
jref = entry.findtext("arxiv:journal_ref", default="", namespaces=NS)
# doi
doi = entry.findtext("arxiv:doi", default="", namespaces=NS)
return {
"title": re.sub(r"\s+", " ", title),
"summary": re.sub(r"\s+", " ", summary),
"published": published,
"updated": updated,
"authors": authors,
"primary_cat": primary,
"journal_ref": jref,
"doi": doi,
}
except Exception as e:
return {"_error": f"xml: {e}"}
def main() -> None:
data = json.loads(PROJECTS_JSON.read_text(encoding="utf-8"))
out = {}
for key, v in data.items():
aid = v["meta"]["arxiv"]
sys.stderr.write(f"[arxiv] {v['meta']['name']} ({aid}) ...\n")
out[key] = fetch_one(aid)
time.sleep(3.5) # arXiv 建议 ≥ 3s
OUT.write_text(json.dumps(out, ensure_ascii=False, indent=2),
encoding="utf-8")
sys.stderr.write(f"[ok] {OUT}\n")
if __name__ == "__main__":
main()