File "fetch_netex.py"
Full path: /home/algopkco/public_html/scraper/fetch_netex.py
File
size: 5.43 B (5.43 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8
Download Open Edit Advanced Editor &nnbsp; Back
import json
import os
import re
import time
import requests
BASE = os.path.dirname(os.path.abspath(__file__))
DATA = os.path.join(BASE, "data")
OUT = os.path.join(DATA, "netex_mcqs.jsonl")
PROG = os.path.join(BASE, "netex_progress.txt")
H = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/126.0 Safari/537.36"}
s = requests.Session()
s.headers.update(H)
LQ = re.compile(r"^\((\d{1,3})\)\s*(.{10,})$")
LA = re.compile(r"^Ans\.?\s*\(([a-eA-E])\)\s*")
LO = re.compile(r"\(([a-eA-E])\)([^()]{2,300}?)(?=\([a-eA-E]\)|$)")
def classify(title):
t = title.lower()
if re.search(r"math|algebra|geometry|trigono", t):
return "Mathematics General"
if re.search(r"physics|electricity|motion|force|light", t):
return "Physics General"
if re.search(r"chemistry|compounds|elements|acid|matter", t):
return "Chemistry General"
if re.search(r"biolog|life process|cell|plant|animal|human", t):
return "Biology General"
if re.search(r"computer|it |program|software|hardware", t):
return "Computer Science General Test"
if re.search(r"english|grammar", t):
return "General English Grammar"
if re.search(r"history|civics|polity", t):
return "History General"
if re.search(r"geograph", t):
return "Geography General"
if re.search(r"science", t):
return "Everyday Science General"
return "General Knowledge"
def parse_content(content):
plain = re.sub(r"<[^>]+>", "\n", content)
rows = []
cur = None
for ln in plain.split("\n"):
ln = ln.strip()
if not ln:
continue
m = LQ.match(ln)
if m:
cur = {"q": m.group(2).strip(), "opts": [], "ans": None}
rows.append(cur)
continue
m = LA.match(ln)
if m:
if cur:
cur["ans"] = m.group(1).lower()
continue
if cur is not None:
for om in LO.finditer(ln):
cur["opts"].append(om.group(2).strip())
out = []
for r in rows:
if not r["opts"] or not r["ans"]:
continue
ci = ord(r["ans"]) - ord("a")
if ci >= len(r["opts"]):
continue
if len(r["q"]) < 8:
continue
out.append((r["q"], r["opts"][:6], ci))
return out
def main():
done = set()
if os.path.exists(PROG):
done = set(l.strip() for l in open(PROG, encoding="utf-8") if l.strip())
seen = set()
if os.path.exists(OUT):
for ln in open(OUT, encoding="utf-8"):
try:
seen.add(json.loads(ln)["question"])
except Exception:
pass
fout = open(OUT, "a", encoding="utf-8")
flog = open(os.path.join(BASE, "netex_run.log"), "a", encoding="utf-8")
new = 0
total_pages = None
page = 1
while True:
try:
r = s.get("https://www.netexplanations.com/wp-json/wp/v2/posts",
params={"search": "mcq", "per_page": 100, "page": page}, timeout=60)
if r.status_code != 200:
flog.write(f"page {page}: HTTP {r.status_code}\n")
flog.flush()
break
posts = r.json()
except Exception as e:
flog.write(f"page {page}: ERR {e}\n")
flog.flush()
time.sleep(5)
page += 1
continue
if total_pages is None:
total_pages = int(r.headers.get("X-WP-TotalPages", "1"))
flog.write(f"total posts {r.headers.get('X-WP-Total')} pages {total_pages}\n")
flog.flush()
flog.write(f"page {page} processing {len(posts)} posts (new total {new})\n")
flog.flush()
posts = r.json()
for p in posts:
try:
pid = str(p.get("id"))
if pid in done:
continue
title = p.get("title", {}).get("rendered", "")
if re.search(r"[\u0900-\u09FF]", title):
continue
content = p.get("content", {}).get("rendered", "")
qs = parse_content(content)
if len(qs) < 3:
continue
subj = classify(title)
topic = ("NetExplanations " + title[:80]).strip()
cnt = 0
for q, opts, ci in qs:
if q in seen:
continue
seen.add(q)
fout.write(json.dumps({
"subject": subj, "topic": topic, "question": q,
"options": opts, "correct": ci + 1, "explanation": "",
"source_url": p.get("link", ""),
}, ensure_ascii=False) + "\n")
cnt += 1
new += 1
fout.flush()
with open(PROG, "a", encoding="utf-8") as f:
f.write(pid + "\n")
if cnt:
print(f"p{page} #{pid} {title[:70]}: {len(qs)} q, {cnt} new (total {new})", flush=True)
flog.write(f"#{pid} {title[:70]}: {len(qs)} q, {cnt} new\n")
flog.flush()
except Exception as e:
flog.write(f"#{p.get('id')} POST ERR {e}\n")
flog.flush()
continue
page += 1
if page > total_pages:
break
time.sleep(0.3)
fout.close()
flog.close()
print("DONE total new:", new)
if __name__ == "__main__":
main()