File "fetch_netex.py"

Full path: /home/algopkco/public_html/scraper/fetch_netex.py
File size: 5.43 B (5.43 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8

Download   Open   Edit   Advanced Editor &nnbsp; Back

import json
import os
import re
import time

import requests

BASE = os.path.dirname(os.path.abspath(__file__))
DATA = os.path.join(BASE, "data")
OUT = os.path.join(DATA, "netex_mcqs.jsonl")
PROG = os.path.join(BASE, "netex_progress.txt")

H = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/126.0 Safari/537.36"}
s = requests.Session()
s.headers.update(H)

LQ = re.compile(r"^\((\d{1,3})\)\s*(.{10,})$")
LA = re.compile(r"^Ans\.?\s*\(([a-eA-E])\)\s*")
LO = re.compile(r"\(([a-eA-E])\)([^()]{2,300}?)(?=\([a-eA-E]\)|$)")


def classify(title):
    t = title.lower()
    if re.search(r"math|algebra|geometry|trigono", t):
        return "Mathematics General"
    if re.search(r"physics|electricity|motion|force|light", t):
        return "Physics General"
    if re.search(r"chemistry|compounds|elements|acid|matter", t):
        return "Chemistry General"
    if re.search(r"biolog|life process|cell|plant|animal|human", t):
        return "Biology General"
    if re.search(r"computer|it |program|software|hardware", t):
        return "Computer Science General Test"
    if re.search(r"english|grammar", t):
        return "General English Grammar"
    if re.search(r"history|civics|polity", t):
        return "History General"
    if re.search(r"geograph", t):
        return "Geography General"
    if re.search(r"science", t):
        return "Everyday Science General"
    return "General Knowledge"


def parse_content(content):
    plain = re.sub(r"<[^>]+>", "\n", content)
    rows = []
    cur = None
    for ln in plain.split("\n"):
        ln = ln.strip()
        if not ln:
            continue
        m = LQ.match(ln)
        if m:
            cur = {"q": m.group(2).strip(), "opts": [], "ans": None}
            rows.append(cur)
            continue
        m = LA.match(ln)
        if m:
            if cur:
                cur["ans"] = m.group(1).lower()
            continue
        if cur is not None:
            for om in LO.finditer(ln):
                cur["opts"].append(om.group(2).strip())
    out = []
    for r in rows:
        if not r["opts"] or not r["ans"]:
            continue
        ci = ord(r["ans"]) - ord("a")
        if ci >= len(r["opts"]):
            continue
        if len(r["q"]) < 8:
            continue
        out.append((r["q"], r["opts"][:6], ci))
    return out


def main():
    done = set()
    if os.path.exists(PROG):
        done = set(l.strip() for l in open(PROG, encoding="utf-8") if l.strip())
    seen = set()
    if os.path.exists(OUT):
        for ln in open(OUT, encoding="utf-8"):
            try:
                seen.add(json.loads(ln)["question"])
            except Exception:
                pass
    fout = open(OUT, "a", encoding="utf-8")
    flog = open(os.path.join(BASE, "netex_run.log"), "a", encoding="utf-8")
    new = 0
    total_pages = None
    page = 1
    while True:
        try:
            r = s.get("https://www.netexplanations.com/wp-json/wp/v2/posts",
                      params={"search": "mcq", "per_page": 100, "page": page}, timeout=60)
            if r.status_code != 200:
                flog.write(f"page {page}: HTTP {r.status_code}\n")
                flog.flush()
                break
            posts = r.json()
        except Exception as e:
            flog.write(f"page {page}: ERR {e}\n")
            flog.flush()
            time.sleep(5)
            page += 1
            continue
        if total_pages is None:
            total_pages = int(r.headers.get("X-WP-TotalPages", "1"))
            flog.write(f"total posts {r.headers.get('X-WP-Total')} pages {total_pages}\n")
            flog.flush()
        flog.write(f"page {page} processing {len(posts)} posts (new total {new})\n")
        flog.flush()
        posts = r.json()
        for p in posts:
            try:
                pid = str(p.get("id"))
                if pid in done:
                    continue
                title = p.get("title", {}).get("rendered", "")
                if re.search(r"[\u0900-\u09FF]", title):
                    continue
                content = p.get("content", {}).get("rendered", "")
                qs = parse_content(content)
                if len(qs) < 3:
                    continue
                subj = classify(title)
                topic = ("NetExplanations " + title[:80]).strip()
                cnt = 0
                for q, opts, ci in qs:
                    if q in seen:
                        continue
                    seen.add(q)
                    fout.write(json.dumps({
                        "subject": subj, "topic": topic, "question": q,
                        "options": opts, "correct": ci + 1, "explanation": "",
                        "source_url": p.get("link", ""),
                    }, ensure_ascii=False) + "\n")
                    cnt += 1
                    new += 1
                fout.flush()
                with open(PROG, "a", encoding="utf-8") as f:
                    f.write(pid + "\n")
                if cnt:
                    print(f"p{page} #{pid} {title[:70]}: {len(qs)} q, {cnt} new (total {new})", flush=True)
                flog.write(f"#{pid} {title[:70]}: {len(qs)} q, {cnt} new\n")
                flog.flush()
            except Exception as e:
                flog.write(f"#{p.get('id')} POST ERR {e}\n")
                flog.flush()
                continue
        page += 1
        if page > total_pages:
            break
        time.sleep(0.3)
    fout.close()
    flog.close()
    print("DONE total new:", new)


if __name__ == "__main__":
    main()