File "refetch_explanations.py"

Full path: /home/algopkco/public_html/scraper/refetch_explanations.py
File size: 3.9 B (3.9 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8

Download   Open   Edit   Advanced Editor &nnbsp; Back

import json
import os
import re
import time

import requests

BASE = os.path.dirname(os.path.abspath(__file__))
DATA = os.path.join(BASE, "data")
FILE = os.path.join(DATA, "examveda_mcqs.jsonl")
EXPL_FILE = os.path.join(DATA, "examveda_explanations.jsonl")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36"
}

session = requests.Session()
session.headers.update(HEADERS)


def get(url, tries=5):
    for i in range(tries):
        try:
            r = session.get(url, timeout=30)
            if r.status_code == 200:
                return r.text
            print("  HTTP", r.status_code, url, flush=True)
            if r.status_code == 404:
                return None
        except Exception as e:
            print("  err", e, "retry", i, flush=True)
        time.sleep(1 + i * 2)
    return None


def clean_html(s):
    s = re.sub(r"<br\s*/?>", "\n", s)
    s = re.sub(r"<[^>]+>", "", s)
    s = re.sub(r"\xa0", " ", s)
    s = re.sub(r"[ \t]+", " ", s)
    s = re.sub(r"\n\s*\n+", "\n", s)
    return s.strip()


def split_articles(html):
    idxs = [m.start() for m in re.finditer(r"<article", html)]
    out = []
    for i, s in enumerate(idxs):
        e = idxs[i + 1] if i + 1 < len(idxs) else len(html)
        out.append(html[s:e])
    return out


def extract_solutions(html):
    sols = {}
    for art in split_articles(html):
        qm = re.search(r'<div class="question-main">(.*?)</div>', art, re.S)
        if not qm:
            continue
        question = clean_html(qm.group(1))
        if not question:
            continue
        m = re.search(
            r'<span class="color">Solution:\s*</span></div>\s*(.*?)</div>',
            art,
            re.S,
        )
        if m:
            expl = clean_html(m.group(1))
            expl = re.sub(r"^Solution:\s*", "", expl)
            if expl:
                sols[question] = expl[:2000]
    return sols


def main():
    rows = []
    with open(FILE, "r", encoding="utf-8") as f:
        for ln in f:
            try:
                rows.append(json.loads(ln))
            except Exception:
                continue
    print("rows:", len(rows), flush=True)

    done = set()
    if os.path.exists(EXPL_FILE):
        with open(EXPL_FILE, encoding="utf-8") as f:
            for ln in f:
                try:
                    done.add(json.loads(ln)["page"])
                except Exception:
                    pass
    print("pages done:", len(done), flush=True)

    pages = {}
    for it in rows:
        subj = it["subject"].strip("-")
        topic = it["topic"].lower().replace(" ", "-")
        topic = re.sub(r"[^a-z0-9-]", "", topic)
        url = f"https://www.examveda.com/{subj}/{topic}/"
        pages.setdefault(url, []).append(it["question"])
    print("unique pages:", len(pages), flush=True)

    fetched = 0
    with open(EXPL_FILE, "a", encoding="utf-8") as out:
        for url, qs in pages.items():
            if url in done:
                continue
            total = {}
            for pg in range(1, 100):
                cur = url if pg == 1 else url + f"?page={pg}"
                html = get(cur)
                if html is None:
                    break
                sols = extract_solutions(html)
                total.update(sols)
                # stop when page contains none of our questions
                page_qs = [q for q in qs if q in sols]
                if pg > 1 and not page_qs and len(sols) < 10:
                    break
                time.sleep(0.3)
            for q, e in total.items():
                out.write(json.dumps({"page": url, "question": q, "explanation": e}, ensure_ascii=False) + "\n")
            out.flush()
            fetched += 1
            if fetched % 50 == 0:
                print("pages fetched:", fetched, "/", len(pages), flush=True)
    print("DONE fetched pages:", fetched, flush=True)


if __name__ == "__main__":
    main()