File "fetch_testfellow.py"

Full path: /home/algopkco/public_html/scraper/fetch_testfellow.py
File size: 7.28 B (7.28 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8

Download   Open   Edit   Advanced Editor &nnbsp; Back

import json
import os
import re
import time

import requests

BASE = os.path.dirname(os.path.abspath(__file__))
DATA = os.path.join(BASE, "data")
OUT = os.path.join(DATA, "testfellow_mcqs.jsonl")
PROG = os.path.join(BASE, "testfellow_progress.txt")
LOG = os.path.join(BASE, "testfellow_run.log")

H = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/126.0 Safari/537.36"}
s = requests.Session()
s.headers.update(H)

INDEXES = [
    "https://testfellow.com/mcqs/",
    "https://testfellow.com/biology-mcqs/",
    "https://testfellow.com/biology-practice-tests/",
    "https://testfellow.com/chemistry-mcqs/",
    "https://testfellow.com/chemistry-mcqs-for-lecturer-test/",
    "https://testfellow.com/physics-mcqs-for-lecturer-test/",
    "https://testfellow.com/zoology-mcqs-for-lecturer-test/",
    "https://testfellow.com/english-literature-mcqs-for-lecturer-test/",
    "https://testfellow.com/ppsc-computer-science-lecturer-solved-past-paper/",
    "https://testfellow.com/computer-mcqs-online-test-quiz/",
    "https://testfellow.com/english-mcqs-online-test/",
    "https://testfellow.com/everyday-science-mcqs/",
    "https://testfellow.com/general-knowledge-mcqs-gk/",
    "https://testfellow.com/current-affairs-mcqs/",
    "https://testfellow.com/world-geography-gk-mcqs/",
    "https://testfellow.com/world-history-mcqs-online-test/",
]

BIOLOGY = re.compile(
    r"(aids|animal|anatomy|biology|biomol|biotech|biodiv|bioinform|cell|disease|ecology|evolution|food|genetics|"
    r"health|histology|human|immun|infectious|liver|living|lysosome|microbe|microbiol|molecular|morphology|neural|"
    r"nitrogen|nutrition|organism|pcr|photosynthesis|phylum|plant|protein|reproduction|respiratory|structural|"
    r"transport|vermicomposting|virus|vitamin|antibiotic|breathing|blood|carbohydrate|chromosome|"
    r"digestion|respiratory|mineral-physiology)",
    re.I,
)
CHEM = re.compile(r"(chem|chlorofluorocarbon|gas|rocket|solution|acid|periodic|atom|chemical)", re.I)
PHYS = re.compile(r"(physic|electric|magnet|light|sound|force|energy|motion|wave|nuclear|optics)", re.I)
COMP = re.compile(r"(computer|software|hardware|internet|programming|dbms|ms-|office)", re.I)
ENG = re.compile(r"(english|grammar|vocabulary|synonym|antonym)", re.I)
GEO = re.compile(r"(geograph|latitude|continent|ocean|river|mountain|country-capital)", re.I)
HIST = re.compile(r"(history|ancient|medieval|freedom|independence|battle)", re.I)
CA = re.compile(r"(current-affairs|appointment|award|scheme|budget)", re.I)
GK = re.compile(r"(gk|general-knowledge|currency|flags|capital|constitution|sports|national)", re.I)
MATH = re.compile(r"(math|algebra|geometry|trigono|arithmetic|percentage|ratio|profit|probability|aptitude)", re.I)
SCI = re.compile(r"(science|pollution|waste|environment)", re.I)

NO_FETCH = re.compile(r"(practice-test|online-test|quiz$|practice-tests)")


def classify(slug):
    if BIOLOGY.search(slug):
        return "Biology General"
    if CHEM.search(slug):
        return "Chemistry General"
    if PHYS.search(slug):
        return "Physics General"
    if COMP.search(slug):
        return "Computer Science General Test"
    if ENG.search(slug):
        return "English General"
    if GEO.search(slug):
        return "Geography General"
    if HIST.search(slug):
        return "History General"
    if CA.search(slug):
        return "Current Affairs General"
    if GK.search(slug):
        return "General Knowledge"
    if MATH.search(slug):
        return "Mathematics General"
    if SCI.search(slug):
        return "Everyday Science General"
    return "Everyday Science General"


def get(url, tries=4):
    for i in range(tries):
        try:
            r = s.get(url, timeout=30)
            if r.status_code == 200:
                return r.text
        except Exception:
            pass
        time.sleep(1 + i)
    return None


def parse_page(html):
    body = re.sub(r"<script.*?</script>|<style.*?</style>", "", html, flags=re.S)
    body = body.replace("&#8211;", "-").replace("&nbsp;", " ").replace("&#038;", "&").replace("&amp;", "&")
    plain = re.sub(r"<[^>]+>", "\n", body)
    rows = []
    cur = None
    LQ = re.compile(r"^(\d{1,3})\.\s+(.+)$")
    LO = re.compile(r"^([a-eA-E])[\.\)]\s+(.+)$")
    LA = re.compile(r"^Answer\s*:\s*([a-eA-E])[\.\)]\s*(.+)$")
    for ln in plain.split("\n"):
        ln = ln.strip()
        if not ln:
            continue
        m = LA.match(ln)
        if m:
            if cur:
                cur["ans_letter"] = m.group(1).lower()
                cur["ans_text"] = m.group(2).strip()
            continue
        m = LQ.match(ln)
        if m:
            q = m.group(2).strip()
            if 8 < len(q) < 500:
                cur = {"q": q, "opts": [], "ans_letter": None, "ans_text": None}
                rows.append(cur)
            else:
                cur = None
            continue
        m = LO.match(ln)
        if m and cur is not None:
            opt = m.group(2).strip()
            if 1 < len(opt) < 400:
                cur["opts"].append(opt)
    out = []
    for r in rows:
        if not r["opts"] or not r["ans_letter"]:
            continue
        if len(r["opts"]) < 2:
            continue
        ci = ord(r["ans_letter"]) - ord("a")
        if ci >= len(r["opts"]):
            continue
        out.append((r["q"], r["opts"], ci))
    return out


def main():
    done = set()
    if os.path.exists(PROG):
        done = set(l.strip() for l in open(PROG, encoding="utf-8") if l.strip())
    seen = set()
    if os.path.exists(OUT):
        for ln in open(OUT, encoding="utf-8"):
            try:
                seen.add(json.loads(ln)["question"])
            except Exception:
                pass
    fout = open(OUT, "a", encoding="utf-8")
    flog = open(LOG, "a", encoding="utf-8")
    new = 0
    for ix in INDEXES:
        html = get(ix)
        if not html:
            flog.write(f"{ix}: NO HTML\n")
            flog.flush()
            continue
        links = sorted(set(re.findall(r'href="(https://testfellow\.com/[a-z0-9\-]+/)"', html)))
        links = [u for u in links if not NO_FETCH.search(u)]
        for url in links:
            if url in done or url + ".done" in done:
                continue
            h = get(url)
            if not h:
                flog.write(f"{url}: NO HTML\n")
                flog.flush()
                continue
            rows = parse_page(h)
            slug = url.rstrip("/").split("/")[-1]
            subj = classify(slug)
            topic = "TestFellow " + slug.replace("-", " ").title()
            cnt = 0
            for q, opts, ci in rows:
                if q in seen:
                    continue
                seen.add(q)
                fout.write(json.dumps({
                    "subject": subj, "topic": topic, "question": q,
                    "options": opts, "correct": ci + 1, "explanation": "", "source_url": url,
                }, ensure_ascii=False) + "\n")
                cnt += 1
                new += 1
            fout.flush()
            with open(PROG, "a", encoding="utf-8") as f:
                f.write(url + "\n")
            print(f"{url}: {len(rows)} parsed, {cnt} new (total {new})", flush=True)
            flog.write(f"{url}: {len(rows)} parsed, {cnt} new\n")
            flog.flush()
            time.sleep(0.5)
    fout.close()
    flog.close()
    print("DONE total new:", new)


if __name__ == "__main__":
    main()