Create New Item
×
Item Type
File
Folder
Item Name
×
Search file in folder and subfolders...
File Manager
/
scraper
Advanced Search
Upload
New Item
Settings
Back
Back Up
Advanced Editor
Save
import json import os import re import time import requests BASE = os.path.dirname(os.path.abspath(__file__)) DATA = os.path.join(BASE, "data") OUT = os.path.join(DATA, "testfellow_mcqs.jsonl") PROG = os.path.join(BASE, "testfellow_progress.txt") LOG = os.path.join(BASE, "testfellow_run.log") H = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/126.0 Safari/537.36"} s = requests.Session() s.headers.update(H) INDEXES = [ "https://testfellow.com/mcqs/", "https://testfellow.com/biology-mcqs/", "https://testfellow.com/biology-practice-tests/", "https://testfellow.com/chemistry-mcqs/", "https://testfellow.com/chemistry-mcqs-for-lecturer-test/", "https://testfellow.com/physics-mcqs-for-lecturer-test/", "https://testfellow.com/zoology-mcqs-for-lecturer-test/", "https://testfellow.com/english-literature-mcqs-for-lecturer-test/", "https://testfellow.com/ppsc-computer-science-lecturer-solved-past-paper/", "https://testfellow.com/computer-mcqs-online-test-quiz/", "https://testfellow.com/english-mcqs-online-test/", "https://testfellow.com/everyday-science-mcqs/", "https://testfellow.com/general-knowledge-mcqs-gk/", "https://testfellow.com/current-affairs-mcqs/", "https://testfellow.com/world-geography-gk-mcqs/", "https://testfellow.com/world-history-mcqs-online-test/", ] BIOLOGY = re.compile( r"(aids|animal|anatomy|biology|biomol|biotech|biodiv|bioinform|cell|disease|ecology|evolution|food|genetics|" r"health|histology|human|immun|infectious|liver|living|lysosome|microbe|microbiol|molecular|morphology|neural|" r"nitrogen|nutrition|organism|pcr|photosynthesis|phylum|plant|protein|reproduction|respiratory|structural|" r"transport|vermicomposting|virus|vitamin|antibiotic|breathing|blood|carbohydrate|chromosome|" r"digestion|respiratory|mineral-physiology)", re.I, ) CHEM = re.compile(r"(chem|chlorofluorocarbon|gas|rocket|solution|acid|periodic|atom|chemical)", re.I) PHYS = re.compile(r"(physic|electric|magnet|light|sound|force|energy|motion|wave|nuclear|optics)", re.I) COMP = re.compile(r"(computer|software|hardware|internet|programming|dbms|ms-|office)", re.I) ENG = re.compile(r"(english|grammar|vocabulary|synonym|antonym)", re.I) GEO = re.compile(r"(geograph|latitude|continent|ocean|river|mountain|country-capital)", re.I) HIST = re.compile(r"(history|ancient|medieval|freedom|independence|battle)", re.I) CA = re.compile(r"(current-affairs|appointment|award|scheme|budget)", re.I) GK = re.compile(r"(gk|general-knowledge|currency|flags|capital|constitution|sports|national)", re.I) MATH = re.compile(r"(math|algebra|geometry|trigono|arithmetic|percentage|ratio|profit|probability|aptitude)", re.I) SCI = re.compile(r"(science|pollution|waste|environment)", re.I) NO_FETCH = re.compile(r"(practice-test|online-test|quiz$|practice-tests)") def classify(slug): if BIOLOGY.search(slug): return "Biology General" if CHEM.search(slug): return "Chemistry General" if PHYS.search(slug): return "Physics General" if COMP.search(slug): return "Computer Science General Test" if ENG.search(slug): return "English General" if GEO.search(slug): return "Geography General" if HIST.search(slug): return "History General" if CA.search(slug): return "Current Affairs General" if GK.search(slug): return "General Knowledge" if MATH.search(slug): return "Mathematics General" if SCI.search(slug): return "Everyday Science General" return "Everyday Science General" def get(url, tries=4): for i in range(tries): try: r = s.get(url, timeout=30) if r.status_code == 200: return r.text except Exception: pass time.sleep(1 + i) return None def parse_page(html): body = re.sub(r"<script.*?</script>|<style.*?</style>", "", html, flags=re.S) body = body.replace("–", "-").replace(" ", " ").replace("&", "&").replace("&", "&") plain = re.sub(r"<[^>]+>", "\n", body) rows = [] cur = None LQ = re.compile(r"^(\d{1,3})\.\s+(.+)$") LO = re.compile(r"^([a-eA-E])[\.\)]\s+(.+)$") LA = re.compile(r"^Answer\s*:\s*([a-eA-E])[\.\)]\s*(.+)$") for ln in plain.split("\n"): ln = ln.strip() if not ln: continue m = LA.match(ln) if m: if cur: cur["ans_letter"] = m.group(1).lower() cur["ans_text"] = m.group(2).strip() continue m = LQ.match(ln) if m: q = m.group(2).strip() if 8 < len(q) < 500: cur = {"q": q, "opts": [], "ans_letter": None, "ans_text": None} rows.append(cur) else: cur = None continue m = LO.match(ln) if m and cur is not None: opt = m.group(2).strip() if 1 < len(opt) < 400: cur["opts"].append(opt) out = [] for r in rows: if not r["opts"] or not r["ans_letter"]: continue if len(r["opts"]) < 2: continue ci = ord(r["ans_letter"]) - ord("a") if ci >= len(r["opts"]): continue out.append((r["q"], r["opts"], ci)) return out def main(): done = set() if os.path.exists(PROG): done = set(l.strip() for l in open(PROG, encoding="utf-8") if l.strip()) seen = set() if os.path.exists(OUT): for ln in open(OUT, encoding="utf-8"): try: seen.add(json.loads(ln)["question"]) except Exception: pass fout = open(OUT, "a", encoding="utf-8") flog = open(LOG, "a", encoding="utf-8") new = 0 for ix in INDEXES: html = get(ix) if not html: flog.write(f"{ix}: NO HTML\n") flog.flush() continue links = sorted(set(re.findall(r'href="(https://testfellow\.com/[a-z0-9\-]+/)"', html))) links = [u for u in links if not NO_FETCH.search(u)] for url in links: if url in done or url + ".done" in done: continue h = get(url) if not h: flog.write(f"{url}: NO HTML\n") flog.flush() continue rows = parse_page(h) slug = url.rstrip("/").split("/")[-1] subj = classify(slug) topic = "TestFellow " + slug.replace("-", " ").title() cnt = 0 for q, opts, ci in rows: if q in seen: continue seen.add(q) fout.write(json.dumps({ "subject": subj, "topic": topic, "question": q, "options": opts, "correct": ci + 1, "explanation": "", "source_url": url, }, ensure_ascii=False) + "\n") cnt += 1 new += 1 fout.flush() with open(PROG, "a", encoding="utf-8") as f: f.write(url + "\n") print(f"{url}: {len(rows)} parsed, {cnt} new (total {new})", flush=True) flog.write(f"{url}: {len(rows)} parsed, {cnt} new\n") flog.flush() time.sleep(0.5) fout.close() flog.close() print("DONE total new:", new) if __name__ == "__main__": main()