File "fetch_testfellow.py"
Full path: /home/algopkco/public_html/scraper/fetch_testfellow.py
File
size: 7.28 B (7.28 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8
Download Open Edit Advanced Editor &nnbsp; Back
import json
import os
import re
import time
import requests
BASE = os.path.dirname(os.path.abspath(__file__))
DATA = os.path.join(BASE, "data")
OUT = os.path.join(DATA, "testfellow_mcqs.jsonl")
PROG = os.path.join(BASE, "testfellow_progress.txt")
LOG = os.path.join(BASE, "testfellow_run.log")
H = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/126.0 Safari/537.36"}
s = requests.Session()
s.headers.update(H)
INDEXES = [
"https://testfellow.com/mcqs/",
"https://testfellow.com/biology-mcqs/",
"https://testfellow.com/biology-practice-tests/",
"https://testfellow.com/chemistry-mcqs/",
"https://testfellow.com/chemistry-mcqs-for-lecturer-test/",
"https://testfellow.com/physics-mcqs-for-lecturer-test/",
"https://testfellow.com/zoology-mcqs-for-lecturer-test/",
"https://testfellow.com/english-literature-mcqs-for-lecturer-test/",
"https://testfellow.com/ppsc-computer-science-lecturer-solved-past-paper/",
"https://testfellow.com/computer-mcqs-online-test-quiz/",
"https://testfellow.com/english-mcqs-online-test/",
"https://testfellow.com/everyday-science-mcqs/",
"https://testfellow.com/general-knowledge-mcqs-gk/",
"https://testfellow.com/current-affairs-mcqs/",
"https://testfellow.com/world-geography-gk-mcqs/",
"https://testfellow.com/world-history-mcqs-online-test/",
]
BIOLOGY = re.compile(
r"(aids|animal|anatomy|biology|biomol|biotech|biodiv|bioinform|cell|disease|ecology|evolution|food|genetics|"
r"health|histology|human|immun|infectious|liver|living|lysosome|microbe|microbiol|molecular|morphology|neural|"
r"nitrogen|nutrition|organism|pcr|photosynthesis|phylum|plant|protein|reproduction|respiratory|structural|"
r"transport|vermicomposting|virus|vitamin|antibiotic|breathing|blood|carbohydrate|chromosome|"
r"digestion|respiratory|mineral-physiology)",
re.I,
)
CHEM = re.compile(r"(chem|chlorofluorocarbon|gas|rocket|solution|acid|periodic|atom|chemical)", re.I)
PHYS = re.compile(r"(physic|electric|magnet|light|sound|force|energy|motion|wave|nuclear|optics)", re.I)
COMP = re.compile(r"(computer|software|hardware|internet|programming|dbms|ms-|office)", re.I)
ENG = re.compile(r"(english|grammar|vocabulary|synonym|antonym)", re.I)
GEO = re.compile(r"(geograph|latitude|continent|ocean|river|mountain|country-capital)", re.I)
HIST = re.compile(r"(history|ancient|medieval|freedom|independence|battle)", re.I)
CA = re.compile(r"(current-affairs|appointment|award|scheme|budget)", re.I)
GK = re.compile(r"(gk|general-knowledge|currency|flags|capital|constitution|sports|national)", re.I)
MATH = re.compile(r"(math|algebra|geometry|trigono|arithmetic|percentage|ratio|profit|probability|aptitude)", re.I)
SCI = re.compile(r"(science|pollution|waste|environment)", re.I)
NO_FETCH = re.compile(r"(practice-test|online-test|quiz$|practice-tests)")
def classify(slug):
if BIOLOGY.search(slug):
return "Biology General"
if CHEM.search(slug):
return "Chemistry General"
if PHYS.search(slug):
return "Physics General"
if COMP.search(slug):
return "Computer Science General Test"
if ENG.search(slug):
return "English General"
if GEO.search(slug):
return "Geography General"
if HIST.search(slug):
return "History General"
if CA.search(slug):
return "Current Affairs General"
if GK.search(slug):
return "General Knowledge"
if MATH.search(slug):
return "Mathematics General"
if SCI.search(slug):
return "Everyday Science General"
return "Everyday Science General"
def get(url, tries=4):
for i in range(tries):
try:
r = s.get(url, timeout=30)
if r.status_code == 200:
return r.text
except Exception:
pass
time.sleep(1 + i)
return None
def parse_page(html):
body = re.sub(r"<script.*?</script>|<style.*?</style>", "", html, flags=re.S)
body = body.replace("–", "-").replace(" ", " ").replace("&", "&").replace("&", "&")
plain = re.sub(r"<[^>]+>", "\n", body)
rows = []
cur = None
LQ = re.compile(r"^(\d{1,3})\.\s+(.+)$")
LO = re.compile(r"^([a-eA-E])[\.\)]\s+(.+)$")
LA = re.compile(r"^Answer\s*:\s*([a-eA-E])[\.\)]\s*(.+)$")
for ln in plain.split("\n"):
ln = ln.strip()
if not ln:
continue
m = LA.match(ln)
if m:
if cur:
cur["ans_letter"] = m.group(1).lower()
cur["ans_text"] = m.group(2).strip()
continue
m = LQ.match(ln)
if m:
q = m.group(2).strip()
if 8 < len(q) < 500:
cur = {"q": q, "opts": [], "ans_letter": None, "ans_text": None}
rows.append(cur)
else:
cur = None
continue
m = LO.match(ln)
if m and cur is not None:
opt = m.group(2).strip()
if 1 < len(opt) < 400:
cur["opts"].append(opt)
out = []
for r in rows:
if not r["opts"] or not r["ans_letter"]:
continue
if len(r["opts"]) < 2:
continue
ci = ord(r["ans_letter"]) - ord("a")
if ci >= len(r["opts"]):
continue
out.append((r["q"], r["opts"], ci))
return out
def main():
done = set()
if os.path.exists(PROG):
done = set(l.strip() for l in open(PROG, encoding="utf-8") if l.strip())
seen = set()
if os.path.exists(OUT):
for ln in open(OUT, encoding="utf-8"):
try:
seen.add(json.loads(ln)["question"])
except Exception:
pass
fout = open(OUT, "a", encoding="utf-8")
flog = open(LOG, "a", encoding="utf-8")
new = 0
for ix in INDEXES:
html = get(ix)
if not html:
flog.write(f"{ix}: NO HTML\n")
flog.flush()
continue
links = sorted(set(re.findall(r'href="(https://testfellow\.com/[a-z0-9\-]+/)"', html)))
links = [u for u in links if not NO_FETCH.search(u)]
for url in links:
if url in done or url + ".done" in done:
continue
h = get(url)
if not h:
flog.write(f"{url}: NO HTML\n")
flog.flush()
continue
rows = parse_page(h)
slug = url.rstrip("/").split("/")[-1]
subj = classify(slug)
topic = "TestFellow " + slug.replace("-", " ").title()
cnt = 0
for q, opts, ci in rows:
if q in seen:
continue
seen.add(q)
fout.write(json.dumps({
"subject": subj, "topic": topic, "question": q,
"options": opts, "correct": ci + 1, "explanation": "", "source_url": url,
}, ensure_ascii=False) + "\n")
cnt += 1
new += 1
fout.flush()
with open(PROG, "a", encoding="utf-8") as f:
f.write(url + "\n")
print(f"{url}: {len(rows)} parsed, {cnt} new (total {new})", flush=True)
flog.write(f"{url}: {len(rows)} parsed, {cnt} new\n")
flog.flush()
time.sleep(0.5)
fout.close()
flog.close()
print("DONE total new:", new)
if __name__ == "__main__":
main()