File "refetch_explanations.py"
Full path: /home/algopkco/public_html/scraper/refetch_explanations.py
File
size: 3.9 B (3.9 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8
Download Open Edit Advanced Editor &nnbsp; Back
import json
import os
import re
import time
import requests
BASE = os.path.dirname(os.path.abspath(__file__))
DATA = os.path.join(BASE, "data")
FILE = os.path.join(DATA, "examveda_mcqs.jsonl")
EXPL_FILE = os.path.join(DATA, "examveda_explanations.jsonl")
HEADERS = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36"
}
session = requests.Session()
session.headers.update(HEADERS)
def get(url, tries=5):
for i in range(tries):
try:
r = session.get(url, timeout=30)
if r.status_code == 200:
return r.text
print(" HTTP", r.status_code, url, flush=True)
if r.status_code == 404:
return None
except Exception as e:
print(" err", e, "retry", i, flush=True)
time.sleep(1 + i * 2)
return None
def clean_html(s):
s = re.sub(r"<br\s*/?>", "\n", s)
s = re.sub(r"<[^>]+>", "", s)
s = re.sub(r"\xa0", " ", s)
s = re.sub(r"[ \t]+", " ", s)
s = re.sub(r"\n\s*\n+", "\n", s)
return s.strip()
def split_articles(html):
idxs = [m.start() for m in re.finditer(r"<article", html)]
out = []
for i, s in enumerate(idxs):
e = idxs[i + 1] if i + 1 < len(idxs) else len(html)
out.append(html[s:e])
return out
def extract_solutions(html):
sols = {}
for art in split_articles(html):
qm = re.search(r'<div class="question-main">(.*?)</div>', art, re.S)
if not qm:
continue
question = clean_html(qm.group(1))
if not question:
continue
m = re.search(
r'<span class="color">Solution:\s*</span></div>\s*(.*?)</div>',
art,
re.S,
)
if m:
expl = clean_html(m.group(1))
expl = re.sub(r"^Solution:\s*", "", expl)
if expl:
sols[question] = expl[:2000]
return sols
def main():
rows = []
with open(FILE, "r", encoding="utf-8") as f:
for ln in f:
try:
rows.append(json.loads(ln))
except Exception:
continue
print("rows:", len(rows), flush=True)
done = set()
if os.path.exists(EXPL_FILE):
with open(EXPL_FILE, encoding="utf-8") as f:
for ln in f:
try:
done.add(json.loads(ln)["page"])
except Exception:
pass
print("pages done:", len(done), flush=True)
pages = {}
for it in rows:
subj = it["subject"].strip("-")
topic = it["topic"].lower().replace(" ", "-")
topic = re.sub(r"[^a-z0-9-]", "", topic)
url = f"https://www.examveda.com/{subj}/{topic}/"
pages.setdefault(url, []).append(it["question"])
print("unique pages:", len(pages), flush=True)
fetched = 0
with open(EXPL_FILE, "a", encoding="utf-8") as out:
for url, qs in pages.items():
if url in done:
continue
total = {}
for pg in range(1, 100):
cur = url if pg == 1 else url + f"?page={pg}"
html = get(cur)
if html is None:
break
sols = extract_solutions(html)
total.update(sols)
# stop when page contains none of our questions
page_qs = [q for q in qs if q in sols]
if pg > 1 and not page_qs and len(sols) < 10:
break
time.sleep(0.3)
for q, e in total.items():
out.write(json.dumps({"page": url, "question": q, "explanation": e}, ensure_ascii=False) + "\n")
out.flush()
fetched += 1
if fetched % 50 == 0:
print("pages fetched:", fetched, "/", len(pages), flush=True)
print("DONE fetched pages:", fetched, flush=True)
if __name__ == "__main__":
main()