File "fetch_t4cpp2.py"
Full path: /home/algopkco/public_html/scraper/fetch_t4cpp2.py
File
size: 10.64 B
MIME-type: text/x-script.python
Charset: utf-8
Download Open Edit Advanced Editor &nnbsp; Back
import html as H
import json
import os
import re
import time
import requests
BASE = os.path.dirname(os.path.abspath(__file__))
DATA = os.path.join(BASE, "data")
OUT = os.path.join(DATA, "t4_mcqs.jsonl")
PROG = os.path.join(BASE, "t4_progress2.txt")
LOG = os.path.join(BASE, "t4_run2.log")
HEADERS = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36",
}
s = requests.Session()
s.headers.update(HEADERS)
SLUGS = [l.strip() for l in open(os.path.join(BASE, "t4cpp_progress.txt"), encoding="utf-8") if l.strip()]
CPP_SLUGS = {
"advanced-c-plus-plus-mcqs", "arrays-mcqs-questions-answers-c", "c-array-solved-mcqs-questions-answers",
"c-mcqs", "c-standard-library-mcqs-questions-answers", "classes-and-inheritance-mcqs-in-c-oop",
"friend-function-mcqs", "function-arguments-pass-by-value-reference-c-mcqs", "inline-function-mcqs",
"mcqs-of-introduction-to-programming", "multi-dimensional-arrays-mcqs",
"one-dimensional-and-multi-dimensional-arrays-c-mcqs",
"one-dimensional-arrays-mcqs", "operator-overloading-solved-mcqs-oop",
"parameter-passing-mechanisms-call-by-value-referencemcqs",
"pointers-and-arrays-c-mcqs", "pointers-solved-mcqs-questions-answers", "pointers-to-functions-c-mcqs",
"pointers-to-pointers-c-mcqs", "polymorphism-mcqs-in-object-oriented-programmingoop",
"programming-c-mcqs-for-programming-competition",
}
CS_SUBJECTS = {
"automata-theory-mcqs": "Computer Science General Test",
"compiler-construction-mcqs": "Computer Science General Test",
"computer-graphics-solved-mcqs-questions-answers": "Computer Science General Test",
"cyber-crime-solved-mcqs-questions-answers": "Computer Science General Test",
"deadlock-mcqs-questions-answers": "Operating Systems",
"digital-image-processing-mcqs": "Computer Science General Test",
"distributed-database-architecture-mcqs-mcqs-in-dbms": "Database Management Systems",
"html-mcqs-test-solved-questions-answers": "Web Technologies",
"html-solved-mcqs": "Web Technologies",
"css-mcqs-for-web-developer-and-managers-jobs-test": "Web Technologies",
"introduction-to-computing-itc-mcqs": "Computer Science General Test",
"java-online-test-mcqs-questions-answers": "Java Programming",
"management-information-system-mis-solved-mcqs-with-answers-pdf": "Management Information Systems",
"mcqs-analysis-of-algorithms-for-jobs-test-solved": "Design And Analysis Of Algorithms",
"mcqs-computer-science-core-courses": "Computer Science General Test",
"mcqs-elective-courses-computer-science": "Computer Science General Test",
"mcqs-on-viruses-and-computer-security": "Computer Security",
"mobile-android-applications-mcqs": "Mobile Application Development",
"mpi-mcqs-message-passing-interface-mcqs": "Computer Science General Test",
"network-layer-osi-model-solved-mcqs": "Computer Networks",
"networking-mcqs-storage-solutions-cloud-computing-mcqs-data-center-technologies-mcqs": "Computer Networks",
"php-mcqs-solved-questions-answers-for-web-developers-and-managers": "Web Technologies",
"social-networks-mcqs-solved-questions-answers": "Computer Science General Test",
"system-programming-mcqs": "Computer Science General Test",
"virtual-memory-mcqs-questions-answers-in-operating-systems": "Operating Systems",
"technical-report-writing-solved-mcqs": "Computer Science General Test",
"technology-management-mcqs": "Computer Science General Test",
"computer-science-mcqs-homepage": "Computer Science General Test",
"computer-science-mcqs-leaks-pdf-ebook-by-fazal-rehman-shamil": "Computer Science General Test",
"software-engineering-mcqs": "Software Engineering",
}
CS_KEYWORDS = re.compile(
r"(computer|software|hardware|programming|algorithm|data|network|database|dbms|html|css|php|java|javascript|"
r"python|android|operating|os-|web|sql|compiler|automata|digital|security|graphic|system|linux|cloud|"
r"information|multimedia|micro|processor|artificial|cryptography|e-?commerce|oop|mcq-of|ic3|ict|itc|pract|"
r"mobile|server|stack|queue|linked|sorting|searching|binary|recursion|hci|cpu|memory|input-output|"
r"windows|ms-office|office)",
re.I,
)
def norm(t):
t = H.unescape(t)
t = re.sub(r"<[^>]+>", "", t)
t = re.sub(r"\s+", " ", t)
return t.strip()
def get(url, tries=4):
for i in range(tries):
try:
r = s.get(url, timeout=30)
if r.status_code == 200:
return r.text
except Exception:
pass
time.sleep(1 + i)
return None
def parse_format_a(html):
rows = []
for b in re.findall(r'<div class="question-container">(.*?)</div>', html, re.S):
qm = re.search(r"question-text\">(.*?)</strong>", b, re.S)
if not qm:
continue
q = re.sub(r"^\d+\.\s*", "", norm(qm.group(1)))
if len(q) < 10 or q.endswith("Back"):
continue
opts, ci = [], -1
for lb in re.findall(r"<label>(.*?)</label>", b, re.S):
vm = re.search(r'value="(.*?)"', lb, re.S)
om = re.search(r"onclick=\"checkAnswer\('[^']*', ?'(.*?)'\)\"", lb, re.S)
if not vm:
continue
label = norm(lb)
m = re.search(r"\(([A-E])\)\s*(.*)$", label, re.S)
if m:
opts.append(norm(m.group(2)))
else:
opts.append(norm(vm.group(1)))
if om and ci < 0:
ans = norm(om.group(1))
for i, o in enumerate(opts):
if o == ans:
ci = i
break
if len(opts) >= 2 and 0 <= ci < len(opts):
rows.append((q, opts, ci))
return rows
def parse_format_c(html):
rows = []
for ch in re.split(r"Q#\s*\d+\s*:", html)[1:]:
ans_part = None
if "Answer:" in ch:
body, ans_part = ch.split("Answer:", 1)
else:
body = ch
pieces = re.split(r"\(([A-E])\)", body)
q = norm(pieces[0])
q = re.split(r"Answer:", q)[0]
if len(q) < 5:
continue
opts = []
for i in range(1, len(pieces) - 1, 2):
opts.append(norm(pieces[i + 1]))
opts = [o for o in opts if o]
ci = -1
if ans_part:
am = re.search(r"^\s*\(([A-E])\)", ans_part)
if am:
ci = "ABCDE".find(am.group(1))
if len(opts) >= 2 and 0 <= ci < len(opts):
rows.append((q, opts, ci))
return rows
def parse_format_d(html):
clean = re.sub(r"<script.*?</script>|<style.*?</style>", "", html, flags=re.S)
clean = re.sub(r"<[^>]+>", " ", clean)
clean = H.unescape(clean)
clean = re.sub(r"\s+", " ", clean)
rows = []
spans = []
for m in re.finditer(r"Answer\s*:\s*([a-eA-E])\s*\)", clean):
spans.append((m.start(), m.group(1)))
for i, (pos, letter) in enumerate(spans):
end = spans[i + 1][0] if i + 1 < len(spans) else min(pos + 400, len(clean))
seg = clean[max(pos - 900, 0):end]
qm = re.search(r"([A-Za-z0-9][^A-Za-z]{0,30}\?)\s", seg)
if not qm:
continue
q = norm(qm.group(1))
if len(q) < 10:
continue
tail = seg[qm.end():]
opts = [norm(x) for x in re.split(r"[a-eA-E]\s*\)\s*", tail) if norm(x)]
ci = "abcde".find(letter.lower())
if len(opts) >= 2 and 0 <= ci < len(opts):
rows.append((q, opts, ci))
return rows
def parse_questions(html):
rows = parse_format_a(html)
if not rows:
rows = parse_format_c(html)
if not rows:
rows = parse_format_d(html)
return rows
def classify(slug):
if slug in CPP_SLUGS:
return "C++ Programming", "T4Tutorials C++ MCQs"
if slug in CS_SUBJECTS:
return CS_SUBJECTS[slug], "T4Tutorials " + slug.replace("-", " ").title()
if CS_KEYWORDS.search(slug):
return "Computer Science General Test", "T4Tutorials " + slug.replace("-", " ").title()
return None, None
def main():
done = set()
if os.path.exists(PROG):
done = set(l.strip() for l in open(PROG, encoding="utf-8") if l.strip())
seen = {}
if os.path.exists(OUT):
for ln in open(OUT, encoding="utf-8"):
try:
j = json.loads(ln)
seen.setdefault(norm(j["question"]), j)
except Exception:
pass
fout = open(OUT, "a", encoding="utf-8")
flog = open(LOG, "a", encoding="utf-8")
crash = open(os.path.join(BASE, "t4_crash.log"), "a", encoding="utf-8")
new = 0
for slug in SLUGS:
if slug in done:
continue
url = f"https://t4tutorials.com/{slug}/"
try:
html = get(url)
if not html:
flog.write(f"{slug}: NO HTML\n")
flog.flush()
with open(PROG, "a", encoding="utf-8") as f:
f.write(slug + "\n")
continue
subj, topic = classify(slug)
import threading
res = {}
thr = threading.Thread(target=lambda: res.update(rows=parse_questions(html)))
thr.start()
thr.join(60)
if thr.is_alive():
flog.write(f"{slug}: PARSE TIMEOUT\n")
flog.flush()
rows = []
else:
rows = res.get("rows", [])
cnt = 0
for q, opts, ci in rows:
k = norm(q)
if k in seen:
continue
it = {
"subject": subj or slug,
"topic": topic or slug,
"question": q,
"options": opts,
"correct": ci + 1,
"explanation": "",
"source_url": url,
}
seen[k] = it
fout.write(json.dumps(it, ensure_ascii=False) + "\n")
cnt += 1
new += 1
fout.flush()
flog.write(f"{slug}: {len(rows)} parsed, {cnt} new\n")
flog.flush()
with open(PROG, "a", encoding="utf-8") as f:
f.write(slug + "\n")
print(f"{slug}: {len(rows)} parsed, {cnt} new (total new {new})", flush=True)
except Exception as e:
import traceback
crash.write(f"CRASH {url}: {e}\n{traceback.format_exc()}\n")
crash.flush()
with open(PROG, "a", encoding="utf-8") as f:
f.write(slug + "\n")
print(f"{slug}: CRASH {e}", flush=True)
time.sleep(0.4)
fout.close()
flog.close()
print("DONE total new:", new)
if __name__ == "__main__":
main()