File "probe_cpp_sites.py"
Full path: /home/algopkco/public_html/scraper/probe_cpp_sites.py
File
size: 1.45 B
MIME-type: text/x-script.python
Charset: utf-8
Download Open Edit Advanced Editor &nnbsp; Back
import re
import requests
headers = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36",
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
"Accept-Language": "en-US,en;q=0.9",
}
sites = [
("javatpoint cpp quiz", "https://www.javatpoint.com/cpp-mcq"),
("javatpoint cpp quiz2", "https://www.javatpoint.com/cpp-quiz"),
("includehelp cpp mcq", "https://www.includehelp.com/cpp-tutorial/cpp-mcq.aspx"),
("studytonight cpp mcq", "https://www.studytonight.com/cpp/mcq-cpp"),
("careerride cpp", "https://www.careerride.com/Cpp-questions-answers.aspx"),
("examsegg cpp mcq", "https://www.examsegg.com/c-plus-plus-mcq.html"),
("interviewbit cpp", "https://www.interviewbit.com/cpp-multiple-choice-questions-mcq/"),
("journaldev cpp", "https://www.journaldev.com/44152/cpp-interview-questions"),
("sanfoundry alt engineeringmcqs", "https://engineeringmcqs.blogspot.com/"),
("mymcqs mirror", "https://www.mymcqs.com/"),
]
for name, url in sites:
try:
r = requests.get(url, headers=headers, timeout=20)
ok = r.status_code == 200 and "just a moment" not in r.text[:500].lower()
t = re.search(r"<title>(.*?)</title>", r.text, re.S)
print(f"{name:35s} {r.status_code:4d} len={len(r.text):6d} ok={ok} | {(t.group(1).strip()[:60] if t else '')}")
except Exception as e:
print(f"{name:35s} ERR {e}")