File "probe_cpp_sites.py"

Full path: /home/algopkco/public_html/scraper/probe_cpp_sites.py
File size: 1.45 B
MIME-type: text/x-script.python
Charset: utf-8

Download   Open   Edit   Advanced Editor &nnbsp; Back

import re

import requests

headers = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "en-US,en;q=0.9",
}

sites = [
    ("javatpoint cpp quiz", "https://www.javatpoint.com/cpp-mcq"),
    ("javatpoint cpp quiz2", "https://www.javatpoint.com/cpp-quiz"),
    ("includehelp cpp mcq", "https://www.includehelp.com/cpp-tutorial/cpp-mcq.aspx"),
    ("studytonight cpp mcq", "https://www.studytonight.com/cpp/mcq-cpp"),
    ("careerride cpp", "https://www.careerride.com/Cpp-questions-answers.aspx"),
    ("examsegg cpp mcq", "https://www.examsegg.com/c-plus-plus-mcq.html"),
    ("interviewbit cpp", "https://www.interviewbit.com/cpp-multiple-choice-questions-mcq/"),
    ("journaldev cpp", "https://www.journaldev.com/44152/cpp-interview-questions"),
    ("sanfoundry alt engineeringmcqs", "https://engineeringmcqs.blogspot.com/"),
    ("mymcqs mirror", "https://www.mymcqs.com/"),
]

for name, url in sites:
    try:
        r = requests.get(url, headers=headers, timeout=20)
        ok = r.status_code == 200 and "just a moment" not in r.text[:500].lower()
        t = re.search(r"<title>(.*?)</title>", r.text, re.S)
        print(f"{name:35s} {r.status_code:4d} len={len(r.text):6d} ok={ok} | {(t.group(1).strip()[:60] if t else '')}")
    except Exception as e:
        print(f"{name:35s} ERR {e}")