Files
QUZ/scrape_cbtbank.py
T
root 9e215baa11 m
2026-06-28 23:48:13 +00:00

406 lines
17 KiB
Python

import os
import sys
import time
import json
import sqlite3
import re
import argparse
import hashlib
import requests
from bs4 import BeautifulSoup
from urllib.parse import quote, unquote, urlparse
def html_to_markdown(html_content):
if not html_content:
return ""
soup = BeautifulSoup(html_content, 'html.parser')
# Process paragraph tags
for p in soup.find_all('p'):
p.insert_after('\n\n')
# Process line breaks
for br in soup.find_all('br'):
br.replace_with('\n')
# Process list items
for li in soup.find_all('li'):
li.insert_before('- ')
li.insert_after('\n')
# Process bold text
for strong in soup.find_all(['strong', 'b']):
text = strong.get_text()
if text:
strong.replace_with(f"**{text}**")
# Process links
for a in soup.find_all('a'):
text = a.get_text()
href = a.get('href', '')
if text and href:
a.replace_with(f"[{text}]({href})")
text = soup.get_text()
# Clean up excess newlines
lines = [line.rstrip() for line in text.split('\n')]
cleaned_text = '\n'.join(lines)
while '\n\n\n' in cleaned_text:
cleaned_text = cleaned_text.replace('\n\n\n', '\n\n')
return cleaned_text.strip()
def get_image_filename(url):
ext = os.path.splitext(urlparse(url).path)[1]
if not ext or len(ext) > 5:
ext = ".png"
h = hashlib.md5(url.encode('utf-8')).hexdigest()
return f"cbtbank_{h}{ext}"
def download_and_save_image(image_url, target_dir, delay=1.0):
if not image_url:
return None
# If it is a relative URL, make it absolute
if image_url.startswith('/'):
image_url = f"https://cbtbank.kr{image_url}"
if not image_url.startswith('http'):
return image_url
try:
os.makedirs(target_dir, exist_ok=True)
filename = get_image_filename(image_url)
dest_path = os.path.join(target_dir, filename)
if os.path.exists(dest_path):
return f"/assets/{filename}"
time.sleep(delay)
headers = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36"
}
r = requests.get(image_url, headers=headers, timeout=15)
if r.status_code == 200:
with open(dest_path, "wb") as f:
f.write(r.content)
print(f" [Image] Downloaded image: {filename}")
return f"/assets/{filename}"
else:
print(f" [Image Warning] Failed to download: {image_url} (HTTP {r.status_code})")
except Exception as e:
print(f" [Image Error] Error downloading image: {e}")
return image_url
def get_existing_question_ids(conn):
cursor = conn.cursor()
cursor.execute("SELECT id FROM QuizQuestions WHERE source = 'cbtbank'")
return {row[0] for row in cursor.fetchall()}
def scrape_exam(exam_name, selected_rounds=None, overwrite=False, delay=1.0, db_file="qple_quiz.db", assets_dir="./assets"):
conn = sqlite3.connect(db_file)
# Ensure correct_answers column exists in QuizQuestions table
cursor = conn.cursor()
cursor.execute("PRAGMA table_info(QuizQuestions)")
columns = [col[1] for col in cursor.fetchall()]
if "correct_answers" not in columns:
cursor.execute("ALTER TABLE QuizQuestions ADD COLUMN correct_answers TEXT DEFAULT ''")
conn.commit()
print(">>> Database Migration: Added correct_answers column to QuizQuestions table.")
existing_ids = get_existing_question_ids(conn)
print(f"\n>>> Connecting to SQLite DB: {db_file}")
print(f">>> Target Exam: {exam_name}")
# 1. Fetch main exam list page to find rounds
url = f"https://cbtbank.kr/category/{quote(exam_name)}"
headers = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36"
}
print(f">>> Fetching list of exam rounds from: {url}")
r = requests.get(url, headers=headers, timeout=15)
r.encoding = 'utf-8'
if r.status_code != 200:
print(f"[Error] Failed to fetch exam page: HTTP {r.status_code}")
conn.close()
return
soup = BeautifulSoup(r.text, 'html.parser')
# Find all links pointing to /exam/...
rounds = []
for a in soup.find_all('a'):
href = a.get('href', '')
if href.startswith('/exam/'):
round_id = href.split('/')[-1]
round_name = a.get_text().strip()
rounds.append({
'id': round_id,
'name': round_name,
'href': href
})
# Filter duplicate rounds preserving order
seen = set()
unique_rounds = []
for rnd in rounds:
if rnd['id'] not in seen:
seen.add(rnd['id'])
unique_rounds.append(rnd)
print(f"Found {len(unique_rounds)} unique rounds on cbtbank.kr:")
for rnd in unique_rounds:
print(f" - {rnd['name']} (ID: {rnd['id']})")
if selected_rounds:
filtered_rounds = [rnd for rnd in unique_rounds if rnd['id'] in selected_rounds or rnd['name'] in selected_rounds]
print(f"\n>>> Filtered to {len(filtered_rounds)} selected rounds: {[rnd['name'] for rnd in filtered_rounds]}")
unique_rounds = filtered_rounds
# 2. Iterate and scrape each round page
for round_idx, rnd in enumerate(unique_rounds, 1):
round_id = rnd['id']
round_name = rnd['name']
round_url = f"https://cbtbank.kr/exam/{round_id}"
print("\n" + "="*50)
print(f"Scraping Round ({round_idx}/{len(unique_rounds)}): {round_name}")
print(f"URL: {round_url}")
print("="*50)
time.sleep(delay)
r = requests.get(round_url, headers=headers, timeout=15)
r.encoding = 'utf-8'
if r.status_code != 200:
print(f" [Error] Failed to fetch round page: HTTP {r.status_code}")
continue
round_soup = BeautifulSoup(r.text, 'html.parser')
exams_div = round_soup.find(class_='exams')
if not exams_div:
print(" [Error] Could not find the questions container (class='exams') on this page.")
continue
current_subject = "일반"
parsed_questions_count = 0
# Iterate over the direct children of div.exams
for child in exams_div.children:
if not child.name:
continue
cl = child.get('class', [])
cl_str = " ".join(cl)
# 1. Update current subject if we encounter exam-class-title
if 'exam-class-title' in cl:
subject_text = child.get_text().strip()
if ':' in subject_text:
current_subject = subject_text.split(':', 1)[1].strip()
else:
current_subject = subject_text.strip()
print(f"\n >>> Subject switched to: {current_subject}")
continue
# 2. Process question containers (exam-box) nested inside row text-dark
if 'text-dark' in cl:
boxes = child.find_all(class_='exam-box')
for box in boxes:
q_id = box.get('question-id')
if not q_id:
continue
import zlib
q_id_int = zlib.crc32(q_id.encode('utf-8')) & 0x7fffffff
print(f" Question ID: {q_id} (Int: {q_id_int}) ({current_subject})")
if q_id_int in existing_ids and not overwrite:
print(f" -> {q_id} (Int: {q_id_int}) already exists in DB. Skipping.")
continue
# Extract title tag
title_tag = box.find(class_='exam-title')
if not title_tag:
print(" [Warning] Could not find exam-title. Skipping.")
continue
# Get question image (if any) in box (excluding choices and replies)
q_img_url = None
all_imgs = box.find_all('img')
for img in all_imgs:
# Check that this image is not inside reply or choices
if img.find_parent(class_='reply') or img.find_parent(class_='question-choice'):
continue
src = img.get('src')
if src and not any(x in src for x in ['kakao.png', 'line.png', 'facebook.png', 'twitter.png', 'band.png']):
q_img_url = src
break
# Clean question text by extracting the span.exam-number (question index)
import copy
title_tag_copy = copy.copy(title_tag)
num_span = title_tag_copy.find(class_='exam-number')
if num_span:
num_span.extract()
# Also extract any images from the copy so they don't leak into text
for img in title_tag_copy.find_all('img'):
img.extract()
q_text = title_tag_copy.get_text().strip()
q_text = re.sub(r'^[\s\.]+', '', q_text).strip()
# Extract passage (if any)
passage = ""
passage_div = box.find(class_='passage')
if passage_div:
passage = html_to_markdown(str(passage_div))
# Extract choices
choices = ["", "", "", ""]
correct_idx = -1
correct_str = ""
correct_answers_serialized = ""
choice_list = box.find(class_='question-choice')
if choice_list:
ol = choice_list.find('ol')
correct_attr = ol.get('correct') if ol else None
li_items = choice_list.find_all('li')
for li_idx, li in enumerate(li_items):
# Check if there is an image in choice
choice_img = li.find('img')
if choice_img:
src = choice_img.get('src')
if src:
local_src = download_and_save_image(src, assets_dir, delay)
choice_text = f'<img src="{local_src}" />'
else:
choice_text = ""
else:
choice_text = li.get_text().strip()
if li_idx < 4:
choices[li_idx] = choice_text
# Check correct choice by class
if 'correct' in li.get('class', []):
correct_idx = li_idx
correct_str = choice_text
# Fallback to correct attribute on ol
if correct_idx == -1 and correct_attr:
try:
correct_idx = int(correct_attr) - 1
if 0 <= correct_idx < len(choices):
correct_str = choices[correct_idx]
except ValueError:
pass
# If it's a descriptive question or no choices were found
if not choice_list or not any(choices):
choices = ["", "", "", ""]
correct_idx = -1
correct_str = "주관식"
correct_answers_serialized = str(correct_idx) if correct_idx != -1 else ""
# Download question image if present
final_image_url = ""
if q_img_url:
final_image_url = download_and_save_image(q_img_url, assets_dir, delay)
# Extract explanations
human_explanations = []
ai_explanations = []
reply_list = box.find(class_='view-reply')
if reply_list:
items = reply_list.find_all(class_='reply-item')
for item in items:
nick_tag = item.find(class_='nick')
nick = nick_tag.get_text().strip() if nick_tag else ""
comment_div = item.find(class_='reply-comment')
if comment_div:
comment_text = html_to_markdown(str(comment_div))
if "AI" in nick or "ai" in nick.lower():
ai_explanations.append(f"### 🤖 {nick} AI 해설\n\n{comment_text}")
else:
human_explanations.append(f"**{nick}**: {comment_text}")
human_explanation = "\n\n---\n\n".join(human_explanations)
ai_explanation = "\n\n---\n\n".join(ai_explanations)
# Construct metadata raw json
meta_json = json.dumps({
"source": "cbtbank.kr",
"q_id": q_id,
"round": round_name,
"subject": current_subject
}, ensure_ascii=False)
# Save to database
cursor = conn.cursor()
try:
cursor.execute('''
INSERT OR REPLACE INTO QuizQuestions (
id, source, category, question_text, choice1, choice2, choice3, choice4,
correct_answer, correct_answer_str, explanation, difficulty, subject,
passage, image_url, raw_json, correct_answers, oid
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
''', (
q_id_int, 'cbtbank', exam_name, q_text,
choices[0], choices[1], choices[2], choices[3],
correct_idx, correct_str, human_explanation, round_name, current_subject,
passage, final_image_url, meta_json, correct_answers_serialized, q_id_int
))
if ai_explanation:
cursor.execute('''
INSERT OR REPLACE INTO QuizExplain (id, source, explain)
VALUES (?, ?, ?)
''', (q_id_int, 'cbtbank', ai_explanation))
conn.commit()
existing_ids.add(q_id_int)
parsed_questions_count += 1
print(f" -> Successfully saved to DB!")
except Exception as db_err:
conn.rollback()
print(f" [DB Error] Failed to save {q_id} (Int: {q_id_int}): {db_err}")
print(f"Completed round {round_name}. Saved {parsed_questions_count} new questions.")
conn.close()
print("\n>>> Scrape operation completed successfully!")
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Cbtbank Scraper for Quiz Database")
parser.add_argument("--exam", type=str, default="정보통신기사", help="Name of the exam/category on cbtbank.kr (e.g. 정보통신기사)")
parser.add_argument("--rounds", type=str, default="", help="Comma-separated round names/IDs to scrape (e.g. 'cey20230617'). Default is all.")
parser.add_argument("--overwrite", action="store_true", help="Overwrite existing questions in the database")
parser.add_argument("--delay", type=float, default=1.0, help="Delay in seconds between requests (default: 1.0)")
parser.add_argument("--db", type=str, default="qple_quiz.db", help="Path to sqlite database file (default: qple_quiz.db)")
parser.add_argument("--assets", type=str, default="./assets", help="Path to assets directory to store downloaded images")
args = parser.parse_args()
selected = [r.strip() for r in args.rounds.split(',') if r.strip()] if args.rounds else None
try:
scrape_exam(
exam_name=args.exam,
selected_rounds=selected,
overwrite=args.overwrite,
delay=args.delay,
db_file=args.db,
assets_dir=args.assets
)
except KeyboardInterrupt:
print("\n[Interrupted] Scraper stopped by user.")
sys.exit(0)