# ===== CodeLab: Python 爬虫入门实战 =====
# 来源: https://aoerliang.dpdns.org/articles/python-scraping
# 以下代码片段按文章出现顺序拼接, 共 5 段

# ----- 片段 1 (python) -----
import requests

url = "https://example.com/news"
resp = requests.get(url, headers={"User-Agent": "Mozilla/5.0"}, timeout=10)
html = resp.text          # 页面源码
print(html[:200])         # 前 200 个字符

# ----- 片段 2 (bash) -----
pip install beautifulsoup4 lxml

# ----- 片段 3 (python) -----
from bs4 import BeautifulSoup

soup = BeautifulSoup(html, "lxml")

# 取第一个标题
h1 = soup.select_one("h1").get_text(strip=True)

# 取所有文章标题
titles = [a.get_text(strip=True) for a in soup.select("article h2 a")]
print(titles)

# ----- 片段 4 (python) -----
from urllib.parse import urljoin

base = "https://example.com"
for a in soup.select("article a[href]"):
    href = a["href"]
    if href.startswith("/"):
        href = urljoin(base, href)   # 转成完整 URL
    print(href)

# ----- 片段 5 (python) -----
import csv

rows = [{"title": t, "url": u} for t, u in zip(titles, urls)]
with open("news.csv", "w", newline="", encoding="utf-8") as f:
    writer = csv.DictWriter(f, fieldnames=["title", "url"])
    writer.writeheader()
    writer.writerows(rows)
