这是Python排名统计脚本,能自动采集个人名次,保存历史记录,并生成折线趋势图。
```python
#!/usr/bin/env python3
-*- coding: utf-8 -*-
"""
个人排名统计自动收集 + 折线统计图
功能:
-
从数据源(网页表格 / JSON 接口 / 模拟数据)自动抓取排名
-
追加保存到 CSV 历史记录(可增量、可重复运行)
-
读取历史记录绘制排名折线图(名次轴倒置,第 1 名在最上方)
依赖:pip install matplotlib requests beautifulsoup4
"""
from future import annotations
import argparse
import csv
import random
import time
from collections import defaultdict
from dataclasses import dataclass
from datetime import datetime, timedelta
from pathlib import Path
from typing import Iterable, Protocol
import matplotlib
import matplotlib.dates as mdates
import matplotlib.pyplot as plt
from matplotlib import font_manager
from matplotlib.ticker import MaxNLocator
---------------------------------------------------------------- 数据模型
@dataclass
class RankRecord:
timestamp: datetime
name: str
rank: int
score: float | None = None
---------------------------------------------------------------- 存储层
class RankStore:
"""以 CSV 形式持久化历史排名(utf-8-sig 方便 Excel 直接打开)"""
FIELDS = ("timestamp", "name", "rank", "score")
def init(self, path: str | Path):
self.path = Path(path)
def append(self, records: IterableRankRecord) -> int:
records = list(records)
if not records:
return 0
self.path.parent.mkdir(parents=True, exist_ok=True)
is_new = not self.path.exists()
with self.path.open("a", newline="", encoding="utf-8-sig") as f:
writer = csv.DictWriter(f, fieldnames=self.FIELDS)
if is_new:
writer.writeheader()
for r in records:
writer.writerow({
"timestamp": r.timestamp.replace(microsecond=0).isoformat(sep=" "),
"name": r.name,
"rank": r.rank,
"score": "" if r.score is None else r.score,
})
return len(records)
def load(self, names: Iterablestr | None = None) -> dictstr, list\[tuple]:
"""返回 {姓名: (时间, 名次, 分数), ...},按时间升序"""
names = set(names) if names else None
series: dictstr, list\[tuple] = defaultdict(list)
if not self.path.exists():
return series
with self.path.open("r", newline="", encoding="utf-8-sig") as f:
for row in csv.DictReader(f):
name = (row.get("name") or "").strip()
if not name or (names and name not in names):
continue
try:
ts = datetime.fromisoformat(row"timestamp")
rank = int(float(row"rank"))
except (ValueError, KeyError, TypeError):
continue
raw_score = (row.get("score") or "").strip()
score = float(raw_score) if raw_score else None
seriesname.append((ts, rank, score))
for pts in series.values():
pts.sort(key=lambda x: x0)
return series
---------------------------------------------------------------- 采集层
class RankSource(Protocol):
def fetch(self) -> listRankRecord:
...
class MockSource:
"""模拟数据源:随机游走的排名,用于演示和测试"""
def init(self, names: Iterablestr, seed: int | None = None):
self.names = list(names)
self.rng = random.Random(seed)
self.state = {n: self.rng.randint(1, 20) for n in self.names}
def fetch(self) -> listRankRecord:
now = datetime.now().replace(microsecond=0)
records = \[\]
for n in self.names:
self.staten = max(1, self.staten + self.rng.choice(-1, 0, 0, 0, 1))
records.append(RankRecord(now, n, self.staten,
round(1000 - self.staten * 7 + self.rng.uniform(-15, 15), 1)))
return records
class HtmlTableSource:
"""
从网页表格抓取排名,适用于形如:
<table>
<tr><th>名次</th><th>姓名</th><th>积分</th></tr>
<tr><td>1</td><td>张三</td><td>980</td></tr>
</table>
"""
def init(self, url: str, *, row_selector: str = "table tr",
rank_col: int = 0, name_col: int = 1, score_col: int | None = 2,
timeout: int = 10):
self.url = url
self.row_selector = row_selector
self.rank_col, self.name_col, self.score_col = rank_col, name_col, score_col
self.timeout = timeout
def fetch(self) -> listRankRecord:
try:
import requests
from bs4 import BeautifulSoup
except ImportError as e:
raise RuntimeError("请先安装依赖:pip install requests beautifulsoup4") from e
headers = {"User-Agent": "Mozilla/5.0 (compatible; RankTracker/1.0)"}
resp = requests.get(self.url, headers=headers, timeout=self.timeout)
resp.raise_for_status()
resp.encoding = resp.apparent_encoding or resp.encoding
soup = BeautifulSoup(resp.text, "html.parser")
now = datetime.now().replace(microsecond=0)
records: listRankRecord = \[\]
for tr in soup.select(self.row_selector):
cells = c.get_text(strip=True) for c in tr.find_all(\["td", "th")]
if len(cells) <= max(self.rank_col, self.name_col):
continue
rank_text = cellsself.rank_col
if not rank_text.replace(".", "", 1).isdigit(): # 跳过表头
continue
rank = int(float(rank_text))
name = cellsself.name_col
score = None
if self.score_col is not None and len(cells) > self.score_col:
try:
score = float(cellsself.score_col.replace(",", ""))
except ValueError:
score = None
records.append(RankRecord(now, name, rank, score))
return records
class JsonApiSource:
"""从 JSON 接口抓取,例如 {"data": {"rank":1,"name":"张三","score":980}}"""
def init(self, url: str, *, list_path: str = "data", rank_key: str = "rank",
name_key: str = "name", score_key: str | None = "score",
timeout: int = 10):
self.url, self.list_path = url, list_path
self.rank_key, self.name_key, self.score_key = rank_key, name_key, score_key
self.timeout = timeout
@staticmethod
def _dig(obj, path: str):
for key in filter(None, path.split(".")):
obj = objint(key) if isinstance(obj, list) else objkey
return obj
def fetch(self) -> listRankRecord:
import requests # 延迟导入
resp = requests.get(self.url, timeout=self.timeout,
headers={"User-Agent": "RankTracker/1.0"})
resp.raise_for_status()
items = self._dig(resp.json(), self.list_path)
now = datetime.now().replace(microsecond=0)
records = \[\]
for it in items:
try:
records.append(RankRecord(
now,
str(itself.name_key),
int(itself.rank_key),
float(itself.score_key) if self.score_key and it.get(self.score_key) is not None else None,
))
except (KeyError, TypeError, ValueError):
continue
return records
---------------------------------------------------------------- 采集 + 落盘
def collect_once(source: RankSource, store: RankStore) -> int:
records = source.fetch()
n = store.append(records)
ts = datetime.now().strftime("%F %T")
print(f"{ts} 采集到 {n} 条记录 -> {store.path}")
return n
---------------------------------------------------------------- 绘图
def setup_chinese_font() -> None:
"""让 matplotlib 正常显示中文"""
candidates = ["Microsoft YaHei", "SimHei", "PingFang SC", "Heiti SC",
"Noto Sans CJK SC", "Source Han Sans SC", "WenQuanYi Micro Hei",
"Arial Unicode MS"]
installed = {f.name for f in font_manager.fontManager.ttflist}
for name in candidates:
if name in installed:
plt.rcParams"font.sans-serif" = name
break
plt.rcParams"axes.unicode_minus" = False
def plot_rank_trend(store: RankStore, names: Iterablestr | None = None,
out_path: str | Path = "rank_trend.png",
title: str = "个人排名变化趋势", show: bool = False) -> None:
series = store.load(names)
if not series:
print("没有可绘制的数据")
return
fig, ax = plt.subplots(figsize=(11, 6), dpi=130)
for name, pts in sorted(series.items()):
xs = p\[0 for p in pts]
ys = p\[1 for p in pts]
line, = ax.plot(xs, ys, marker="o", markersize=3.5, linewidth=1.8, label=name)
在最后一个点上标注当前名次
ax.annotate(f"{ys-1}", xy=(xs-1, ys-1), xytext=(6, 0),
textcoords="offset points", va="center",
fontsize=9, color=line.get_color(), fontweight="bold")
all_ranks = p\[1 for pts in series.values() for p in pts]
lo, hi = min(all_ranks), max(all_ranks)
ax.set_ylim(hi + 1, lo - 1) # 名次轴倒置:第 1 名在顶部
ax.yaxis.set_major_locator(MaxNLocator(integer=True, nbins=min(12, hi - lo + 2)))
ax.xaxis.set_major_locator(mdates.AutoDateLocator())
ax.xaxis.set_major_formatter(mdates.ConciseDateFormatter(ax.xaxis.get_major_locator()))
ax.set_xlabel("时间")
ax.set_ylabel("名次(越靠上越靠前)")
ax.set_title(f"{title} 更新于 {datetime.now():%Y-%m-%d %H:%M}", fontsize=13)
ax.grid(True, linestyle="--", alpha=0.35)
ax.legend(loc="best", framealpha=0.9)
fig.tight_layout()
out_path = Path(out_path)
out_path.parent.mkdir(parents=True, exist_ok=True)
fig.savefig(out_path, bbox_inches="tight")
print(f"图表已保存:{out_path.resolve()}")
if show:
plt.show()
plt.close(fig)
---------------------------------------------------------------- 演示数据
def seed_mock_history(store: RankStore, names: Iterablestr,
days: int = 45, step_hours: int = 6, seed: int = 2024) -> None:
"""生成一段模拟历史数据,方便直接看效果(已有数据则跳过)"""
if store.path.exists() and store.path.stat().st_size > 0:
print(f"已存在历史数据 {store.path},跳过演示数据生成")
return
rng = random.Random(seed)
names = list(names)
state = {n: rng.randint(3, 15) for n in names}
now = datetime.now().replace(minute=0, second=0, microsecond=0)
t = now - timedelta(days=days)
records: listRankRecord = \[\]
while t <= now:
for n in names:
staten = max(1, staten + rng.choice(-1, -1, 0, 0, 0, 1, 1))
records.append(RankRecord(
t, n, staten,
round(1000 - staten * 7 + rng.uniform(-15, 15), 1),
))
t += timedelta(hours=step_hours)
store.append(records)
print(f"已生成 {len(records)} 条演示数据 -> {store.path}")
---------------------------------------------------------------- 入口
DATA_FILE = Path("data/rank_history.csv")
CHART_FILE = Path("data/rank_trend.png")
DEFAULT_NAMES = "张三", "李四", "王五"
def build_source() -> RankSource:
"""★ 在这里切换成你自己的真实数据源 ★"""
示例 1:网页表格
return HtmlTableSource("https://example.com/rank", rank_col=0, name_col=1, score_col=2)
示例 2:JSON 接口
return JsonApiSource("https://example.com/api/rank", list_path="data")
return MockSource(DEFAULT_NAMES)
def main() -> None:
parser = argparse.ArgumentParser(description="个人排名统计自动收集与折线图")
parser.add_argument("--mode", choices=("once", "loop", "demo"), default="once",
help="once=采集一次并绘图;loop=定时循环采集;demo=生成模拟数据并绘图")
parser.add_argument("--interval", type=int, default=3600, help="loop 模式的采集间隔(秒)")
parser.add_argument("--data", default=str(DATA_FILE), help="历史数据 CSV 路径")
parser.add_argument("--chart", default=str(CHART_FILE), help="输出图片路径")
parser.add_argument("--names", nargs="*", default=None, help="只统计指定姓名")
args = parser.parse_args()
setup_chinese_font()
store = RankStore(args.data)
if args.mode == "demo":
seed_mock_history(store, DEFAULT_NAMES)
plot_rank_trend(store, args.names, args.chart)
return
source = build_source()
if args.mode == "once":
collect_once(source, store)
plot_rank_trend(store, args.names, args.chart)
return
loop 模式:常驻进程定时采集
print(f"开始定时采集,间隔 {args.interval} 秒(Ctrl+C 退出)")
while True:
try:
collect_once(source, store)
plot_rank_trend(store, args.names, args.chart)
except Exception as exc: # 网络抖动不要中断循环
print(f"{datetime.now():%F %T} 采集失败:{exc!r}")
time.sleep(args.interval)
if name == "main":
main()
```
采集、存储与可视化流程
脚本围绕"自动记录名次变化"来设计,您只需选择数据源就能跑起来。
· 采集层:内置了模拟数据、网页表格和JSON接口三种来源,其中MockSource可直接生成随机游走的排名用于演示,HtmlTableSource和JsonApiSource则负责从真实网站抓取。
· 存储层:每次采集到的记录都会追加到CSV文件,按时间排序加载,并自动跳过格式异常的行,方便后续分析。
· 绘图层:读取历史数据后按姓名分组绘制折线,名次轴倒置让第1名显示在顶部,每条线的最后一个数据点会直接标注当前名次。
· 命令行入口:支持--mode once单次采集并绘图、--mode loop定时循环采集、--mode demo生成演示数据三种模式,配合--interval可控制采集频率。
优化建议: 可以在build_source()函数中取消注释并替换HtmlTableSource或JsonApiSource里的示例URL,改成自己真实的排名页面;DEFAULT_NAMES也可以换成实际要跟踪的姓名列表。
仅供参考用。