Python排名统计折线图

这是Python排名统计脚本,能自动采集个人名次,保存历史记录,并生成折线趋势图。

```python

#!/usr/bin/env python3

-*- coding: utf-8 -*-

"""

个人排名统计自动收集 + 折线统计图

功能:

  1. 从数据源(网页表格 / JSON 接口 / 模拟数据)自动抓取排名

  2. 追加保存到 CSV 历史记录(可增量、可重复运行)

  3. 读取历史记录绘制排名折线图(名次轴倒置,第 1 名在最上方)

依赖:pip install matplotlib requests beautifulsoup4

"""

from future import annotations

import argparse

import csv

import random

import time

from collections import defaultdict

from dataclasses import dataclass

from datetime import datetime, timedelta

from pathlib import Path

from typing import Iterable, Protocol

import matplotlib

import matplotlib.dates as mdates

import matplotlib.pyplot as plt

from matplotlib import font_manager

from matplotlib.ticker import MaxNLocator

---------------------------------------------------------------- 数据模型

@dataclass

class RankRecord:

timestamp: datetime

name: str

rank: int

score: float | None = None

---------------------------------------------------------------- 存储层

class RankStore:

"""以 CSV 形式持久化历史排名(utf-8-sig 方便 Excel 直接打开)"""

FIELDS = ("timestamp", "name", "rank", "score")

def init(self, path: str | Path):

self.path = Path(path)

def append(self, records: IterableRankRecord) -> int:

records = list(records)

if not records:

return 0

self.path.parent.mkdir(parents=True, exist_ok=True)

is_new = not self.path.exists()

with self.path.open("a", newline="", encoding="utf-8-sig") as f:

writer = csv.DictWriter(f, fieldnames=self.FIELDS)

if is_new:

writer.writeheader()

for r in records:

writer.writerow({

"timestamp": r.timestamp.replace(microsecond=0).isoformat(sep=" "),

"name": r.name,

"rank": r.rank,

"score": "" if r.score is None else r.score,

})

return len(records)

def load(self, names: Iterablestr | None = None) -> dictstr, list\[tuple]:

"""返回 {姓名: (时间, 名次, 分数), ...},按时间升序"""

names = set(names) if names else None

series: dictstr, list\[tuple] = defaultdict(list)

if not self.path.exists():

return series

with self.path.open("r", newline="", encoding="utf-8-sig") as f:

for row in csv.DictReader(f):

name = (row.get("name") or "").strip()

if not name or (names and name not in names):

continue

try:

ts = datetime.fromisoformat(row"timestamp")

rank = int(float(row"rank"))

except (ValueError, KeyError, TypeError):

continue

raw_score = (row.get("score") or "").strip()

score = float(raw_score) if raw_score else None

seriesname.append((ts, rank, score))

for pts in series.values():

pts.sort(key=lambda x: x0)

return series

---------------------------------------------------------------- 采集层

class RankSource(Protocol):

def fetch(self) -> listRankRecord:

...

class MockSource:

"""模拟数据源:随机游走的排名,用于演示和测试"""

def init(self, names: Iterablestr, seed: int | None = None):

self.names = list(names)

self.rng = random.Random(seed)

self.state = {n: self.rng.randint(1, 20) for n in self.names}

def fetch(self) -> listRankRecord:

now = datetime.now().replace(microsecond=0)

records = \[\]

for n in self.names:

self.staten = max(1, self.staten + self.rng.choice(-1, 0, 0, 0, 1))

records.append(RankRecord(now, n, self.staten,

round(1000 - self.staten * 7 + self.rng.uniform(-15, 15), 1)))

return records

class HtmlTableSource:

"""

从网页表格抓取排名,适用于形如:

<table>

<tr><th>名次</th><th>姓名</th><th>积分</th></tr>

<tr><td>1</td><td>张三</td><td>980</td></tr>

</table>

"""

def init(self, url: str, *, row_selector: str = "table tr",

rank_col: int = 0, name_col: int = 1, score_col: int | None = 2,

timeout: int = 10):

self.url = url

self.row_selector = row_selector

self.rank_col, self.name_col, self.score_col = rank_col, name_col, score_col

self.timeout = timeout

def fetch(self) -> listRankRecord:

try:

import requests

from bs4 import BeautifulSoup

except ImportError as e:

raise RuntimeError("请先安装依赖:pip install requests beautifulsoup4") from e

headers = {"User-Agent": "Mozilla/5.0 (compatible; RankTracker/1.0)"}

resp = requests.get(self.url, headers=headers, timeout=self.timeout)

resp.raise_for_status()

resp.encoding = resp.apparent_encoding or resp.encoding

soup = BeautifulSoup(resp.text, "html.parser")

now = datetime.now().replace(microsecond=0)

records: listRankRecord = \[\]

for tr in soup.select(self.row_selector):

cells = c.get_text(strip=True) for c in tr.find_all(\["td", "th")]

if len(cells) <= max(self.rank_col, self.name_col):

continue

rank_text = cellsself.rank_col

if not rank_text.replace(".", "", 1).isdigit(): # 跳过表头

continue

rank = int(float(rank_text))

name = cellsself.name_col

score = None

if self.score_col is not None and len(cells) > self.score_col:

try:

score = float(cellsself.score_col.replace(",", ""))

except ValueError:

score = None

records.append(RankRecord(now, name, rank, score))

return records

class JsonApiSource:

"""从 JSON 接口抓取,例如 {"data": {"rank":1,"name":"张三","score":980}}"""

def init(self, url: str, *, list_path: str = "data", rank_key: str = "rank",

name_key: str = "name", score_key: str | None = "score",

timeout: int = 10):

self.url, self.list_path = url, list_path

self.rank_key, self.name_key, self.score_key = rank_key, name_key, score_key

self.timeout = timeout

@staticmethod

def _dig(obj, path: str):

for key in filter(None, path.split(".")):

obj = objint(key) if isinstance(obj, list) else objkey

return obj

def fetch(self) -> listRankRecord:

import requests # 延迟导入

resp = requests.get(self.url, timeout=self.timeout,

headers={"User-Agent": "RankTracker/1.0"})

resp.raise_for_status()

items = self._dig(resp.json(), self.list_path)

now = datetime.now().replace(microsecond=0)

records = \[\]

for it in items:

try:

records.append(RankRecord(

now,

str(itself.name_key),

int(itself.rank_key),

float(itself.score_key) if self.score_key and it.get(self.score_key) is not None else None,

))

except (KeyError, TypeError, ValueError):

continue

return records

---------------------------------------------------------------- 采集 + 落盘

def collect_once(source: RankSource, store: RankStore) -> int:

records = source.fetch()

n = store.append(records)

ts = datetime.now().strftime("%F %T")

print(f"{ts} 采集到 {n} 条记录 -> {store.path}")

return n

---------------------------------------------------------------- 绘图

def setup_chinese_font() -> None:

"""让 matplotlib 正常显示中文"""

candidates = ["Microsoft YaHei", "SimHei", "PingFang SC", "Heiti SC",

"Noto Sans CJK SC", "Source Han Sans SC", "WenQuanYi Micro Hei",

"Arial Unicode MS"]

installed = {f.name for f in font_manager.fontManager.ttflist}

for name in candidates:

if name in installed:

plt.rcParams"font.sans-serif" = name

break

plt.rcParams"axes.unicode_minus" = False

def plot_rank_trend(store: RankStore, names: Iterablestr | None = None,

out_path: str | Path = "rank_trend.png",

title: str = "个人排名变化趋势", show: bool = False) -> None:

series = store.load(names)

if not series:

print("没有可绘制的数据")

return

fig, ax = plt.subplots(figsize=(11, 6), dpi=130)

for name, pts in sorted(series.items()):

xs = p\[0 for p in pts]

ys = p\[1 for p in pts]

line, = ax.plot(xs, ys, marker="o", markersize=3.5, linewidth=1.8, label=name)

在最后一个点上标注当前名次

ax.annotate(f"{ys-1}", xy=(xs-1, ys-1), xytext=(6, 0),

textcoords="offset points", va="center",

fontsize=9, color=line.get_color(), fontweight="bold")

all_ranks = p\[1 for pts in series.values() for p in pts]

lo, hi = min(all_ranks), max(all_ranks)

ax.set_ylim(hi + 1, lo - 1) # 名次轴倒置:第 1 名在顶部

ax.yaxis.set_major_locator(MaxNLocator(integer=True, nbins=min(12, hi - lo + 2)))

ax.xaxis.set_major_locator(mdates.AutoDateLocator())

ax.xaxis.set_major_formatter(mdates.ConciseDateFormatter(ax.xaxis.get_major_locator()))

ax.set_xlabel("时间")

ax.set_ylabel("名次(越靠上越靠前)")

ax.set_title(f"{title} 更新于 {datetime.now():%Y-%m-%d %H:%M}", fontsize=13)

ax.grid(True, linestyle="--", alpha=0.35)

ax.legend(loc="best", framealpha=0.9)

fig.tight_layout()

out_path = Path(out_path)

out_path.parent.mkdir(parents=True, exist_ok=True)

fig.savefig(out_path, bbox_inches="tight")

print(f"图表已保存:{out_path.resolve()}")

if show:

plt.show()

plt.close(fig)

---------------------------------------------------------------- 演示数据

def seed_mock_history(store: RankStore, names: Iterablestr,

days: int = 45, step_hours: int = 6, seed: int = 2024) -> None:

"""生成一段模拟历史数据,方便直接看效果(已有数据则跳过)"""

if store.path.exists() and store.path.stat().st_size > 0:

print(f"已存在历史数据 {store.path},跳过演示数据生成")

return

rng = random.Random(seed)

names = list(names)

state = {n: rng.randint(3, 15) for n in names}

now = datetime.now().replace(minute=0, second=0, microsecond=0)

t = now - timedelta(days=days)

records: listRankRecord = \[\]

while t <= now:

for n in names:

staten = max(1, staten + rng.choice(-1, -1, 0, 0, 0, 1, 1))

records.append(RankRecord(

t, n, staten,

round(1000 - staten * 7 + rng.uniform(-15, 15), 1),

))

t += timedelta(hours=step_hours)

store.append(records)

print(f"已生成 {len(records)} 条演示数据 -> {store.path}")

---------------------------------------------------------------- 入口

DATA_FILE = Path("data/rank_history.csv")

CHART_FILE = Path("data/rank_trend.png")

DEFAULT_NAMES = "张三", "李四", "王五"

def build_source() -> RankSource:

"""★ 在这里切换成你自己的真实数据源 ★"""

示例 1:网页表格

return HtmlTableSource("https://example.com/rank", rank_col=0, name_col=1, score_col=2)

示例 2:JSON 接口

return JsonApiSource("https://example.com/api/rank", list_path="data")

return MockSource(DEFAULT_NAMES)

def main() -> None:

parser = argparse.ArgumentParser(description="个人排名统计自动收集与折线图")

parser.add_argument("--mode", choices=("once", "loop", "demo"), default="once",

help="once=采集一次并绘图;loop=定时循环采集;demo=生成模拟数据并绘图")

parser.add_argument("--interval", type=int, default=3600, help="loop 模式的采集间隔(秒)")

parser.add_argument("--data", default=str(DATA_FILE), help="历史数据 CSV 路径")

parser.add_argument("--chart", default=str(CHART_FILE), help="输出图片路径")

parser.add_argument("--names", nargs="*", default=None, help="只统计指定姓名")

args = parser.parse_args()

setup_chinese_font()

store = RankStore(args.data)

if args.mode == "demo":

seed_mock_history(store, DEFAULT_NAMES)

plot_rank_trend(store, args.names, args.chart)

return

source = build_source()

if args.mode == "once":

collect_once(source, store)

plot_rank_trend(store, args.names, args.chart)

return

loop 模式:常驻进程定时采集

print(f"开始定时采集,间隔 {args.interval} 秒(Ctrl+C 退出)")

while True:

try:

collect_once(source, store)

plot_rank_trend(store, args.names, args.chart)

except Exception as exc: # 网络抖动不要中断循环

print(f"{datetime.now():%F %T} 采集失败:{exc!r}")

time.sleep(args.interval)

if name == "main":

main()

```

采集、存储与可视化流程

脚本围绕"自动记录名次变化"来设计,您只需选择数据源就能跑起来。

· 采集层:内置了模拟数据、网页表格和JSON接口三种来源,其中MockSource可直接生成随机游走的排名用于演示,HtmlTableSource和JsonApiSource则负责从真实网站抓取。

· 存储层:每次采集到的记录都会追加到CSV文件,按时间排序加载,并自动跳过格式异常的行,方便后续分析。

· 绘图层:读取历史数据后按姓名分组绘制折线,名次轴倒置让第1名显示在顶部,每条线的最后一个数据点会直接标注当前名次。

· 命令行入口:支持--mode once单次采集并绘图、--mode loop定时循环采集、--mode demo生成演示数据三种模式,配合--interval可控制采集频率。


优化建议: 可以在build_source()函数中取消注释并替换HtmlTableSource或JsonApiSource里的示例URL,改成自己真实的排名页面;DEFAULT_NAMES也可以换成实际要跟踪的姓名列表。

仅供参考用。

相关推荐
浩瀚地学2 小时前
deepagents学习打卡day08
经验分享·笔记·python·学习·agent
卷无止境2 小时前
便宜模型和贵模型到底差在哪儿?一份关于Claude与GPT分级体系的深度梳理
后端·python
玩美移动2 小时前
AI Skin Analysis API 技术解析:文件上传、异步任务与结果读取
java·人工智能·python
我是小白呀2 小时前
19-Temporal项目实战:将客户开通流程迁移到持久执行架构
java·开发语言·人工智能·架构·workflow
励志不掉头发的内向程序员2 小时前
【从零写一个CAD 02】画完第一条线之后:实体容器、Esc 取消,和按了没反应的键盘
开发语言·c++·qt·学习·系统架构
摇滚侠3 小时前
《On Java 中文版 基础卷》阅读笔记 安装 Java 和本书示例 02
java·开发语言·笔记
FfHUCisI3 小时前
sync.Once 与 sync.Cond 源码与并发控制陷阱
服务器·开发语言·后端·golang
阿钱真强道3 小时前
21 嵌入式操作系统 | JSON 与 json-c:生成与解析
c语言·开发语言·json·json-c
j7~3 小时前
【Python】(篇六)《基础语法(函数、列表和元组、字典、文件)》---详解
开发语言·python·文件·函数·字典·编程学习·元组与列表