项目初始化

This commit is contained in:
2026-07-02 11:31:16 +08:00
commit eef4c76e3f
64 changed files with 13369 additions and 0 deletions

103
daily/skills_group.py Normal file
View File

@@ -0,0 +1,103 @@
"""Skills 榜单:同 source 合并为一条(企微 Top N"""
from __future__ import annotations
from typing import Any
def skill_id(item: dict[str, Any]) -> str:
return str(item.get("id") or f"{item.get('source')}/{item.get('title')}")
def format_installs(n: int | float) -> str:
if n >= 1_000_000:
return f"{n / 1_000_000:.1f}M"
if n >= 1_000:
return f"{n / 1_000:.1f}K"
return str(int(n))
def _installs_range(items: list[dict[str, Any]]) -> tuple[int, int, str]:
values = [int(item.get("installs") or 0) for item in items]
lo, hi = min(values), max(values)
if lo == hi:
return lo, hi, format_installs(hi)
return lo, hi, f"{format_installs(lo)}{format_installs(hi)}"
def _best_description(items: list[dict[str, Any]]) -> str:
for item in items:
desc = (item.get("description") or "").strip()
if desc:
return desc
return ""
def _cluster_skill(items: list[dict[str, Any]]) -> dict[str, Any]:
ranked = sorted(items, key=lambda x: int(x.get("installs") or 0), reverse=True)
top = ranked[0]
_lo, _hi, installs_fmt = _installs_range(ranked)
titles = [str(x.get("title") or "") for x in ranked if x.get("title")]
sample = ", ".join(titles[:4])
if len(titles) > 4:
sample = f"{sample}"
return {
"id": skill_id(top),
"title": top.get("title", ""),
"source": top.get("source", ""),
"installs": int(top.get("installs") or 0),
"installs_min": _lo,
"installs_max": _hi,
"installs_fmt": installs_fmt,
"link": top.get("link", ""),
"description": _best_description(ranked),
"cluster": True,
"cluster_count": len(ranked),
"cluster_skills": titles,
"cluster_titles": sample,
}
def _single_skill(item: dict[str, Any]) -> dict[str, Any]:
installs = int(item.get("installs") or 0)
return {
"id": skill_id(item),
"title": item.get("title", ""),
"source": item.get("source", ""),
"installs": installs,
"installs_fmt": format_installs(installs),
"link": item.get("link", ""),
"description": item.get("description", ""),
"cluster": False,
}
def group_skills_by_source(
items: list[dict[str, Any]],
*,
limit: int = 10,
pool_size: int = 50,
) -> list[dict[str, Any]]:
"""按 source 去重合并;保留各 source 在榜内的最佳名次顺序。"""
if not items or limit <= 0:
return []
pool = items[: max(pool_size, limit)]
by_source: dict[str, list[dict[str, Any]]] = {}
first_rank: dict[str, int] = {}
for rank, item in enumerate(pool, 1):
source = (item.get("source") or "?").strip() or "?"
by_source.setdefault(source, []).append(item)
first_rank.setdefault(source, rank)
ordered_sources = sorted(by_source.keys(), key=lambda s: first_rank[s])
result: list[dict[str, Any]] = []
for source in ordered_sources:
group = by_source[source]
if len(group) == 1:
result.append(_single_skill(group[0]))
else:
result.append(_cluster_skill(group))
if len(result) >= limit:
break
return result