Compare commits

..

1 Commits

Author SHA1 Message Date
guke 7c241b4772 比价记录失败卡展示具体原因(新增 fail_reason)
失败记录不再一律「网络开小差」:新增记录级 fail_reason 派生列——information
具体则直出,笼统则从 platform_results 救出业务原因(找不到店/菜、未起送、打烊、
单点不配送等),纯系统失败为 None → 端侧品牌兜底。store_closed/no_delivery 被
pricebot 漏成 status=failed 的按 reason 补判,打烊脏店名统一简短模板。接入
harvest_done 与灰度期 upsert_record 两条写路径。

- models: comparison_record.fail_reason 列
- repositories: _derive_fail_display + 补判/清洗 helper,两条写路径接入
- schemas: ComparisonRecordOut 暴露 fail_reason
- alembic: 加列 + 回填老 specific 失败记录
- tests: _derive_fail_display 单测(8 例)+ harvest 失败落库集成测试

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-28 13:13:14 +08:00
5 changed files with 7 additions and 506 deletions
+7 -17
View File
@@ -298,31 +298,21 @@ def dashboard_overview(
if period_comparison_success_denominator > 0
else None
)
# “平均节省金额”表示一次可计算的成功比价平均能省多少钱:
# 原平台已经最便宜时 saved_amount_cents=0,也必须进入分母;否则只统计
# “确实省到钱”的记录会系统性抬高大盘指标。源价/最优价缺失导致的 NULL
# 无法判断实际节省额,仍排除在分母之外。
period_saved_calculable_count = _count(
period_saved_positive_count = _count(
ComparisonRecord,
*period_comparison_conds,
ComparisonRecord.status == "success",
ComparisonRecord.saved_amount_cents.is_not(None),
ComparisonRecord.saved_amount_cents > 0,
)
period_saved_sum = _sum(
case(
(
ComparisonRecord.saved_amount_cents > 0,
ComparisonRecord.saved_amount_cents,
),
else_=0,
),
period_saved_positive_sum = _sum(
ComparisonRecord.saved_amount_cents,
*period_comparison_conds,
ComparisonRecord.status == "success",
ComparisonRecord.saved_amount_cents.is_not(None),
ComparisonRecord.saved_amount_cents > 0,
)
period_avg_saved_cents = (
round(period_saved_sum / period_saved_calculable_count)
if period_saved_calculable_count
round(period_saved_positive_sum / period_saved_positive_count)
if period_saved_positive_count
else None
)
period_avg_duration_ms = db.execute(
-201
View File
@@ -1,201 +0,0 @@
"""分批修正历史比价记录的最优平台与节省金额派生列。
该回填刻意不放进 Alembic:生产启动迁移不应一次性读取并更新全部历史记录。
调用方按主键游标分批执行,每批独立事务;默认 dry-run 时只计算差异、不写库。
"""
from __future__ import annotations
import json
from dataclasses import dataclass
from decimal import ROUND_HALF_UP, Decimal, InvalidOperation
from typing import Any
from sqlalchemy import func, select
from sqlalchemy.orm import Session
from app.models.comparison import ComparisonRecord
_DERIVED_FIELDS = (
"source_price_cents",
"best_platform_id",
"best_platform_name",
"best_price_cents",
"saved_amount_cents",
"is_source_best",
)
@dataclass(frozen=True)
class SavingsBackfillBatch:
"""一批回填结果;last_id 可直接作为下一批 after_id。"""
scanned: int
changed: int
last_id: int
done: bool
def _json_value(value: Any, fallback: Any) -> Any:
if isinstance(value, str):
try:
return json.loads(value)
except (TypeError, ValueError):
return fallback
return value
def _decimal(value: Any) -> Decimal | None:
if value is None or isinstance(value, bool):
return None
try:
return Decimal(str(value))
except (InvalidOperation, TypeError, ValueError):
return None
def _yuan_to_cents(value: Any) -> int | None:
amount = _decimal(value)
if amount is None:
return None
return int((amount * 100).quantize(Decimal("1"), rounding=ROUND_HALF_UP))
def _rank(value: Any) -> Decimal:
rank = _decimal(value)
return rank if rank is not None else Decimal("1e18")
def derive_historical_savings(
source_price_cents: int | None,
comparison_results: Any,
raw_payload: Any,
) -> dict[str, Any] | None:
"""按当前完整购物篮规则,从历史原始数据重新派生结构化金额字段。"""
results = _json_value(comparison_results, [])
raw = _json_value(raw_payload, {})
if not isinstance(results, list):
return None
platform_results = raw.get("platform_results", {}) if isinstance(raw, dict) else {}
if not isinstance(platform_results, dict):
platform_results = {}
clean: list[tuple[dict[str, Any], Decimal]] = []
for result in results:
if not isinstance(result, dict):
continue
price = _decimal(result.get("price"))
if price is None:
continue
platform_id = result.get("platform_id")
platform_info = platform_results.get(platform_id) if platform_id else None
skipped_count = (
_decimal(platform_info.get("skipped_dish_count"))
if isinstance(platform_info, dict)
else None
)
if skipped_count is not None and skipped_count > 0:
continue
clean.append((result, price))
if not clean:
return None
best, best_price = min(
clean,
key=lambda item: (_rank(item[0].get("rank")), item[1]),
)
if source_price_cents is None:
source = next(
(
result
for result in results
if isinstance(result, dict)
and result.get("is_source")
and _decimal(result.get("price")) is not None
),
None,
)
source_price_cents = _yuan_to_cents(source.get("price")) if source else None
best_price_cents = _yuan_to_cents(best_price)
saved_amount_cents = (
source_price_cents - best_price_cents
if source_price_cents is not None and best_price_cents is not None
else None
)
return {
"source_price_cents": source_price_cents,
"best_platform_id": best.get("platform_id"),
"best_platform_name": best.get("platform_name"),
"best_price_cents": best_price_cents,
"saved_amount_cents": saved_amount_cents,
"is_source_best": bool(best.get("is_source")),
}
def latest_success_id(db: Session) -> int:
"""固定本次任务的扫描上界,避免执行期间新增记录让任务无限延长。"""
return int(
db.execute(
select(func.coalesce(func.max(ComparisonRecord.id), 0)).where(
ComparisonRecord.status == "success"
)
).scalar_one()
)
def repair_comparison_savings_batch(
db: Session,
*,
after_id: int,
max_id: int,
batch_size: int = 500,
apply: bool = False,
) -> SavingsBackfillBatch:
"""按 id 游标处理一批;apply=True 时仅更新实际发生变化的记录并提交。"""
if after_id < 0 or max_id < 0:
raise ValueError("after_id and max_id must be non-negative")
if batch_size <= 0:
raise ValueError("batch_size must be positive")
records = list(
db.scalars(
select(ComparisonRecord)
.where(
ComparisonRecord.status == "success",
ComparisonRecord.id > after_id,
ComparisonRecord.id <= max_id,
)
.order_by(ComparisonRecord.id)
.limit(batch_size)
)
)
changed = 0
for record in records:
derived = derive_historical_savings(
record.source_price_cents,
record.comparison_results,
record.raw_payload,
)
if derived is None:
continue
differs = any(
getattr(record, field) != derived[field] for field in _DERIVED_FIELDS
)
if not differs:
continue
changed += 1
if apply:
for field in _DERIVED_FIELDS:
setattr(record, field, derived[field])
if apply:
db.commit()
last_id = records[-1].id if records else after_id
return SavingsBackfillBatch(
scanned=len(records),
changed=changed,
last_id=last_id,
done=not records or len(records) < batch_size or last_id >= max_id,
)
-103
View File
@@ -1,103 +0,0 @@
"""分批回填历史比价节省金额。
默认只预览,不写库:
python -m scripts.backfill_comparison_savings
确认预览后实际执行:
python -m scripts.backfill_comparison_savings --apply
中断后可从日志最后一个 last_id 继续:
python -m scripts.backfill_comparison_savings --apply --after-id 12345
"""
from __future__ import annotations
import argparse
from app.db.session import SessionLocal
from app.services.comparison_savings_backfill import (
latest_success_id,
repair_comparison_savings_batch,
)
def _positive_int(value: str) -> int:
parsed = int(value)
if parsed <= 0:
raise argparse.ArgumentTypeError("必须为正整数")
return parsed
def _non_negative_int(value: str) -> int:
parsed = int(value)
if parsed < 0:
raise argparse.ArgumentTypeError("不能为负数")
return parsed
def main() -> None:
parser = argparse.ArgumentParser(
description="按完整购物篮口径分批回填历史比价节省金额(默认 dry-run)"
)
parser.add_argument("--apply", action="store_true", help="实际写库;默认仅预览")
parser.add_argument("--batch-size", type=_positive_int, default=500)
parser.add_argument(
"--after-id",
type=_non_negative_int,
default=0,
help="只处理大于该主键的记录,用于断点续跑",
)
parser.add_argument(
"--max-id",
type=_non_negative_int,
help="固定扫描上界;不传则启动时取成功记录最大主键",
)
parser.add_argument(
"--max-batches",
type=_positive_int,
help="本次最多处理多少批,便于灰度限量执行",
)
args = parser.parse_args()
with SessionLocal() as db:
max_id = args.max_id if args.max_id is not None else latest_success_id(db)
mode = "APPLY" if args.apply else "DRY-RUN"
after_id = args.after_id
scanned = 0
changed = 0
batches = 0
print(
f"comparison savings backfill mode={mode} "
f"after_id={after_id} max_id={max_id} batch_size={args.batch_size}"
)
while after_id < max_id:
with SessionLocal() as db:
result = repair_comparison_savings_batch(
db,
after_id=after_id,
max_id=max_id,
batch_size=args.batch_size,
apply=args.apply,
)
batches += 1
scanned += result.scanned
changed += result.changed
after_id = result.last_id
print(
f"batch={batches} scanned={result.scanned} changed={result.changed} "
f"last_id={result.last_id} done={result.done}"
)
if result.done or (
args.max_batches is not None and batches >= args.max_batches
):
break
print(
f"completed mode={mode} batches={batches} scanned={scanned} "
f"changed={changed} last_id={after_id} max_id={max_id}"
)
if __name__ == "__main__":
main()
-49
View File
@@ -120,55 +120,6 @@ def test_dashboard_period_comparison_is_aggregated_by_backend(
assert comparison["token_cost_total_yuan"] == pytest.approx(1.0)
def test_dashboard_average_saved_includes_zero_and_excludes_uncalculable(
admin_client: TestClient, admin_token: str
) -> None:
created_at = datetime(2037, 2, 15, 12)
db = SessionLocal()
try:
db.add_all(
[
ComparisonRecord(
trace_id="dashboard-saved-positive",
status="success",
saved_amount_cents=100,
created_at=created_at,
),
ComparisonRecord(
trace_id="dashboard-saved-source-best",
status="success",
saved_amount_cents=0,
created_at=created_at,
),
ComparisonRecord(
trace_id="dashboard-saved-uncalculable",
status="success",
saved_amount_cents=None,
created_at=created_at,
),
ComparisonRecord(
trace_id="dashboard-saved-failed",
status="failed",
saved_amount_cents=900,
created_at=created_at,
),
]
)
db.commit()
finally:
db.close()
response = admin_client.get(
"/admin/api/stats/overview",
params={"date_from": "2037-02-15", "date_to": "2037-02-15"},
headers=_auth(admin_token),
)
assert response.status_code == 200, response.text
# (100 + 0) / 2NULL 无法计算,failed 不属于成功比价。
assert response.json()["period"]["comparison"]["average_saved_cents"] == 50
def test_user_list_and_detail(admin_client: TestClient, admin_token: str) -> None:
uid = _seed_user_with_data("13800000002")
r = admin_client.get("/admin/api/users", headers=_auth(admin_token))
-136
View File
@@ -1,136 +0,0 @@
from __future__ import annotations
from datetime import datetime
from app.db.session import SessionLocal
from app.models.comparison import ComparisonRecord
from app.services.comparison_savings_backfill import (
latest_success_id,
repair_comparison_savings_batch,
)
def _record(trace_id: str, *, status: str = "success") -> ComparisonRecord:
return ComparisonRecord(
trace_id=trace_id,
status=status,
source_price_cents=4200,
best_platform_id="missing-dishes",
best_platform_name="Incomplete",
best_price_cents=2500,
saved_amount_cents=1700,
is_source_best=False,
comparison_results=[
{
"platform_id": "source",
"platform_name": "Source",
"price": 42,
"rank": 3,
"is_source": True,
},
{
"platform_id": "missing-dishes",
"platform_name": "Incomplete",
"price": 25,
"rank": 1,
"is_source": False,
},
{
"platform_id": "complete",
"platform_name": "Complete",
"price": 38.5,
"rank": 2,
"is_source": False,
},
],
raw_payload={
"platform_results": {
"missing-dishes": {"skipped_dish_count": 1},
"complete": {"skipped_dish_count": 0},
}
},
created_at=datetime(2037, 3, 1, 12),
)
def test_comparison_savings_backfill_is_batched_dry_run_and_idempotent() -> None:
db = SessionLocal()
try:
affected = _record("savings-backfill-affected")
unchanged = _record("savings-backfill-unchanged")
unchanged.best_platform_id = "complete"
unchanged.best_platform_name = "Complete"
unchanged.best_price_cents = 3850
unchanged.saved_amount_cents = 350
failed = _record("savings-backfill-failed", status="failed")
db.add_all([affected, unchanged, failed])
db.commit()
affected_id = affected.id
unchanged_id = unchanged.id
failed_id = failed.id
finally:
db.close()
db = SessionLocal()
try:
assert latest_success_id(db) >= unchanged_id
max_id = unchanged_id
affected = db.get(ComparisonRecord, affected_id)
assert affected is not None
preview = repair_comparison_savings_batch(
db,
after_id=affected_id - 1,
max_id=max_id,
batch_size=1,
)
assert preview.scanned == 1
assert preview.changed == 1
assert preview.last_id == affected_id
assert not preview.done
db.refresh(affected)
assert affected.best_platform_id == "missing-dishes"
assert affected.saved_amount_cents == 1700
applied = repair_comparison_savings_batch(
db,
after_id=affected_id - 1,
max_id=max_id,
batch_size=1,
apply=True,
)
assert applied.changed == 1
db.refresh(affected)
assert affected.best_platform_id == "complete"
assert affected.best_platform_name == "Complete"
assert affected.best_price_cents == 3850
assert affected.saved_amount_cents == 350
second_batch = repair_comparison_savings_batch(
db,
after_id=applied.last_id,
max_id=max_id,
batch_size=1,
apply=True,
)
assert second_batch.scanned == 1
assert second_batch.changed == 0
assert second_batch.last_id == unchanged_id
assert second_batch.done
rerun = repair_comparison_savings_batch(
db,
after_id=affected_id - 1,
max_id=max_id,
batch_size=10,
apply=True,
)
assert rerun.scanned == 2
assert rerun.changed == 0
failed_row = db.get(ComparisonRecord, failed_id)
assert failed_row is not None
assert failed_row.best_platform_id == "missing-dishes"
assert failed_row.saved_amount_cents == 1700
finally:
db.close()