import argparse import csv import hashlib from collections import Counter from datetime import datetime, timedelta, timezone from pathlib import Path def read_csv(path): with path.open("r", encoding="utf-8-sig", newline="") as handle: return list(csv.DictReader(handle)) def write_csv(path, fieldnames, rows): path.parent.mkdir(parents=True, exist_ok=True) with path.open("w", encoding="utf-8-sig", newline="") as handle: writer = csv.DictWriter(handle, fieldnames=fieldnames) writer.writeheader() writer.writerows(rows) def sha256_file(path): return hashlib.sha256(path.read_bytes()).hexdigest() def classify_source(row): source_name = row.get("source_name", "") source_url = row.get("source_url", "") if "LBMA" in source_name: return { "priority": "HIGH", "source_access_status": "HELD_BY_WEB_VERIFICATION_PAGE", "source_row_status": "SOURCE_PAGE_NOT_ARCHIVED", "date_verify_result": "NEEDS_REPORT_DATE_RESOLUTION", "instrument_verify_result": "SOURCE_SCOPE_RELEVANT_PRECIOUS_METALS", "unit_verify_result": "SOURCE_UNIT_EXPECTED_USD_PER_TROY_OUNCE", "value_consistency_result": "NOT_VERIFIED_SOURCE_ACCESS_HELD", "verified_source_url": source_url, "verification_note": "LBMA official precious metal price entry identified, but current access presented verification gate; do not use as verified value until downloadable/table row is archived.", "next_action": "retry_lbma_download_or_archive_table_row", } if "Shanghai Gold Exchange" in source_name: return { "priority": "HIGH", "source_access_status": "SOURCE_PAGE_LOCATED", "source_row_status": "BENCHMARK_ENTRY_LOCATED", "date_verify_result": "NEEDS_REPORT_DATE_RESOLUTION", "instrument_verify_result": "SHAU_SHAG_BENCHMARK_SCOPE_RELEVANT", "unit_verify_result": "SOURCE_UNIT_EXPECTED_CNY_PER_GRAM", "value_consistency_result": "NOT_VERIFIED_CANDIDATE_DATE_AMBIGUOUS", "verified_source_url": "https://en.sge.com.cn/data_BenchmarkPrice", "verification_note": "SGE benchmark price data entry for Shanghai Gold/Silver Benchmark Price is locatable; candidate fact only has relative date wording, so value/date consistency is not established.", "next_action": "map_report_publication_date_to_sge_benchmark_row", } if "CME" in source_name or "COMEX" in source_name: return { "priority": "HIGH", "source_access_status": "SOURCE_PAGE_LOCATED", "source_row_status": "WAREHOUSE_STOCK_DOWNLOAD_LINKS_LOCATED", "date_verify_result": "NEEDS_REPORT_DATE_RESOLUTION", "instrument_verify_result": "COMEX_GOLD_SILVER_STOCK_SCOPE_RELEVANT", "unit_verify_result": "SOURCE_UNIT_EXPECTED_TROY_OUNCE_OR_REPORT_UNIT", "value_consistency_result": "NOT_VERIFIED_CANDIDATE_DATE_AMBIGUOUS", "verified_source_url": source_url, "verification_note": "CME delivery notice page exposes gold and silver stock download links; report candidate uses relative week wording, so exact inventory row cannot be matched yet.", "next_action": "download_cme_gold_silver_stock_file_for_report_week", } if "SMM" in source_name: return { "priority": "HIGH", "source_access_status": "PARTIAL_ACCESS_VALUE_SIGNIN_REQUIRED", "source_row_status": "PRICE_SPEC_ENTRY_LOCATED_WITH_VALUE_HELD", "date_verify_result": "SOURCE_DATE_VISIBLE_BUT_VALUE_HELD", "instrument_verify_result": "ANTIMONY_99_70_PERCENT_SCOPE_RELEVANT", "unit_verify_result": "SOURCE_UNIT_NEEDS_VISIBLE_VALUE_CONFIRMATION", "value_consistency_result": "NOT_VERIFIED_VALUE_ACCESS_HELD", "verified_source_url": source_url, "verification_note": "SMM antimony 99.70% Sb minimum page is locatable, but price value is not accessible in current environment; use alternative source or archived screenshot before formal use.", "next_action": "replace_or_archive_smm_antimony_quote", } return { "priority": "MEDIUM", "source_access_status": "SOURCE_REVIEW_REQUIRED", "source_row_status": "SOURCE_ROW_REVIEW_REQUIRED", "date_verify_result": "NEEDS_MANUAL_REVIEW", "instrument_verify_result": "NEEDS_MANUAL_REVIEW", "unit_verify_result": "NEEDS_MANUAL_REVIEW", "value_consistency_result": "NOT_VERIFIED", "verified_source_url": source_url, "verification_note": "Source type not recognized by PASS-010 rule set.", "next_action": "manual_source_review", } def main(): parser = argparse.ArgumentParser() parser.add_argument("--project-root", default=".") parser.add_argument("--queue-input", required=True) parser.add_argument("--output", required=True) parser.add_argument("--manifest", required=True) parser.add_argument("--summary", required=True) parser.add_argument("--run-id", default="RUN-ANA-YS-SOURCE-VERIFY-010") args = parser.parse_args() project_root = Path(args.project_root).resolve() queue_rows = read_csv(project_root / args.queue_input) created_at = datetime.now(timezone(timedelta(hours=8))).isoformat(timespec="seconds") rows = [] for row in queue_rows: result = classify_source(row) review_status = "DRAFT_FOR_REVIEW" if result["value_consistency_result"].startswith("NOT_VERIFIED"): evidence_status = "SOURCE_ENTRY_VERIFIED_VALUE_HELD" else: evidence_status = "SOURCE_ENTRY_REVIEW_REQUIRED" rows.append( { "source_verify_result_id": f"YS-SOURCE-VERIFY-010-{len(rows) + 1:04d}", "case_id": row.get("case_id", ""), "batch_id": row.get("batch_id", ""), "run_id": args.run_id, "source_verify_id": row.get("source_verify_id", ""), "source_supplement_id": row.get("source_supplement_id", ""), "gap_priority_id": row.get("gap_priority_id", ""), "priority": result["priority"], "evidence_card_id": row.get("evidence_card_id", ""), "fact_id": row.get("fact_id", ""), "doc_id": row.get("doc_id", ""), "metal_tags": row.get("metal_tags", ""), "theme_tags": row.get("theme_tags", ""), "source_name": row.get("source_name", ""), "source_url": row.get("source_url", ""), "verified_source_url": result["verified_source_url"], "source_access_status": result["source_access_status"], "source_row_status": result["source_row_status"], "date_verify_result": result["date_verify_result"], "instrument_verify_result": result["instrument_verify_result"], "unit_verify_result": result["unit_verify_result"], "value_consistency_result": result["value_consistency_result"], "evidence_status": evidence_status, "value_unit_candidates": row.get("value_unit_candidates", ""), "date_candidates": row.get("date_candidates", ""), "verification_note": result["verification_note"], "next_action": result["next_action"], "review_status": review_status, "created_at": created_at, } ) output_path = project_root / args.output fields = list(rows[0].keys()) if rows else ["source_verify_result_id", "case_id", "run_id", "review_status"] write_csv(output_path, fields, rows) output_sha = sha256_file(output_path) manifest_path = project_root / args.manifest write_csv( manifest_path, ["case_id", "batch_id", "run_id", "artifact_type", "artifact_path", "row_count", "sha256", "review_status", "created_at"], [ { "case_id": "ANA-YS-INDUSTRY-001", "batch_id": "BATCH-001+BATCH-003", "run_id": args.run_id, "artifact_type": "key_fact_source_verify_result", "artifact_path": output_path.relative_to(project_root).as_posix(), "row_count": str(len(rows)), "sha256": output_sha, "review_status": "DRAFT_FOR_REVIEW", "created_at": created_at, } ], ) access_counts = Counter(row["source_access_status"] for row in rows) value_counts = Counter(row["value_consistency_result"] for row in rows) source_counts = Counter(row["source_name"] for row in rows) summary_path = project_root / args.summary lines = [ "# 来源核验结果 PASS-010 摘要", "", "状态:DRAFT_FOR_REVIEW", f"生成时间:{created_at}", "", "## 输出", "", f"- 核验结果表:`{output_path.relative_to(project_root).as_posix()}`", f"- manifest:`{manifest_path.relative_to(project_root).as_posix()}`", f"- 记录数:{len(rows)}", f"- sha256:`{output_sha}`", "", "## 来源访问状态", "", "| 状态 | 数量 |", "|---|---:|", ] for key, count in access_counts.most_common(): lines.append(f"| {key} | {count} |") lines.extend(["", "## 数值一致性状态", "", "| 状态 | 数量 |", "|---|---:|"]) for key, count in value_counts.most_common(): lines.append(f"| {key} | {count} |") lines.extend(["", "## 来源分布", "", "| 来源 | 数量 |", "|---|---:|"]) for key, count in source_counts.most_common(): lines.append(f"| {key} | {count} |") lines.extend( [ "", "## 边界", "", "本轮只完成来源入口、来源类型、单位口径和访问状态核验;由于研报候选事实多为“本周”等相对日期,未完成具体日期、原始表格行和数值一致性核验。所有记录仍为 DRAFT_FOR_REVIEW,不作为正式指标或正式结论。", ] ) summary_path.parent.mkdir(parents=True, exist_ok=True) summary_path.write_text("\n".join(lines) + "\n", encoding="utf-8") print(f"rows={len(rows)}") print(dict(access_counts)) print(dict(value_counts)) print(f"output={output_path.relative_to(project_root).as_posix()}") print(f"manifest={manifest_path.relative_to(project_root).as_posix()}") print(f"summary={summary_path.relative_to(project_root).as_posix()}") if __name__ == "__main__": main()