Cai
9 days ago 2fbc2b9ee0dfcf211f57b04769b7694d539d7312
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
import argparse
import csv
import hashlib
from collections import Counter, defaultdict
from datetime import datetime, timezone, timedelta
from pathlib import Path
 
 
CN_TZ = timezone(timedelta(hours=8))
 
 
def read_csv(path: Path):
    with path.open("r", encoding="utf-8-sig", newline="") as f:
        return list(csv.DictReader(f))
 
 
def write_csv(path: Path, fieldnames, rows):
    path.parent.mkdir(parents=True, exist_ok=True)
    with path.open("w", encoding="utf-8-sig", newline="") as f:
        writer = csv.DictWriter(f, fieldnames=fieldnames, extrasaction="ignore")
        writer.writeheader()
        writer.writerows(rows)
 
 
def sha256_file(path: Path) -> str:
    h = hashlib.sha256()
    with path.open("rb") as f:
        for chunk in iter(lambda: f.read(1024 * 1024), b""):
            h.update(chunk)
    return h.hexdigest()
 
 
def split_values(value: str):
    return [item.strip() for item in (value or "").split("|") if item.strip()]
 
 
def first_non_empty(values):
    for value in values:
        if value:
            return value
    return ""
 
 
def build_rows(company_cards, official_sources, recovery_rows, broker_rows, run_id, created_at):
    official_by_company = defaultdict(list)
    recovery_by_company = defaultdict(list)
    broker_by_company = defaultdict(list)
    for row in official_sources:
        official_by_company[row.get("company_name", "")].append(row)
    for row in recovery_rows:
        recovery_by_company[row.get("company_name", "")].append(row)
    for row in broker_rows:
        broker_by_company[row.get("company_name_fragment", "")].append(row)
 
    rows = []
    for idx, card in enumerate(company_cards, start=1):
        company = card.get("company_name", "")
        official = official_by_company.get(company, [])
        recovery = recovery_by_company.get(company, [])
        broker = broker_by_company.get(company, [])
        source_types = sorted({r.get("source_type", "") for r in official if r.get("source_type", "")})
        recovery_types = sorted({r.get("archive_recovery_type", "") for r in recovery if r.get("archive_recovery_type", "")})
        value_candidates = []
        for row in official:
            value_candidates.extend(split_values(row.get("value_unit_candidates", "")))
        value_candidates = sorted(set(value_candidates))
        has_official = bool(official)
        has_recovery = bool(recovery)
        has_broker_split = bool(broker)
        rows.append(
            {
                "structured_company_card_id": f"YS-COMP-STRUCT-017-{idx:04d}",
                "case_id": card.get("case_id", "ANA-YS-INDUSTRY-001"),
                "batch_id": "BATCH-003",
                "run_id": run_id,
                "source_company_card_id": card.get("company_card_id", ""),
                "company_name": company,
                "mention_count": card.get("mention_count", ""),
                "metal_tags": card.get("metal_tags_in_context", ""),
                "stock_code": first_non_empty(row.get("stock_code", "") for row in official),
                "broker_phrase_split_status": "SPLIT_LEDGER_LINKED" if has_broker_split else "NO_SPLIT_LEDGER_MATCH",
                "broker_phrase_split_count": str(len(broker)),
                "sample_context_policy": "DO_NOT_USE_AS_ANALYST_CONCLUSION" if has_broker_split else "RAW_SAMPLE_REVIEW_REQUIRED",
                "official_source_status": "OFFICIAL_SOURCE_ENTRY_DRAFT" if has_official else "NEEDS_OFFICIAL_SOURCE",
                "official_source_count": str(len(official)),
                "official_source_types": "|".join(source_types),
                "archive_recovery_status": "ARCHIVE_RECOVERY_QUEUE_LINKED" if has_recovery else "NO_ARCHIVE_RECOVERY_QUEUE",
                "archive_recovery_count": str(len(recovery)),
                "archive_recovery_types": "|".join(recovery_types),
                "business_exposure_field": "TO_BE_EXTRACTED_FROM_OFFICIAL_SOURCE",
                "resource_or_capacity_field": "TO_BE_EXTRACTED_FROM_OFFICIAL_SOURCE",
                "profit_sensitivity_field": "TO_BE_MODELED_AFTER_VALUE_VERIFICATION",
                "trigger_condition_field": "TO_BE_FILLED_AFTER_EVIDENCE_REVIEW",
                "failure_condition_field": "TO_BE_FILLED_AFTER_EVIDENCE_REVIEW",
                "risk_field": "TO_BE_FILLED_AFTER_EVIDENCE_REVIEW",
                "value_unit_candidates": "|".join(value_candidates),
                "official_archive_blockers": "archive_path_or_hash_missing; original_line_or_table_row_missing; value_consistency_not_verified" if has_recovery else "official_source_missing_or_not_prioritized",
                "analyst_conclusion_status": "NOT_GENERATED_DRAFT_INPUT_ONLY",
                "evidence_status": "STRUCTURED_DRAFT_FIELD_SHELL",
                "next_action": "fill_official_archive_and_original_line_then_value_consistency" if has_official else "find_company_official_or_exchange_source",
                "review_status": "DRAFT_FOR_REVIEW",
                "created_at": created_at,
            }
        )
    return rows
 
 
def write_summary(path: Path, output_path: Path, manifest_path: Path, rows, digest: str, created_at: str):
    by_official = Counter(r["official_source_status"] for r in rows)
    by_split = Counter(r["broker_phrase_split_status"] for r in rows)
    lines = [
        "# 公司证据卡字段化 PASS-017 摘要",
        "",
        "状态:DRAFT_FOR_REVIEW",
        f"生成时间:{created_at}",
        "",
        "## 输出",
        "",
        f"- 字段化公司证据卡:`{output_path.as_posix()}`",
        f"- manifest:`{manifest_path.as_posix()}`",
        f"- 记录数:{len(rows)}",
        f"- sha256:`{digest}`",
        "",
        "## 官方源状态",
        "",
        "| 状态 | 数量 |",
        "|---|---:|",
    ]
    for key, count in by_official.items():
        lines.append(f"| {key} | {count} |")
    lines += ["", "## 券商原文分账链接", "", "| 状态 | 数量 |", "|---|---:|"]
    for key, count in by_split.items():
        lines.append(f"| {key} | {count} |")
    lines += [
        "",
        "## 边界",
        "",
        "本轮只把公司候选卡字段化为证据卡草稿壳,并链接 PASS-016 分账清单、PASS-012 官方源入口和 PASS-014 归档恢复队列。所有公司字段仍为 DRAFT,不构成正式公司证据、正式投资读法、交易指令或收益承诺。",
        "",
    ]
    path.parent.mkdir(parents=True, exist_ok=True)
    path.write_text("\n".join(lines), encoding="utf-8")
 
 
def main():
    parser = argparse.ArgumentParser()
    parser.add_argument("--company-cards", required=True)
    parser.add_argument("--official-sources", required=True)
    parser.add_argument("--archive-recovery", required=True)
    parser.add_argument("--broker-split", required=True)
    parser.add_argument("--output", required=True)
    parser.add_argument("--manifest", required=True)
    parser.add_argument("--summary", required=True)
    parser.add_argument("--run-id", required=True)
    args = parser.parse_args()
    created_at = datetime.now(CN_TZ).isoformat(timespec="seconds")
    output_path = Path(args.output)
    manifest_path = Path(args.manifest)
    rows = build_rows(
        read_csv(Path(args.company_cards)),
        read_csv(Path(args.official_sources)),
        read_csv(Path(args.archive_recovery)),
        read_csv(Path(args.broker_split)),
        args.run_id,
        created_at,
    )
    fieldnames = list(rows[0].keys()) if rows else ["structured_company_card_id", "review_status"]
    write_csv(output_path, fieldnames, rows)
    digest = sha256_file(output_path)
    manifest_rows = [
        {
            "case_id": "ANA-YS-INDUSTRY-001",
            "batch_id": "BATCH-003",
            "run_id": args.run_id,
            "artifact_type": "structured_company_evidence_card",
            "artifact_path": output_path.as_posix(),
            "row_count": str(len(rows)),
            "sha256": digest,
            "review_status": "DRAFT_FOR_REVIEW",
            "created_at": created_at,
        }
    ]
    write_csv(manifest_path, list(manifest_rows[0].keys()), manifest_rows)
    write_summary(Path(args.summary), output_path, manifest_path, rows, digest, created_at)
    print(f"OK rows={len(rows)} output={output_path} sha256={digest}")
 
 
if __name__ == "__main__":
    main()