From 7b6bc7c126657ae943f3d37e5f19133d8bb187c5 Mon Sep 17 00:00:00 2001
From: zhizhijie <zhizhijie@users.noreply.gitee.com>
Date: 星期四, 17 九月 2026 14:54:44 +0800
Subject: [PATCH] fix(投资): 全部改为「按表头定位」——目标明细表不再写死列号 + 同名列按数据消歧

---
 docs/投资/工具/格式探测.py |   77 +++++++++++++++++++++++++++++++++-----
 1 files changed, 67 insertions(+), 10 deletions(-)

diff --git "a/docs/\346\212\225\350\265\204/\345\267\245\345\205\267/\346\240\274\345\274\217\346\216\242\346\265\213.py" "b/docs/\346\212\225\350\265\204/\345\267\245\345\205\267/\346\240\274\345\274\217\346\216\242\346\265\213.py"
index f23585c..a6ed4dd 100644
--- "a/docs/\346\212\225\350\265\204/\345\267\245\345\205\267/\346\240\274\345\274\217\346\216\242\346\265\213.py"
+++ "b/docs/\346\212\225\350\265\204/\345\267\245\345\205\267/\346\240\274\345\274\217\346\216\242\346\265\213.py"
@@ -193,10 +193,66 @@
     return bool(re.match(r'^-?\d+(\.\d+)?$', norm(v)))
 
 
-def profile(cell, nrows, ncols, hits, city=None):
-    namecol = None
-    if '椤圭洰鍚嶇О' in hits:
-        namecol = hits['椤圭洰鍚嶇О'][0][1]
+def _num_count(cell, nrows, col, start_row1):
+    """缁熻璇ュ垪鍦ㄨ〃澶磋浠ヤ笅鐨勬暟鍊煎崟鍏冩牸涓暟锛堢敤浜庡尯鍒嗗悓鍚嶅垪锛屽銆岃嚜寮�濮嬪缓璁剧疮璁″畬鎴愭姇璧勩�嶄笌銆屸�︽柊澧炲缓绛戦潰绉�嶏級銆�"""
+    n = 0
+    for r in range(start_row1, min(nrows, 800)):
+        v = cell(r, col)
+        if isinstance(v, (int, float)) and not isinstance(v, bool):
+            n += 1
+    return n
+
+
+def pick_columns(cell, nrows, ncols, merges=None, fields=None):
+    """鎸夎〃澶村叧閿瓧瀹氫綅鍒� 鈫� ({瀛楁: 0 鍩哄垪鍙穧, hits)銆�
+
+    涓ょ骇琛ㄥご娑堟锛氳嫢鏌愬瓧娈电殑鍛戒腑钀藉湪**璺ㄥ鍒楃殑鍚堝苟鍗曞厓鏍�**閲岋紝璇存槑瀹冩槸缁勮〃澶达紝涓嶈兘鐩存帴褰撹瀛楁鐨勫垪锛�
+    姝ゆ椂鍦ㄦ湰鍚堝苟鍖哄唴鎺掗櫎銆屽叾瀹冨瓧娈垫洿娣辩殑瀛愯〃澶淬�嶆墍鍗犵殑鍒楋紝鑻ュ彧鍓╁敮涓�涓�鍒楋紝鍒欒鍒楁墠鏄湰瀛楁鐨勫垪銆�
+    渚嬶細閯傚窞銆婁紒涓氭湀鎶ユ眹鎬汇�婻5 缁勮〃澶淬�岃嚜骞村垵绱瀹屾垚鎶曡祫锛堜竾鍏冿級銆嶅悎骞� I5:K5锛�
+        瀛愯〃澶� I=鏈勾璁″垝鎶曡祫銆丣=鍚堣銆並=鍏朵腑锛氭湰鏈堝畬鎴愭姇璧� 鈫�
+        銆岃嚜骞村垵绱銆嶈惤鍦� J锛堜笌鎶曡祫绯荤粺 G 鍒椾竴鑷达級锛岃�屼笉鏄� I锛堥偅鏄湰骞磋鍒掓姇璧勶級銆�
+    """
+    hits = scan_fields(cell, nrows, ncols)
+    fields = list(fields) if fields else [k for k, _ in FIELD_PAT]
+    best = {}
+    for f in fields:
+        if f not in hits:
+            continue
+        uniq, seen = [], set()
+        for h in sorted(hits[f], key=lambda h: (h[0], colindex(h[1]))):
+            if h[1] in seen:
+                continue
+            seen.add(h[1]); uniq.append(h)
+        # 鍚屽悕鍒楁秷姝э細浼樺厛銆岃〃澶翠笅鏂圭湡鐨勬湁鏁板瓧銆嶇殑閭d竴鍒椼��
+        # 渚嬶細姝︽眽鐗╂祦銆岃嚜寮�濮嬪缓璁剧疮璁″畬鎴愭姇璧勶紙涓囧厓锛夈�岺 涓庛�岃嚜寮�濮嬪缓璁剧疮璁℃柊澧炲缓绛戦潰绉�峆 鍚屽悕锛�
+        #     鍙湁 H 鍒楁湁鏁板瓧 鈫� 鍙� H锛堟寜鈥滃懡涓瘝鏈�鐭�濅細閿欏彇 P锛岄偅鏄缓绛戦潰绉級銆�
+        withnum = [h for h in uniq if _num_count(cell, nrows, colindex(h[1]), h[0]) > 0]
+        pool = withnum or uniq
+        best[f] = sorted(pool, key=lambda h: (h[0], colindex(h[1])))[0]
+    cols = {}
+    for f, h in best.items():
+        row1, letter, _txt = h
+        c0 = colindex(letter)
+        span = merge_span(merges, row1, c0)
+        if span and (span[1] - span[0]) > 1:
+            taken = set()
+            for f2, h2 in best.items():
+                if f2 == f or h2[0] <= row1:
+                    continue
+                c2 = colindex(h2[1])
+                if span[0] <= c2 < span[1]:
+                    taken.add(c2)
+            free = [c for c in range(span[0], span[1]) if c not in taken]
+            if len(free) == 1:
+                cols[f] = free[0]
+                continue
+        cols[f] = c0
+    return cols, hits
+
+
+def profile(cell, nrows, ncols, hits, city=None, merges=None):
+    cols, _ = pick_columns(cell, nrows, ncols, merges)
+    namecol = colname(cols['椤圭洰鍚嶇О']) if '椤圭洰鍚嶇О' in cols else None
     data_start = data_end = None
     project_rows = 0
     groups = []
@@ -211,8 +267,8 @@
                 hi = segs[i + 1][0] if i + 1 < len(segs) else 10 ** 9
                 break
     if namecol is not None:
-        nc = colindex(namecol)
-        anchors = [colindex(hits[k][0][1]) for k in ('鑷紑濮嬪缓璁剧疮璁�', '鑷勾鍒濈疮璁�', '鎬绘姇璧�') if k in hits]
+        nc = cols['椤圭洰鍚嶇О']
+        anchors = [cols[k] for k in ('鑷紑濮嬪缓璁剧疮璁�', '鑷勾鍒濈疮璁�', '鎬绘姇璧�') if k in cols]
         for r in range(min(nrows, 800)):
             if not (lo <= r + 1 < hi):
                 continue
@@ -235,7 +291,7 @@
                 data_start = data_start or r + 1
                 data_end = r + 1
                 project_rows += 1
-    return dict(namecol=namecol, data_start=data_start, data_end=data_end,
+    return dict(namecol=namecol, cols=cols, data_start=data_start, data_end=data_end,
                 project_rows=project_rows, groups=groups[:6],
                 city=city or '', city_row=city_row or '', segments=segs[:8],
                 header_rows=sorted({h[0] for v in hits.values() for h in v}))
@@ -302,7 +358,8 @@
                 sel = pick_sheet(open_sheets(p), ym=ym)
                 if sel:
                     nm, nr, nc, cell, hits = sel
-                    pr = profile(cell, nr, nc, hits, city=file_city)
+                    pr = profile(cell, nr, nc, hits, city=file_city,
+                                 merges=sheet_merges(p).get(nm, []))
                     rec.update(宸ヤ綔琛�=nm, 琛ㄥご琛�=','.join(map(str, pr['header_rows'])),
                                鏁版嵁璧�=pr['data_start'], 鏁版嵁姝�=pr['data_end'],
                                椤圭洰琛屾暟=pr['project_rows'], 鍚嶇О鍒�=pr['namecol'] or '',
@@ -310,8 +367,8 @@
                                甯傚窞=pr['city'] or (file_city or ''), 甯傚窞琛�=pr['city_row'],
                                甯傚窞娈�=';'.join('R%s:%s' % s for s in pr['segments']))
                     for k in KEY:
-                        if k in hits:
-                            rec[k + '鍒�'] = hits[k][0][1]
+                        if k in pr['cols']:
+                            rec[k + '鍒�'] = colname(pr['cols'][k])
             except Exception as e:
                 rec['閿欒'] = '%s: %s' % (type(e).__name__, e)
             rows.append(rec)

--
Gitblit v1.9.1