zyj
2026-09-21 13e9b89313f1e9525d5bf805bfcc59c9da8bed52
docs/Ͷ×Ê/¹¤¾ß/¸ñʽ̽²â.py
@@ -106,6 +106,45 @@
    return n - 1
def sheet_merges(path):
    """返回 {sheet_name: [(rlo, rhi, clo, chi), ...]}(0 åŸºã€å³å¼€ï¼‰ã€‚
    ç”¨äºŽè¯†åˆ«ã€Œä¸¤çº§è¡¨å¤´ã€ï¼šç»„表头(跨多列的合并单元格)不能直接当成某字段的列。
    è¯»ä¸åˆ°åˆå¹¶ä¿¡æ¯æ—¶è¿”回已解析到的部分(不抛异常)。
    """
    with open(path, 'rb') as fh:
        magic = fh.read(4)
    out = {}
    try:
        if magic[:4] == b'PK\x03\x04':
            if openpyxl is None:
                return out
            wb = openpyxl.load_workbook(path, data_only=True)
            for nm in wb.sheetnames:
                ws = wb[nm]
                out[nm] = [(r.min_row - 1, r.max_row, r.min_col - 1, r.max_col)
                           for r in ws.merged_cells.ranges]
            wb.close()
        else:
            if xlrd is None:
                return out
            bk = xlrd.open_workbook(path, formatting_info=True)
            for si in range(bk.nsheets):
                sh = bk.sheet_by_index(si)
                out[sh.name] = list(getattr(sh, 'merged_cells', []) or [])
    except Exception:
        return out
    return out
def merge_span(merges, row1, col0):
    """该单元格(1 åŸºè¡Œå·ã€0 åŸºåˆ—号)所在合并区的 (列0èµ·, åˆ—0止右开);不在合并区返回 None。"""
    for (rlo, rhi, clo, chi) in (merges or ()):
        if rlo <= row1 - 1 < rhi and clo <= col0 < chi:
            return (clo, chi)
    return None
def open_sheets(path):
    """返回 [(sheet_name, nrows, ncols, cell_fn), ...]。
@@ -154,10 +193,66 @@
    return bool(re.match(r'^-?\d+(\.\d+)?$', norm(v)))
def profile(cell, nrows, ncols, hits, city=None):
    namecol = None
    if '项目名称' in hits:
        namecol = hits['项目名称'][0][1]
def _num_count(cell, nrows, col, start_row1):
    """统计该列在表头行以下的数值单元格个数(用于区分同名列,如「自开始建设累计完成投资」与「…新增建筑面积」)。"""
    n = 0
    for r in range(start_row1, min(nrows, 800)):
        v = cell(r, col)
        if isinstance(v, (int, float)) and not isinstance(v, bool):
            n += 1
    return n
def pick_columns(cell, nrows, ncols, merges=None, fields=None):
    """按表头关键字定位列 â†’ ({字段: 0 åŸºåˆ—号}, hits)。
    ä¸¤çº§è¡¨å¤´æ¶ˆæ­§ï¼šè‹¥æŸå­—段的命中落在**跨多列的合并单元格**里,说明它是组表头,不能直接当该字段的列;
    æ­¤æ—¶åœ¨æœ¬åˆå¹¶åŒºå†…排除「其它字段更深的子表头」所占的列,若只剩唯一一列,则该列才是本字段的列。
    ä¾‹ï¼šé„‚州《企业月报汇总》R5 ç»„表头「自年初累计完成投资(万元)」合并 I5:K5,
        å­è¡¨å¤´ I=本年计划投资、J=合计、K=其中:本月完成投资 â†’
        ã€Œè‡ªå¹´åˆç´¯è®¡ã€è½åœ¨ J(与投资系统 G åˆ—一致),而不是 I(那是本年计划投资)。
    """
    hits = scan_fields(cell, nrows, ncols)
    fields = list(fields) if fields else [k for k, _ in FIELD_PAT]
    best = {}
    for f in fields:
        if f not in hits:
            continue
        uniq, seen = [], set()
        for h in sorted(hits[f], key=lambda h: (h[0], colindex(h[1]))):
            if h[1] in seen:
                continue
            seen.add(h[1]); uniq.append(h)
        # åŒååˆ—消歧:优先「表头下方真的有数字」的那一列。
        # ä¾‹ï¼šæ­¦æ±‰ç‰©æµã€Œè‡ªå¼€å§‹å»ºè®¾ç´¯è®¡å®ŒæˆæŠ•资(万元)」H ä¸Žã€Œè‡ªå¼€å§‹å»ºè®¾ç´¯è®¡æ–°å¢žå»ºç­‘面积」P åŒåï¼Œ
        #     åªæœ‰ H åˆ—有数字 â†’ å– H(按“命中词最短”会错取 P,那是建筑面积)。
        withnum = [h for h in uniq if _num_count(cell, nrows, colindex(h[1]), h[0]) > 0]
        pool = withnum or uniq
        best[f] = sorted(pool, key=lambda h: (h[0], colindex(h[1])))[0]
    cols = {}
    for f, h in best.items():
        row1, letter, _txt = h
        c0 = colindex(letter)
        span = merge_span(merges, row1, c0)
        if span and (span[1] - span[0]) > 1:
            taken = set()
            for f2, h2 in best.items():
                if f2 == f or h2[0] <= row1:
                    continue
                c2 = colindex(h2[1])
                if span[0] <= c2 < span[1]:
                    taken.add(c2)
            free = [c for c in range(span[0], span[1]) if c not in taken]
            if len(free) == 1:
                cols[f] = free[0]
                continue
        cols[f] = c0
    return cols, hits
def profile(cell, nrows, ncols, hits, city=None, merges=None):
    cols, _ = pick_columns(cell, nrows, ncols, merges)
    namecol = colname(cols['项目名称']) if '项目名称' in cols else None
    data_start = data_end = None
    project_rows = 0
    groups = []
@@ -172,8 +267,8 @@
                hi = segs[i + 1][0] if i + 1 < len(segs) else 10 ** 9
                break
    if namecol is not None:
        nc = colindex(namecol)
        anchors = [colindex(hits[k][0][1]) for k in ('自开始建设累计', '自年初累计', '总投资') if k in hits]
        nc = cols['项目名称']
        anchors = [cols[k] for k in ('自开始建设累计', '自年初累计', '总投资') if k in cols]
        for r in range(min(nrows, 800)):
            if not (lo <= r + 1 < hi):
                continue
@@ -196,7 +291,7 @@
                data_start = data_start or r + 1
                data_end = r + 1
                project_rows += 1
    return dict(namecol=namecol, data_start=data_start, data_end=data_end,
    return dict(namecol=namecol, cols=cols, data_start=data_start, data_end=data_end,
                project_rows=project_rows, groups=groups[:6],
                city=city or '', city_row=city_row or '', segments=segs[:8],
                header_rows=sorted({h[0] for v in hits.values() for h in v}))
@@ -263,7 +358,8 @@
                sel = pick_sheet(open_sheets(p), ym=ym)
                if sel:
                    nm, nr, nc, cell, hits = sel
                    pr = profile(cell, nr, nc, hits, city=file_city)
                    pr = profile(cell, nr, nc, hits, city=file_city,
                                 merges=sheet_merges(p).get(nm, []))
                    rec.update(工作表=nm, è¡¨å¤´è¡Œ=','.join(map(str, pr['header_rows'])),
                               æ•°æ®èµ·=pr['data_start'], æ•°æ®æ­¢=pr['data_end'],
                               é¡¹ç›®è¡Œæ•°=pr['project_rows'], åç§°åˆ—=pr['namecol'] or '',
@@ -271,8 +367,8 @@
                               å¸‚å·ž=pr['city'] or (file_city or ''), å¸‚州行=pr['city_row'],
                               å¸‚州段=';'.join('R%s:%s' % s for s in pr['segments']))
                    for k in KEY:
                        if k in hits:
                            rec[k + '列'] = hits[k][0][1]
                        if k in pr['cols']:
                            rec[k + '列'] = colname(pr['cols'][k])
            except Exception as e:
                rec['错误'] = '%s: %s' % (type(e).__name__, e)
            rows.append(rec)