| | |
| | | import org.apache.poi.xwpf.usermodel.XWPFParagraph; |
| | | import org.apache.poi.xwpf.usermodel.XWPFTable; |
| | | import org.apache.poi.xwpf.usermodel.XWPFTableRow; |
| | | import org.w3c.dom.Document; |
| | | import org.w3c.dom.Element; |
| | | import org.w3c.dom.Node; |
| | | import org.w3c.dom.NodeList; |
| | | import cn.hutool.core.io.IoUtil; |
| | | import org.springframework.stereotype.Service; |
| | | import org.springframework.transaction.annotation.Transactional; |
| | |
| | | } |
| | | if (name == null || name.trim().isEmpty()) continue; |
| | | String nt = name.trim().replaceAll("[\r\n]+", ""); |
| | | if (nt.contains("合计") || nt.contains("总计")) continue; |
| | | String ntFlat = nt.replaceAll("\\s+", ""); |
| | | if (ntFlat.contains("合计") || ntFlat.contains("总计") || ntFlat.contains("小计")) continue; |
| | | if (nt.matches("^[一二三四五六七八九十]+、.*")) continue; |
| | | if (nt.matches("^([一二三四五六七八九十]+).*")) continue; |
| | | if (nt.replace(" ", "").matches(".*小\\s*计.*")) continue; |
| | | if (nt.startsWith("填报") || nt.startsWith("单位负责人") || nt.matches("^\\d+、.*")) continue; |
| | | if (hasSeq) { |
| | | if (seq == null || seq.trim().isEmpty()) { |
| | |
| | | } |
| | | if (!hasName) continue; |
| | | Map<Integer, String> colTexts = new HashMap<>(); |
| | | for (int rr = r; rr <= Math.min(r + 3, sh.getLastRowNum()); rr++) { |
| | | int scanEnd = investHeaderScanEnd(sh, r, row); |
| | | for (int rr = r; rr <= scanEnd; rr++) { |
| | | Row rr2 = sh.getRow(rr); |
| | | if (rr2 == null) continue; |
| | | for (int c = 0; c <= rr2.getLastCellNum(); c++) { |
| | |
| | | else if (v.contains("自年初累计")) col.putIfAbsent("yearcum", c); |
| | | else if (v.contains("本月完成")) col.putIfAbsent("monthdone", c); |
| | | else if (v.contains("建设阶段")) col.putIfAbsent("stage", c); |
| | | else if (v.contains("形象进度")) col.putIfAbsent("desc", c); |
| | | else if (v.contains("形象进度") || v.contains("建设进展")) col.putIfAbsent("desc", c); |
| | | else if (v.contains("工可") && !v.contains("项目名称")) col.putIfAbsent("gk", c); |
| | | else if (v.contains("初设") && !v.contains("项目名称")) col.putIfAbsent("cs", c); |
| | | else if (v.contains("建筑面积") && !v.contains("新增")) col.putIfAbsent("area", c); |
| | |
| | | int md = col.get("yearcum") + 1; |
| | | if (md <= row.getLastCellNum()) col.put("monthdone", md); |
| | | } |
| | | // desc 缺失时按 stage+1 兜底:省标「项目建设阶段、进度情况」是 L/M 合并表头, |
| | | // 孝感/恩施/林区/随州等市州只在 L 写「建设阶段」、M 直接写进度文字而不写标题 |
| | | if (!col.containsKey("desc") && col.containsKey("stage") && col.get("stage") >= 0) { |
| | | int dc = col.get("stage") + 1; |
| | | if (dc <= row.getLastCellNum()) col.put("desc", dc); |
| | | } |
| | | for (String k : new String[]{"builder", "nature", "start", "end", "total", "startcum", |
| | | "yearplan", "yearcum", "monthdone", "stage", "desc", "gk", "cs", "area"}) { |
| | | col.putIfAbsent(k, -1); |
| | |
| | | return col; |
| | | } |
| | | return null; |
| | | } |
| | | |
| | | /** 表头文本扫描下界:默认表头行 +3;若「项目名称」列首个数据行更靠下,则扩展到数据行前一行(上限 +12)。 |
| | | * 背景:咸宁/十堰等市州把「建设阶段/形象进度」标签写在表头下第 4~8 行,固定 +3 会漏列(形象进度读不到)。 */ |
| | | private int investHeaderScanEnd(Sheet sh, int headerRow, Row headerRowObj) { |
| | | int hardEnd = Math.min(headerRow + 12, sh.getLastRowNum()); |
| | | int nameCol = -1; |
| | | if (headerRowObj != null) { |
| | | for (int c = 0; c <= headerRowObj.getLastCellNum(); c++) { |
| | | String v = cellStr(headerRowObj, c); |
| | | if (v != null && v.contains("项目名称")) { nameCol = c; break; } |
| | | } |
| | | } |
| | | int firstData = findFirstInvestDataRow(sh, headerRow, nameCol); |
| | | int scanEnd = (firstData > headerRow) |
| | | ? Math.min(hardEnd, firstData - 1) |
| | | : Math.min(headerRow + 3, sh.getLastRowNum()); |
| | | if (scanEnd < headerRow) scanEnd = headerRow; |
| | | return scanEnd; |
| | | } |
| | | |
| | | /** 「项目名称」列首个数据行下标(跳过表头续行与空行);找不到返回 -1 */ |
| | | private int findFirstInvestDataRow(Sheet sh, int headerRow, int nameCol) { |
| | | if (nameCol < 0) return -1; |
| | | for (int r = headerRow + 1; r <= sh.getLastRowNum(); r++) { |
| | | Row row = sh.getRow(r); |
| | | if (row == null) continue; |
| | | String name = cellStr(row, nameCol); |
| | | if (name == null || name.trim().isEmpty()) continue; |
| | | String nt = name.trim().replaceAll("[\r\n]+", ""); |
| | | if (nt.contains("项目名称")) continue; |
| | | String flat = nt.replaceAll("\\s+", ""); |
| | | if (flat.contains("合计") || flat.contains("总计") || flat.contains("小计")) continue; |
| | | if (nt.matches("^[一二三四五六七八九十]+、.*")) continue; |
| | | if (nt.matches("^([一二三四五六七八九十]+).*")) continue; |
| | | if (nt.startsWith("填报") || nt.startsWith("单位负责人") || nt.matches("^\\d+、.*")) continue; |
| | | if (investmentCity(nt) != null) continue; |
| | | return r; |
| | | } |
| | | return -1; |
| | | } |
| | | |
| | | /** 统计 sheet 内项目数据行数(用于多 sheet 历史文件选当月 sheet) */ |
| | |
| | | String name = cellStr(row, nameCol); |
| | | if (name == null || name.trim().isEmpty()) continue; |
| | | String nt = name.trim().replaceAll("[\r\n]+", ""); |
| | | if (nt.contains("合计") || nt.contains("总计")) continue; |
| | | String ntFlat = nt.replaceAll("\\s+", ""); |
| | | if (ntFlat.contains("合计") || ntFlat.contains("总计") || ntFlat.contains("小计")) continue; |
| | | if (nt.matches("^[一二三四五六七八九十]+、.*")) continue; |
| | | if (nt.replace(" ", "").matches(".*小\\s*计.*")) continue; |
| | | if (nt.startsWith("填报") || nt.startsWith("单位负责人") || nt.matches("^\\d+、.*")) continue; |
| | | if (!hasSeq) { |
| | | if (nt.length() > 4) count++; |
| | |
| | | Map<String, Double> orders = new LinkedHashMap<>(); |
| | | double[] sumHolder = new double[]{0.0}; |
| | | String docPeriod; |
| | | try (java.io.InputStream in = file.getInputStream(); |
| | | XWPFDocument doc = new XWPFDocument(in)) { |
| | | docPeriod = periodOfDocx(doc); |
| | | parseWycOrderReportTables(doc, orders, sumHolder); |
| | | byte[] bytes; |
| | | try (java.io.InputStream in = file.getInputStream()) { |
| | | bytes = IoUtil.readBytes(in); |
| | | } |
| | | boolean isZip = bytes.length > 3 && (bytes[0] & 0xFF) == 0x50 && (bytes[1] & 0xFF) == 0x4B; |
| | | boolean isWordMl = !isZip && looksLikeWordMl(bytes); |
| | | try { |
| | | if (isZip && !isWordPackage(bytes)) { |
| | | throw new RuntimeException("这不是 Word 文档(压缩包里没有 word/document.xml)," |
| | | + "多半是 Excel 或其它格式。请上传网约车订单分析报告的 .docx(或 Word 2003 XML)"); |
| | | } |
| | | if (isZip) { |
| | | try (XWPFDocument doc = new XWPFDocument(new java.io.ByteArrayInputStream(bytes))) { |
| | | docPeriod = periodOfParagraphs(paragraphsOfDocx(doc)); |
| | | parseWycOrderReportTables(doc, orders, sumHolder); |
| | | } |
| | | } else if (isWordMl) { |
| | | // 网约车监管平台下发的报告常被 WPS/Word 存成「Word 2003 XML(平铺 XML)」后再改名 .docx: |
| | | // 这种文件不是 zip 包,XWPFDocument 打不开(报 "File is not a zip file"),需按 XML 直接解析。 |
| | | String xml = new String(bytes, java.nio.charset.StandardCharsets.UTF_8); |
| | | Document dom = parseXmlSafely(xml); |
| | | docPeriod = periodOfParagraphs(paragraphsOfDocxXml(dom)); |
| | | parseWycOrderReportTablesXml(dom, orders, sumHolder); |
| | | } else if (isOle2(bytes)) { |
| | | throw new RuntimeException("这是 Word 97-2003 老格式(.doc),请用 WPS/Word 打开后「另存为」docx 格式再导入"); |
| | | } else { |
| | | throw new RuntimeException("无法识别的文件格式:既不是 docx(zip 包),也不是 Word 2003 XML 或 .doc"); |
| | | } |
| | | } catch (RuntimeException e) { |
| | | throw e; |
| | | } catch (Exception e) { |
| | | throw new RuntimeException("无法读取 Word 文档,请上传 .docx 格式的《网约车行业YYYY年M月运行监测情况分析报告》;" |
| | | + "若上传的其实是 Excel/PDF,请改用对应模板导入。(原始错误:" + e.getMessage() + ")"); |
| | |
| | | } |
| | | |
| | | /** 报告标题中的报表期:网约车行业2026年7月 -> 2026-07;取不到返回 null */ |
| | | private String periodOfDocx(XWPFDocument doc) { |
| | | private String periodOfParagraphs(List<String> paragraphs) { |
| | | java.util.regex.Pattern titlePat = java.util.regex.Pattern.compile("网约车行业\\s*(\\d{4})\\s*年\\s*(\\d{1,2})\\s*月"); |
| | | java.util.regex.Pattern anyPat = java.util.regex.Pattern.compile("(\\d{4})\\s*年\\s*(\\d{1,2})\\s*月"); |
| | | String fallback = null; |
| | | int scanned = 0; |
| | | for (XWPFParagraph para : doc.getParagraphs()) { |
| | | String text = para.getText(); |
| | | for (String text : paragraphs) { |
| | | if (text == null) continue; |
| | | String flat = text.replaceAll("\\s+", ""); |
| | | if (flat.isEmpty()) continue; |
| | |
| | | return fallback; |
| | | } |
| | | |
| | | /** docx 段落文本(保持原始顺序) */ |
| | | private List<String> paragraphsOfDocx(XWPFDocument doc) { |
| | | List<String> out = new ArrayList<>(); |
| | | for (XWPFParagraph para : doc.getParagraphs()) out.add(para.getText()); |
| | | return out; |
| | | } |
| | | |
| | | /** Word 2003 XML(WordML)段落文本:<w:p> 下所有 <w:t> 拼接后压掉空白 */ |
| | | private List<String> paragraphsOfDocxXml(Document dom) { |
| | | List<String> out = new ArrayList<>(); |
| | | for (Element p : elementsByLocalName(dom, "p")) out.add(textOfXmlElement(p)); |
| | | return out; |
| | | } |
| | | |
| | | /** 是否为 Word 2003 XML(WordML 平铺 XML)——这种文件常被改名成 .docx */ |
| | | private boolean looksLikeWordMl(byte[] bytes) { |
| | | int n = Math.min(bytes.length, 4096); |
| | | String head = new String(bytes, 0, n, java.nio.charset.StandardCharsets.UTF_8); |
| | | return head.contains("wordml") && head.contains("<w:wordDocument"); |
| | | } |
| | | |
| | | /** zip 包内是否含 word/document.xml(用于区分 docx 与 xlsx/其它 OPC 包) */ |
| | | private boolean isWordPackage(byte[] bytes) { |
| | | try (java.util.zip.ZipInputStream zin = |
| | | new java.util.zip.ZipInputStream(new java.io.ByteArrayInputStream(bytes))) { |
| | | java.util.zip.ZipEntry e; |
| | | while ((e = zin.getNextEntry()) != null) { |
| | | String n = e.getName(); |
| | | if ("word/document.xml".equals(n)) return true; |
| | | } |
| | | } catch (Exception ignore) { |
| | | return false; |
| | | } |
| | | return false; |
| | | } |
| | | |
| | | /** 是否为 OLE2 复合文档(Word 97-2003 .doc) */ |
| | | private boolean isOle2(byte[] b) { |
| | | return b.length > 8 && (b[0] & 0xFF) == 0xD0 && (b[1] & 0xFF) == 0xCF |
| | | && (b[2] & 0xFF) == 0x11 && (b[3] & 0xFF) == 0xE0; |
| | | } |
| | | |
| | | /** 安全解析 XML:禁用外部实体/DTD(报告来自外部平台,防 XXE) */ |
| | | private Document parseXmlSafely(String xml) throws Exception { |
| | | javax.xml.parsers.DocumentBuilderFactory f = javax.xml.parsers.DocumentBuilderFactory.newInstance(); |
| | | f.setNamespaceAware(true); |
| | | try { f.setFeature("http://apache.org/xml/features/nonvalidating/load-external-dtd", false); } catch (Exception ignore) { } |
| | | try { f.setFeature("http://xml.org/sax/features/external-general-entities", false); } catch (Exception ignore) { } |
| | | try { f.setFeature("http://xml.org/sax/features/external-parameter-entities", false); } catch (Exception ignore) { } |
| | | javax.xml.parsers.DocumentBuilder b = f.newDocumentBuilder(); |
| | | b.setEntityResolver(new org.xml.sax.EntityResolver() { |
| | | @Override |
| | | public org.xml.sax.InputSource resolveEntity(String publicId, String systemId) { |
| | | return new org.xml.sax.InputSource(new java.io.StringReader("")); |
| | | } |
| | | }); |
| | | return b.parse(new org.xml.sax.InputSource(new java.io.StringReader(xml))); |
| | | } |
| | | |
| | | /** 按 localName 取全部后代元素(命名空间无关,兼容 WordML 与 docx 的 w: 前缀) */ |
| | | private List<Element> elementsByLocalName(Document dom, String localName) { |
| | | List<Element> out = new ArrayList<>(); |
| | | NodeList list = dom.getElementsByTagNameNS("*", localName); |
| | | for (int i = 0; i < list.getLength(); i++) out.add((Element) list.item(i)); |
| | | return out; |
| | | } |
| | | |
| | | /** 只取直接子元素(避免嵌套表格/嵌套属性元素被算进来) */ |
| | | private List<Element> childElementsByLocalName(Element parent, String localName) { |
| | | List<Element> out = new ArrayList<>(); |
| | | NodeList kids = parent.getChildNodes(); |
| | | for (int i = 0; i < kids.getLength(); i++) { |
| | | Node n = kids.item(i); |
| | | if (n.getNodeType() == Node.ELEMENT_NODE && localName.equals(n.getLocalName())) out.add((Element) n); |
| | | } |
| | | return out; |
| | | } |
| | | |
| | | /** 元素内所有 <w:t> 文本拼接并压掉空白 */ |
| | | private String textOfXmlElement(Element el) { |
| | | StringBuilder sb = new StringBuilder(); |
| | | NodeList ts = el.getElementsByTagNameNS("*", "t"); |
| | | for (int i = 0; i < ts.getLength(); i++) sb.append(ts.item(i).getTextContent()); |
| | | return flat(sb.toString()); |
| | | } |
| | | |
| | | private String ymd(String y, String mo) { |
| | | try { |
| | | int month = Integer.parseInt(mo); |
| | |
| | | |
| | | /** 解析报告中的「地市名称 | … | 订单数总数」表,orders=市州订单,sumHolder[0]=合计行订单数 */ |
| | | private void parseWycOrderReportTables(XWPFDocument doc, Map<String, Double> orders, double[] sumHolder) { |
| | | List<List<List<String>>> tables = new ArrayList<>(); |
| | | for (XWPFTable table : doc.getTables()) { |
| | | List<XWPFTableRow> rows = table.getRows(); |
| | | List<List<String>> rows = new ArrayList<>(); |
| | | for (XWPFTableRow row : table.getRows()) { |
| | | List<String> cells = new ArrayList<>(); |
| | | for (org.apache.poi.xwpf.usermodel.XWPFTableCell c : row.getTableCells()) cells.add(flat(c.getText())); |
| | | rows.add(cells); |
| | | } |
| | | tables.add(rows); |
| | | } |
| | | matchWycOrderTable(tables, orders, sumHolder); |
| | | } |
| | | |
| | | /** Word 2003 XML(WordML)分支:<w:tbl>/<w:tr>/<w:tc> 抽成文本表后走同一套匹配逻辑 */ |
| | | private void parseWycOrderReportTablesXml(Document dom, Map<String, Double> orders, double[] sumHolder) { |
| | | List<List<List<String>>> tables = new ArrayList<>(); |
| | | for (Element tbl : elementsByLocalName(dom, "tbl")) { |
| | | List<List<String>> rows = new ArrayList<>(); |
| | | for (Element tr : childElementsByLocalName(tbl, "tr")) { |
| | | List<String> cells = new ArrayList<>(); |
| | | for (Element tc : childElementsByLocalName(tr, "tc")) cells.add(textOfXmlElement(tc)); |
| | | rows.add(cells); |
| | | } |
| | | tables.add(rows); |
| | | } |
| | | matchWycOrderTable(tables, orders, sumHolder); |
| | | } |
| | | |
| | | /** |
| | | * 在「文本化表格」里找附件1(表头含「订单数总数」/「订单数」)并取 17 市州订单数与合计。 |
| | | * 入参 tables = 表 -> 行 -> 单元格文本(已压空白),与来源格式无关。 |
| | | */ |
| | | private void matchWycOrderTable(List<List<List<String>>> tables, Map<String, Double> orders, double[] sumHolder) { |
| | | for (List<List<String>> rows : tables) { |
| | | if (rows.size() < 3) continue; |
| | | int headerRow = -1; |
| | | int orderCol = -1; |
| | | for (int r = 0; r < Math.min(rows.size(), 3); r++) { |
| | | List<org.apache.poi.xwpf.usermodel.XWPFTableCell> cells = rows.get(r).getTableCells(); |
| | | List<String> cells = rows.get(r); |
| | | for (int c = 0; c < cells.size(); c++) { |
| | | String norm = flat(cells.get(c).getText()); |
| | | String norm = cells.get(c); |
| | | if (norm.contains("订单数总数") || "订单数".equals(norm)) { |
| | | headerRow = r; |
| | | orderCol = c; |
| | |
| | | double citySum = 0.0; |
| | | boolean hasTotal = false; |
| | | for (int r = headerRow + 1; r < rows.size(); r++) { |
| | | List<org.apache.poi.xwpf.usermodel.XWPFTableCell> cells = rows.get(r).getTableCells(); |
| | | if (orderCol >= cells.size()) continue; |
| | | String name = flat(cells.get(0).getText()); |
| | | List<String> cells = rows.get(r); |
| | | if (cells.isEmpty() || orderCol >= cells.size()) continue; |
| | | String name = cells.get(0); |
| | | if (name.isEmpty()) continue; |
| | | Double val = parsePlainNumber(cells.get(orderCol).getText()); |
| | | Double val = parsePlainNumber(cells.get(orderCol)); |
| | | if (val == null) continue; |
| | | String city = RegionUtil.normalizeCityName(name); |
| | | if (city != null && RegionUtil.cityList().contains(city)) { |