| | |
| | | import org.apache.poi.xwpf.usermodel.XWPFParagraph; |
| | | import org.apache.poi.xwpf.usermodel.XWPFTable; |
| | | import org.apache.poi.xwpf.usermodel.XWPFTableRow; |
| | | import org.w3c.dom.Document; |
| | | import org.w3c.dom.Element; |
| | | import org.w3c.dom.Node; |
| | | import org.w3c.dom.NodeList; |
| | | import cn.hutool.core.io.IoUtil; |
| | | import org.springframework.stereotype.Service; |
| | | import org.springframework.transaction.annotation.Transactional; |
| | |
| | | Map<String, Double> orders = new LinkedHashMap<>(); |
| | | double[] sumHolder = new double[]{0.0}; |
| | | String docPeriod; |
| | | try (java.io.InputStream in = file.getInputStream(); |
| | | XWPFDocument doc = new XWPFDocument(in)) { |
| | | docPeriod = periodOfDocx(doc); |
| | | byte[] bytes; |
| | | try (java.io.InputStream in = file.getInputStream()) { |
| | | bytes = IoUtil.readBytes(in); |
| | | } |
| | | boolean isZip = bytes.length > 3 && (bytes[0] & 0xFF) == 0x50 && (bytes[1] & 0xFF) == 0x4B; |
| | | boolean isWordMl = !isZip && looksLikeWordMl(bytes); |
| | | try { |
| | | if (isZip && !isWordPackage(bytes)) { |
| | | throw new RuntimeException("这不是 Word 文档(压缩包里没有 word/document.xml)," |
| | | + "多半是 Excel 或其它格式。请上传网约车订单分析报告的 .docx(或 Word 2003 XML)"); |
| | | } |
| | | if (isZip) { |
| | | try (XWPFDocument doc = new XWPFDocument(new java.io.ByteArrayInputStream(bytes))) { |
| | | docPeriod = periodOfParagraphs(paragraphsOfDocx(doc)); |
| | | parseWycOrderReportTables(doc, orders, sumHolder); |
| | | } |
| | | } else if (isWordMl) { |
| | | // 网约车监管平台下发的报告常被 WPS/Word 存成「Word 2003 XML(平铺 XML)」后再改名 .docx: |
| | | // 这种文件不是 zip 包,XWPFDocument 打不开(报 "File is not a zip file"),需按 XML 直接解析。 |
| | | String xml = new String(bytes, java.nio.charset.StandardCharsets.UTF_8); |
| | | Document dom = parseXmlSafely(xml); |
| | | docPeriod = periodOfParagraphs(paragraphsOfDocxXml(dom)); |
| | | parseWycOrderReportTablesXml(dom, orders, sumHolder); |
| | | } else if (isOle2(bytes)) { |
| | | throw new RuntimeException("这是 Word 97-2003 老格式(.doc),请用 WPS/Word 打开后「另存为」docx 格式再导入"); |
| | | } else { |
| | | throw new RuntimeException("无法识别的文件格式:既不是 docx(zip 包),也不是 Word 2003 XML 或 .doc"); |
| | | } |
| | | } catch (RuntimeException e) { |
| | | throw e; |
| | | } catch (Exception e) { |
| | | throw new RuntimeException("无法读取 Word 文档,请上传 .docx 格式的《网约车行业YYYY年M月运行监测情况分析报告》;" |
| | | + "若上传的其实是 Excel/PDF,请改用对应模板导入。(原始错误:" + e.getMessage() + ")"); |
| | |
| | | } |
| | | |
| | | /** 报告标题中的报表期:网约车行业2026年7月 -> 2026-07;取不到返回 null */ |
| | | private String periodOfDocx(XWPFDocument doc) { |
| | | private String periodOfParagraphs(List<String> paragraphs) { |
| | | java.util.regex.Pattern titlePat = java.util.regex.Pattern.compile("网约车行业\\s*(\\d{4})\\s*年\\s*(\\d{1,2})\\s*月"); |
| | | java.util.regex.Pattern anyPat = java.util.regex.Pattern.compile("(\\d{4})\\s*年\\s*(\\d{1,2})\\s*月"); |
| | | String fallback = null; |
| | | int scanned = 0; |
| | | for (XWPFParagraph para : doc.getParagraphs()) { |
| | | String text = para.getText(); |
| | | for (String text : paragraphs) { |
| | | if (text == null) continue; |
| | | String flat = text.replaceAll("\\s+", ""); |
| | | if (flat.isEmpty()) continue; |
| | |
| | | return fallback; |
| | | } |
| | | |
| | | /** docx 段落文本(保持原始顺序) */ |
| | | private List<String> paragraphsOfDocx(XWPFDocument doc) { |
| | | List<String> out = new ArrayList<>(); |
| | | for (XWPFParagraph para : doc.getParagraphs()) out.add(para.getText()); |
| | | return out; |
| | | } |
| | | |
| | | /** Word 2003 XML(WordML)段落文本:<w:p> 下所有 <w:t> 拼接后压掉空白 */ |
| | | private List<String> paragraphsOfDocxXml(Document dom) { |
| | | List<String> out = new ArrayList<>(); |
| | | for (Element p : elementsByLocalName(dom, "p")) out.add(textOfXmlElement(p)); |
| | | return out; |
| | | } |
| | | |
| | | /** 是否为 Word 2003 XML(WordML 平铺 XML)——这种文件常被改名成 .docx */ |
| | | private boolean looksLikeWordMl(byte[] bytes) { |
| | | int n = Math.min(bytes.length, 4096); |
| | | String head = new String(bytes, 0, n, java.nio.charset.StandardCharsets.UTF_8); |
| | | return head.contains("wordml") && head.contains("<w:wordDocument"); |
| | | } |
| | | |
| | | /** zip 包内是否含 word/document.xml(用于区分 docx 与 xlsx/其它 OPC 包) */ |
| | | private boolean isWordPackage(byte[] bytes) { |
| | | try (java.util.zip.ZipInputStream zin = |
| | | new java.util.zip.ZipInputStream(new java.io.ByteArrayInputStream(bytes))) { |
| | | java.util.zip.ZipEntry e; |
| | | while ((e = zin.getNextEntry()) != null) { |
| | | String n = e.getName(); |
| | | if ("word/document.xml".equals(n)) return true; |
| | | } |
| | | } catch (Exception ignore) { |
| | | return false; |
| | | } |
| | | return false; |
| | | } |
| | | |
| | | /** 是否为 OLE2 复合文档(Word 97-2003 .doc) */ |
| | | private boolean isOle2(byte[] b) { |
| | | return b.length > 8 && (b[0] & 0xFF) == 0xD0 && (b[1] & 0xFF) == 0xCF |
| | | && (b[2] & 0xFF) == 0x11 && (b[3] & 0xFF) == 0xE0; |
| | | } |
| | | |
| | | /** 安全解析 XML:禁用外部实体/DTD(报告来自外部平台,防 XXE) */ |
| | | private Document parseXmlSafely(String xml) throws Exception { |
| | | javax.xml.parsers.DocumentBuilderFactory f = javax.xml.parsers.DocumentBuilderFactory.newInstance(); |
| | | f.setNamespaceAware(true); |
| | | try { f.setFeature("http://apache.org/xml/features/nonvalidating/load-external-dtd", false); } catch (Exception ignore) { } |
| | | try { f.setFeature("http://xml.org/sax/features/external-general-entities", false); } catch (Exception ignore) { } |
| | | try { f.setFeature("http://xml.org/sax/features/external-parameter-entities", false); } catch (Exception ignore) { } |
| | | javax.xml.parsers.DocumentBuilder b = f.newDocumentBuilder(); |
| | | b.setEntityResolver(new org.xml.sax.EntityResolver() { |
| | | @Override |
| | | public org.xml.sax.InputSource resolveEntity(String publicId, String systemId) { |
| | | return new org.xml.sax.InputSource(new java.io.StringReader("")); |
| | | } |
| | | }); |
| | | return b.parse(new org.xml.sax.InputSource(new java.io.StringReader(xml))); |
| | | } |
| | | |
| | | /** 按 localName 取全部后代元素(命名空间无关,兼容 WordML 与 docx 的 w: 前缀) */ |
| | | private List<Element> elementsByLocalName(Document dom, String localName) { |
| | | List<Element> out = new ArrayList<>(); |
| | | NodeList list = dom.getElementsByTagNameNS("*", localName); |
| | | for (int i = 0; i < list.getLength(); i++) out.add((Element) list.item(i)); |
| | | return out; |
| | | } |
| | | |
| | | /** 只取直接子元素(避免嵌套表格/嵌套属性元素被算进来) */ |
| | | private List<Element> childElementsByLocalName(Element parent, String localName) { |
| | | List<Element> out = new ArrayList<>(); |
| | | NodeList kids = parent.getChildNodes(); |
| | | for (int i = 0; i < kids.getLength(); i++) { |
| | | Node n = kids.item(i); |
| | | if (n.getNodeType() == Node.ELEMENT_NODE && localName.equals(n.getLocalName())) out.add((Element) n); |
| | | } |
| | | return out; |
| | | } |
| | | |
| | | /** 元素内所有 <w:t> 文本拼接并压掉空白 */ |
| | | private String textOfXmlElement(Element el) { |
| | | StringBuilder sb = new StringBuilder(); |
| | | NodeList ts = el.getElementsByTagNameNS("*", "t"); |
| | | for (int i = 0; i < ts.getLength(); i++) sb.append(ts.item(i).getTextContent()); |
| | | return flat(sb.toString()); |
| | | } |
| | | |
| | | private String ymd(String y, String mo) { |
| | | try { |
| | | int month = Integer.parseInt(mo); |
| | |
| | | |
| | | /** 解析报告中的「地市名称 | … | 订单数总数」表,orders=市州订单,sumHolder[0]=合计行订单数 */ |
| | | private void parseWycOrderReportTables(XWPFDocument doc, Map<String, Double> orders, double[] sumHolder) { |
| | | List<List<List<String>>> tables = new ArrayList<>(); |
| | | for (XWPFTable table : doc.getTables()) { |
| | | List<XWPFTableRow> rows = table.getRows(); |
| | | List<List<String>> rows = new ArrayList<>(); |
| | | for (XWPFTableRow row : table.getRows()) { |
| | | List<String> cells = new ArrayList<>(); |
| | | for (org.apache.poi.xwpf.usermodel.XWPFTableCell c : row.getTableCells()) cells.add(flat(c.getText())); |
| | | rows.add(cells); |
| | | } |
| | | tables.add(rows); |
| | | } |
| | | matchWycOrderTable(tables, orders, sumHolder); |
| | | } |
| | | |
| | | /** Word 2003 XML(WordML)分支:<w:tbl>/<w:tr>/<w:tc> 抽成文本表后走同一套匹配逻辑 */ |
| | | private void parseWycOrderReportTablesXml(Document dom, Map<String, Double> orders, double[] sumHolder) { |
| | | List<List<List<String>>> tables = new ArrayList<>(); |
| | | for (Element tbl : elementsByLocalName(dom, "tbl")) { |
| | | List<List<String>> rows = new ArrayList<>(); |
| | | for (Element tr : childElementsByLocalName(tbl, "tr")) { |
| | | List<String> cells = new ArrayList<>(); |
| | | for (Element tc : childElementsByLocalName(tr, "tc")) cells.add(textOfXmlElement(tc)); |
| | | rows.add(cells); |
| | | } |
| | | tables.add(rows); |
| | | } |
| | | matchWycOrderTable(tables, orders, sumHolder); |
| | | } |
| | | |
| | | /** |
| | | * 在「文本化表格」里找附件1(表头含「订单数总数」/「订单数」)并取 17 市州订单数与合计。 |
| | | * 入参 tables = 表 -> 行 -> 单元格文本(已压空白),与来源格式无关。 |
| | | */ |
| | | private void matchWycOrderTable(List<List<List<String>>> tables, Map<String, Double> orders, double[] sumHolder) { |
| | | for (List<List<String>> rows : tables) { |
| | | if (rows.size() < 3) continue; |
| | | int headerRow = -1; |
| | | int orderCol = -1; |
| | | for (int r = 0; r < Math.min(rows.size(), 3); r++) { |
| | | List<org.apache.poi.xwpf.usermodel.XWPFTableCell> cells = rows.get(r).getTableCells(); |
| | | List<String> cells = rows.get(r); |
| | | for (int c = 0; c < cells.size(); c++) { |
| | | String norm = flat(cells.get(c).getText()); |
| | | String norm = cells.get(c); |
| | | if (norm.contains("订单数总数") || "订单数".equals(norm)) { |
| | | headerRow = r; |
| | | orderCol = c; |
| | |
| | | double citySum = 0.0; |
| | | boolean hasTotal = false; |
| | | for (int r = headerRow + 1; r < rows.size(); r++) { |
| | | List<org.apache.poi.xwpf.usermodel.XWPFTableCell> cells = rows.get(r).getTableCells(); |
| | | if (orderCol >= cells.size()) continue; |
| | | String name = flat(cells.get(0).getText()); |
| | | List<String> cells = rows.get(r); |
| | | if (cells.isEmpty() || orderCol >= cells.size()) continue; |
| | | String name = cells.get(0); |
| | | if (name.isEmpty()) continue; |
| | | Double val = parsePlainNumber(cells.get(orderCol).getText()); |
| | | Double val = parsePlainNumber(cells.get(orderCol)); |
| | | if (val == null) continue; |
| | | String city = RegionUtil.normalizeCityName(name); |
| | | if (city != null && RegionUtil.cityList().contains(city)) { |