From df6928d0cb27594b4ff770e4ce7b10ceb5caa126 Mon Sep 17 00:00:00 2001 From: mediabot-pt <295750538+mediabot-pt@users.noreply.github.com> Date: Wed, 1 Jul 2026 10:13:33 +0800 Subject: [PATCH] =?UTF-8?q?refactor:=20JSON=E8=A7=A3=E6=9E=90=E6=94=B9?= =?UTF-8?q?=E4=B8=BA=E5=8C=B9=E9=85=8D=E9=A1=B5=E9=9D=A2=E8=A7=84=E5=88=99?= =?UTF-8?q?+selector=E8=B7=AF=E5=BE=84=E6=8F=90=E5=8F=96=E5=AD=97=E6=AE=B5?= =?UTF-8?q?=EF=BC=8C=E4=B8=8D=E5=86=8Ddump=E5=8E=9F=E5=A7=8B=E7=BB=93?= =?UTF-8?q?=E6=9E=84?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../com/par/core/adapter/AdapterEngine.java | 285 ++++++++---------- 1 file changed, 129 insertions(+), 156 deletions(-) diff --git a/par-core/src/main/java/com/par/core/adapter/AdapterEngine.java b/par-core/src/main/java/com/par/core/adapter/AdapterEngine.java index ba5c758..30d7c93 100644 --- a/par-core/src/main/java/com/par/core/adapter/AdapterEngine.java +++ b/par-core/src/main/java/com/par/core/adapter/AdapterEngine.java @@ -11,10 +11,9 @@ import org.springframework.stereotype.Service; import java.util.*; import java.util.regex.Matcher; import java.util.regex.Pattern; -import java.util.stream.Collectors; /** - * 适配器解析引擎:根据配置文件的 CSS 选择器从 HTML 中提取数据 + * 适配器解析引擎:根据配置文件的 CSS 选择器(HTML)或点分路径(JSON)提取数据 */ @Slf4j @Service @@ -22,9 +21,6 @@ public class AdapterEngine { private static final ObjectMapper objectMapper = new ObjectMapper(); - /** - * 解析页面内容(兼容 HTML 和 JSON) - */ public Map extract(AdapterConfig config, String content, String contentType, String accessUrl) { if ("json".equalsIgnoreCase(contentType)) { return extractJson(config, content, accessUrl); @@ -32,68 +28,75 @@ public class AdapterEngine { return extractHtml(config, content, accessUrl); } - /** 保留旧签名兼容单条解析 */ public Map extract(AdapterConfig config, String content, String accessUrl) { return extract(config, content, "html", accessUrl); } - /** - * 解析 JSON 内容:直接展示 JSON 结构 - */ + // ==================== JSON ==================== + private Map extractJson(AdapterConfig config, String jsonStr, String accessUrl) { Map result = new LinkedHashMap<>(); try { Object parsed = objectMapper.readValue(jsonStr, Object.class); result.put("contentType", "json"); - result.put("matched", true); - result.put("pageName", "JSON响应"); - List> rows = new ArrayList<>(); - - if (parsed instanceof Map) { - @SuppressWarnings("unchecked") - Map map = (Map) parsed; - result.put("message", "JSON 对象,共 " + map.size() + " 个顶层字段"); - Map row = new LinkedHashMap<>(); - for (String key : map.keySet()) { - Object val = map.get(key); - row.put(key, toDisplayString(val)); - } - rows.add(row); - result.put("rowCount", 1); - result.put("fieldCount", map.size()); - } else if (parsed instanceof List) { - @SuppressWarnings("unchecked") - List list = (List) parsed; - result.put("message", "JSON 数组,共 " + list.size() + " 条记录"); - int max = Math.min(list.size(), 20); - for (int i = 0; i < max; i++) { - Object item = list.get(i); - Map row = new LinkedHashMap<>(); - if (item instanceof Map) { - @SuppressWarnings("unchecked") - Map m = (Map) item; - for (String key : m.keySet()) { - row.put(key, toDisplayString(m.get(key))); - } - } else { - row.put("value", item); - } - rows.add(row); - } - result.put("rowCount", list.size()); - result.put("fieldCount", rows.isEmpty() ? 0 : rows.get(0).size()); - } else { - // 简单值(string/number/boolean/null) - result.put("message", "JSON 简单值"); - Map row = new LinkedHashMap<>(); - row.put("value", parsed); - rows.add(row); - result.put("rowCount", 1); - result.put("fieldCount", 1); + Map.Entry matched = matchPage(config.getPages(), accessUrl); + if (matched == null) { + result.put("matched", false); + result.put("message", "未匹配到页面规则"); + result.put("availablePages", new ArrayList<>(config.getPages() != null ? config.getPages().keySet() : Set.of())); + return result; } + PageConfig page = matched.getValue(); + result.put("matched", true); + result.put("pageName", matched.getKey()); + + if (page.getFields() == null || page.getFields().isEmpty()) { + result.put("message", "页面规则未定义字段"); + result.put("extracted", List.of()); + result.put("rowCount", 0); + result.put("fieldCount", 0); + return result; + } + + // 列表容器 + List items = new ArrayList<>(); + if (page.getListContainer() != null && !page.getListContainer().isBlank()) { + Object container = resolveJsonPath(parsed, page.getListContainer()); + if (container instanceof List) items = (List) container; + else items.add(parsed); + } else { + items.add(parsed); + } + + // 逐行提取字段 + List> rows = new ArrayList<>(); + int max = Math.min(items.size(), 50); + for (int i = 0; i < max; i++) { + Map row = new LinkedHashMap<>(); + boolean hasData = false; + for (Map.Entry field : page.getFields().entrySet()) { + String fieldName = field.getKey(); + FieldExtractor fe = field.getValue(); + if (fe.getSelector() == null) continue; + Object val = resolveJsonPath(items.get(i), fe.getSelector()); + if (val != null) hasData = true; + if (fe.getOptional() != null && fe.getOptional() && val == null) continue; + if (val instanceof String s) { + row.put(fieldName, applyTransform(s, fe.getTransform())); + } else if (val != null) { + row.put(fieldName, toDisplayString(val)); + } else { + row.put(fieldName, null); + } + } + if (hasData) rows.add(row); + } result.put("extracted", rows); + result.put("rowCount", rows.size()); + result.put("fieldCount", page.getFields().size()); + result.put("message", rows.isEmpty() ? "selector 未匹配到数据" : "共 " + rows.size() + " 行"); } catch (Exception e) { result.put("contentType", "json"); result.put("matched", false); @@ -102,9 +105,40 @@ public class AdapterEngine { return result; } - /** - * 解析 HTML 内容:Jsoup + CSS 选择器 - */ + @SuppressWarnings("unchecked") + private Object resolveJsonPath(Object root, String path) { + if (root == null || path == null || path.isBlank()) return null; + Object current = root; + for (String part : path.split("\\.")) { + if (current == null) return null; + String key = part; + int bracket = key.indexOf('['); + int arrIdx = -1; + if (bracket > 0) { + try { + arrIdx = Integer.parseInt(key.substring(bracket + 1, key.indexOf(']'))); + key = key.substring(0, bracket); + } catch (Exception ignored) {} + } + if (current instanceof Map) { + current = ((Map) current).get(key); + } else if (current instanceof List && !key.isEmpty()) { + try { + int li = Integer.parseInt(key); + current = ((List) current).size() > li ? ((List) current).get(li) : null; + } catch (NumberFormatException e) { return null; } + } else { + return null; + } + if (arrIdx >= 0 && current instanceof List) { + current = ((List) current).size() > arrIdx ? ((List) current).get(arrIdx) : null; + } + } + return current; + } + + // ==================== HTML ==================== + private Map extractHtml(AdapterConfig config, String html, String accessUrl) { Map result = new LinkedHashMap<>(); Document doc = Jsoup.parse(html); @@ -115,66 +149,43 @@ public class AdapterEngine { result.put("contentType", "html"); result.put("message", "未匹配到页面规则"); result.put("availablePages", new ArrayList<>(config.getPages() != null ? config.getPages().keySet() : Set.of())); - result.put("accessUrl", accessUrl); return result; } - String pageName = matched.getKey(); PageConfig page = matched.getValue(); result.put("matched", true); result.put("contentType", "html"); - result.put("pageName", pageName); - result.put("pageUrl", accessUrl); + result.put("pageName", matched.getKey()); + result.put("fieldCount", page.getFields() != null ? page.getFields().size() : 0); List> extracted = extractRows(doc, page); result.put("extracted", extracted); result.put("rowCount", extracted.size()); - result.put("fieldCount", page.getFields() != null ? page.getFields().size() : 0); - return result; } - /** - * 匹配 URL 对应的页面规则 - */ + // ==================== 页面匹配 ==================== + private Map.Entry matchPage(Map pages, String accessUrl) { if (pages == null || pages.isEmpty() || accessUrl == null) return null; - for (Map.Entry entry : pages.entrySet()) { PageConfig page = entry.getValue(); - - // 精确 URL 匹配 if (page.getUrl() != null) { - String normalizedPageUrl = page.getUrl().replaceAll("\\?.+", ""); - String normalizedAccessUrl = accessUrl.replaceAll("\\?.+", ""); - if (normalizedAccessUrl.contains(normalizedPageUrl) - || normalizedPageUrl.contains(normalizedAccessUrl)) { - return entry; - } + String a = page.getUrl().replaceAll("\\?.+", ""); + String b = accessUrl.replaceAll("\\?.+", ""); + if (b.contains(a) || a.contains(b)) return entry; } - - // URL 正则匹配 if (page.getUrlPattern() != null) { - try { - if (Pattern.compile(page.getUrlPattern()).matcher(accessUrl).find()) { - return entry; - } - } catch (Exception ignored) {} - } - - // 模糊匹配:URL 路径中包含页面 key - if (accessUrl.contains("/" + entry.getKey())) { - return entry; + try { if (Pattern.compile(page.getUrlPattern()).matcher(accessUrl).find()) return entry; } + catch (Exception ignored) {} } + if (accessUrl.contains("/" + entry.getKey())) return entry; } - return null; } - /** - * 从文档中提取所有行的数据 - */ - @SuppressWarnings("unchecked") + // ==================== HTML 提取 ==================== + private List> extractRows(Document doc, PageConfig page) { if (page.getFields() == null || page.getFields().isEmpty()) return List.of(); @@ -182,101 +193,64 @@ public class AdapterEngine { if (page.getListContainer() != null && !page.getListContainer().isBlank()) { rows = doc.select(page.getListContainer()); } else { - // 无列表容器 → 整页提取一行 - rows = new Elements(); - rows.add(doc); + rows = new Elements(); rows.add(doc); } List> results = new ArrayList<>(); - int skip = Math.max(0, page.getSkipRows()); - for (int i = skip; i < rows.size(); i++) { + for (int i = Math.max(0, page.getSkipRows()); i < rows.size(); i++) { Element row = rows.get(i); Map rowData = new LinkedHashMap<>(); boolean hasData = false; - - for (Map.Entry field : page.getFields().entrySet()) { - String fieldName = field.getKey(); - FieldExtractor fe = field.getValue(); + for (Map.Entry f : page.getFields().entrySet()) { + FieldExtractor fe = f.getValue(); if (fe.getSelector() == null) continue; - Element el = row.selectFirst(fe.getSelector()); if (el == null) { - if (fe.getOptional() == null || !fe.getOptional()) { - rowData.put(fieldName, null); - } + if (fe.getOptional() == null || !fe.getOptional()) rowData.put(f.getKey(), null); continue; } - - String raw; - String type = fe.getType() != null ? fe.getType() : "text"; - switch (type) { - case "attribute": - raw = fe.getAttribute() != null ? el.attr(fe.getAttribute()) : el.text(); - break; - case "html": - raw = el.html(); - break; - default: - raw = el.text(); - } - - Object value = applyTransform(raw, fe.getTransform()); - rowData.put(fieldName, value); + String raw = switch (fe.getType() != null ? fe.getType() : "text") { + case "attribute" -> fe.getAttribute() != null ? el.attr(fe.getAttribute()) : el.text(); + case "html" -> el.html(); + default -> el.text(); + }; + rowData.put(f.getKey(), applyTransform(raw, fe.getTransform())); hasData = true; } - if (hasData) results.add(rowData); } - return results; } - /** - * 应用内置转换器 - */ + // ==================== 转换器 ==================== + private Object applyTransform(String raw, String transform) { if (raw == null || transform == null) return raw; - try { return switch (transform.toLowerCase()) { - case "int", "parseint" -> { - String clean = raw.replaceAll("[^\\d-]", ""); - yield clean.isEmpty() ? null : Integer.parseInt(clean); - } - case "long" -> { - String clean = raw.replaceAll("[^\\d-]", ""); - yield clean.isEmpty() ? null : Long.parseLong(clean); - } - case "double", "float" -> { - String clean = raw.replaceAll("[^\\d.-]", ""); - yield clean.isEmpty() ? null : Double.parseDouble(clean); - } + case "int", "parseint" -> { String c = raw.replaceAll("[^\\d-]", ""); yield c.isEmpty() ? null : Integer.parseInt(c); } + case "long" -> { String c = raw.replaceAll("[^\\d-]", ""); yield c.isEmpty() ? null : Long.parseLong(c); } + case "double", "float" -> { String c = raw.replaceAll("[^\\d.-]", ""); yield c.isEmpty() ? null : Double.parseDouble(c); } case "filesize" -> parseFileSize(raw); case "boolean", "bool" -> parseBoolean(raw); default -> raw; }; - } catch (Exception e) { - return raw; - } + } catch (Exception e) { return raw; } } private Long parseFileSize(String raw) { if (raw == null) return null; - Matcher m = Pattern.compile("([\\d.]+)\\s*(GB|MB|KB|TB|B|GiB|MiB|KiB|TiB)", - Pattern.CASE_INSENSITIVE).matcher(raw); + Matcher m = Pattern.compile("([\\d.]+)\\s*(GB|MB|KB|TB|B|GiB|MiB|KiB|TiB)", Pattern.CASE_INSENSITIVE).matcher(raw); if (!m.find()) { - // try just number - String num = raw.replaceAll("[^\\d.]", ""); - if (!num.isEmpty()) return (long) Double.parseDouble(num); - return null; + String n = raw.replaceAll("[^\\d.]", ""); + return n.isEmpty() ? null : (long) Double.parseDouble(n); } - double val = Double.parseDouble(m.group(1)); - String unit = m.group(2).toUpperCase(); - return (long) (val * switch (unit) { - case "TB", "TIB" -> 1024L * 1024 * 1024 * 1024; - case "GB", "GIB" -> 1024L * 1024 * 1024; - case "MB", "MIB" -> 1024L * 1024; - case "KB", "KIB" -> 1024L; + double v = Double.parseDouble(m.group(1)); + return (long) (v * switch (m.group(2).toUpperCase()) { + case "TB","TIB" -> 1024L*1024*1024*1024; + case "GB","GIB" -> 1024L*1024*1024; + case "MB","MIB" -> 1024L*1024; + case "KB","KIB" -> 1024L; default -> 1L; }); } @@ -289,7 +263,6 @@ public class AdapterEngine { return null; } - /** 值转展示字符串(嵌套对象序列化,null → "null") */ private String toDisplayString(Object val) { if (val == null) return "null"; if (val instanceof String s) return s;