refactor: JSON解析改为匹配页面规则+selector路径提取字段,不再dump原始结构
This commit is contained in:
@@ -11,10 +11,9 @@ import org.springframework.stereotype.Service;
|
||||
import java.util.*;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* 适配器解析引擎:根据配置文件的 CSS 选择器从 HTML 中提取数据
|
||||
* 适配器解析引擎:根据配置文件的 CSS 选择器(HTML)或点分路径(JSON)提取数据
|
||||
*/
|
||||
@Slf4j
|
||||
@Service
|
||||
@@ -22,9 +21,6 @@ public class AdapterEngine {
|
||||
|
||||
private static final ObjectMapper objectMapper = new ObjectMapper();
|
||||
|
||||
/**
|
||||
* 解析页面内容(兼容 HTML 和 JSON)
|
||||
*/
|
||||
public Map<String, Object> extract(AdapterConfig config, String content, String contentType, String accessUrl) {
|
||||
if ("json".equalsIgnoreCase(contentType)) {
|
||||
return extractJson(config, content, accessUrl);
|
||||
@@ -32,68 +28,75 @@ public class AdapterEngine {
|
||||
return extractHtml(config, content, accessUrl);
|
||||
}
|
||||
|
||||
/** 保留旧签名兼容单条解析 */
|
||||
public Map<String, Object> extract(AdapterConfig config, String content, String accessUrl) {
|
||||
return extract(config, content, "html", accessUrl);
|
||||
}
|
||||
|
||||
/**
|
||||
* 解析 JSON 内容:直接展示 JSON 结构
|
||||
*/
|
||||
// ==================== JSON ====================
|
||||
|
||||
private Map<String, Object> extractJson(AdapterConfig config, String jsonStr, String accessUrl) {
|
||||
Map<String, Object> result = new LinkedHashMap<>();
|
||||
try {
|
||||
Object parsed = objectMapper.readValue(jsonStr, Object.class);
|
||||
result.put("contentType", "json");
|
||||
result.put("matched", true);
|
||||
result.put("pageName", "JSON响应");
|
||||
|
||||
List<Map<String, Object>> rows = new ArrayList<>();
|
||||
|
||||
if (parsed instanceof Map) {
|
||||
@SuppressWarnings("unchecked")
|
||||
Map<String, Object> map = (Map<String, Object>) parsed;
|
||||
result.put("message", "JSON 对象,共 " + map.size() + " 个顶层字段");
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
for (String key : map.keySet()) {
|
||||
Object val = map.get(key);
|
||||
row.put(key, toDisplayString(val));
|
||||
}
|
||||
rows.add(row);
|
||||
result.put("rowCount", 1);
|
||||
result.put("fieldCount", map.size());
|
||||
} else if (parsed instanceof List) {
|
||||
@SuppressWarnings("unchecked")
|
||||
List<Object> list = (List<Object>) parsed;
|
||||
result.put("message", "JSON 数组,共 " + list.size() + " 条记录");
|
||||
int max = Math.min(list.size(), 20);
|
||||
for (int i = 0; i < max; i++) {
|
||||
Object item = list.get(i);
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
if (item instanceof Map) {
|
||||
@SuppressWarnings("unchecked")
|
||||
Map<String, Object> m = (Map<String, Object>) item;
|
||||
for (String key : m.keySet()) {
|
||||
row.put(key, toDisplayString(m.get(key)));
|
||||
}
|
||||
} else {
|
||||
row.put("value", item);
|
||||
}
|
||||
rows.add(row);
|
||||
}
|
||||
result.put("rowCount", list.size());
|
||||
result.put("fieldCount", rows.isEmpty() ? 0 : rows.get(0).size());
|
||||
} else {
|
||||
// 简单值(string/number/boolean/null)
|
||||
result.put("message", "JSON 简单值");
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
row.put("value", parsed);
|
||||
rows.add(row);
|
||||
result.put("rowCount", 1);
|
||||
result.put("fieldCount", 1);
|
||||
Map.Entry<String, PageConfig> matched = matchPage(config.getPages(), accessUrl);
|
||||
if (matched == null) {
|
||||
result.put("matched", false);
|
||||
result.put("message", "未匹配到页面规则");
|
||||
result.put("availablePages", new ArrayList<>(config.getPages() != null ? config.getPages().keySet() : Set.of()));
|
||||
return result;
|
||||
}
|
||||
|
||||
PageConfig page = matched.getValue();
|
||||
result.put("matched", true);
|
||||
result.put("pageName", matched.getKey());
|
||||
|
||||
if (page.getFields() == null || page.getFields().isEmpty()) {
|
||||
result.put("message", "页面规则未定义字段");
|
||||
result.put("extracted", List.of());
|
||||
result.put("rowCount", 0);
|
||||
result.put("fieldCount", 0);
|
||||
return result;
|
||||
}
|
||||
|
||||
// 列表容器
|
||||
List<Object> items = new ArrayList<>();
|
||||
if (page.getListContainer() != null && !page.getListContainer().isBlank()) {
|
||||
Object container = resolveJsonPath(parsed, page.getListContainer());
|
||||
if (container instanceof List) items = (List<Object>) container;
|
||||
else items.add(parsed);
|
||||
} else {
|
||||
items.add(parsed);
|
||||
}
|
||||
|
||||
// 逐行提取字段
|
||||
List<Map<String, Object>> rows = new ArrayList<>();
|
||||
int max = Math.min(items.size(), 50);
|
||||
for (int i = 0; i < max; i++) {
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
boolean hasData = false;
|
||||
for (Map.Entry<String, FieldExtractor> field : page.getFields().entrySet()) {
|
||||
String fieldName = field.getKey();
|
||||
FieldExtractor fe = field.getValue();
|
||||
if (fe.getSelector() == null) continue;
|
||||
Object val = resolveJsonPath(items.get(i), fe.getSelector());
|
||||
if (val != null) hasData = true;
|
||||
if (fe.getOptional() != null && fe.getOptional() && val == null) continue;
|
||||
if (val instanceof String s) {
|
||||
row.put(fieldName, applyTransform(s, fe.getTransform()));
|
||||
} else if (val != null) {
|
||||
row.put(fieldName, toDisplayString(val));
|
||||
} else {
|
||||
row.put(fieldName, null);
|
||||
}
|
||||
}
|
||||
if (hasData) rows.add(row);
|
||||
}
|
||||
result.put("extracted", rows);
|
||||
result.put("rowCount", rows.size());
|
||||
result.put("fieldCount", page.getFields().size());
|
||||
result.put("message", rows.isEmpty() ? "selector 未匹配到数据" : "共 " + rows.size() + " 行");
|
||||
} catch (Exception e) {
|
||||
result.put("contentType", "json");
|
||||
result.put("matched", false);
|
||||
@@ -102,9 +105,40 @@ public class AdapterEngine {
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* 解析 HTML 内容:Jsoup + CSS 选择器
|
||||
*/
|
||||
@SuppressWarnings("unchecked")
|
||||
private Object resolveJsonPath(Object root, String path) {
|
||||
if (root == null || path == null || path.isBlank()) return null;
|
||||
Object current = root;
|
||||
for (String part : path.split("\\.")) {
|
||||
if (current == null) return null;
|
||||
String key = part;
|
||||
int bracket = key.indexOf('[');
|
||||
int arrIdx = -1;
|
||||
if (bracket > 0) {
|
||||
try {
|
||||
arrIdx = Integer.parseInt(key.substring(bracket + 1, key.indexOf(']')));
|
||||
key = key.substring(0, bracket);
|
||||
} catch (Exception ignored) {}
|
||||
}
|
||||
if (current instanceof Map) {
|
||||
current = ((Map<String, Object>) current).get(key);
|
||||
} else if (current instanceof List && !key.isEmpty()) {
|
||||
try {
|
||||
int li = Integer.parseInt(key);
|
||||
current = ((List<Object>) current).size() > li ? ((List<Object>) current).get(li) : null;
|
||||
} catch (NumberFormatException e) { return null; }
|
||||
} else {
|
||||
return null;
|
||||
}
|
||||
if (arrIdx >= 0 && current instanceof List) {
|
||||
current = ((List<Object>) current).size() > arrIdx ? ((List<Object>) current).get(arrIdx) : null;
|
||||
}
|
||||
}
|
||||
return current;
|
||||
}
|
||||
|
||||
// ==================== HTML ====================
|
||||
|
||||
private Map<String, Object> extractHtml(AdapterConfig config, String html, String accessUrl) {
|
||||
Map<String, Object> result = new LinkedHashMap<>();
|
||||
Document doc = Jsoup.parse(html);
|
||||
@@ -115,66 +149,43 @@ public class AdapterEngine {
|
||||
result.put("contentType", "html");
|
||||
result.put("message", "未匹配到页面规则");
|
||||
result.put("availablePages", new ArrayList<>(config.getPages() != null ? config.getPages().keySet() : Set.of()));
|
||||
result.put("accessUrl", accessUrl);
|
||||
return result;
|
||||
}
|
||||
|
||||
String pageName = matched.getKey();
|
||||
PageConfig page = matched.getValue();
|
||||
result.put("matched", true);
|
||||
result.put("contentType", "html");
|
||||
result.put("pageName", pageName);
|
||||
result.put("pageUrl", accessUrl);
|
||||
result.put("pageName", matched.getKey());
|
||||
result.put("fieldCount", page.getFields() != null ? page.getFields().size() : 0);
|
||||
|
||||
List<Map<String, Object>> extracted = extractRows(doc, page);
|
||||
result.put("extracted", extracted);
|
||||
result.put("rowCount", extracted.size());
|
||||
result.put("fieldCount", page.getFields() != null ? page.getFields().size() : 0);
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* 匹配 URL 对应的页面规则
|
||||
*/
|
||||
// ==================== 页面匹配 ====================
|
||||
|
||||
private Map.Entry<String, PageConfig> matchPage(Map<String, PageConfig> pages, String accessUrl) {
|
||||
if (pages == null || pages.isEmpty() || accessUrl == null) return null;
|
||||
|
||||
for (Map.Entry<String, PageConfig> entry : pages.entrySet()) {
|
||||
PageConfig page = entry.getValue();
|
||||
|
||||
// 精确 URL 匹配
|
||||
if (page.getUrl() != null) {
|
||||
String normalizedPageUrl = page.getUrl().replaceAll("\\?.+", "");
|
||||
String normalizedAccessUrl = accessUrl.replaceAll("\\?.+", "");
|
||||
if (normalizedAccessUrl.contains(normalizedPageUrl)
|
||||
|| normalizedPageUrl.contains(normalizedAccessUrl)) {
|
||||
return entry;
|
||||
}
|
||||
String a = page.getUrl().replaceAll("\\?.+", "");
|
||||
String b = accessUrl.replaceAll("\\?.+", "");
|
||||
if (b.contains(a) || a.contains(b)) return entry;
|
||||
}
|
||||
|
||||
// URL 正则匹配
|
||||
if (page.getUrlPattern() != null) {
|
||||
try {
|
||||
if (Pattern.compile(page.getUrlPattern()).matcher(accessUrl).find()) {
|
||||
return entry;
|
||||
}
|
||||
} catch (Exception ignored) {}
|
||||
}
|
||||
|
||||
// 模糊匹配:URL 路径中包含页面 key
|
||||
if (accessUrl.contains("/" + entry.getKey())) {
|
||||
return entry;
|
||||
try { if (Pattern.compile(page.getUrlPattern()).matcher(accessUrl).find()) return entry; }
|
||||
catch (Exception ignored) {}
|
||||
}
|
||||
if (accessUrl.contains("/" + entry.getKey())) return entry;
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* 从文档中提取所有行的数据
|
||||
*/
|
||||
@SuppressWarnings("unchecked")
|
||||
// ==================== HTML 提取 ====================
|
||||
|
||||
private List<Map<String, Object>> extractRows(Document doc, PageConfig page) {
|
||||
if (page.getFields() == null || page.getFields().isEmpty()) return List.of();
|
||||
|
||||
@@ -182,101 +193,64 @@ public class AdapterEngine {
|
||||
if (page.getListContainer() != null && !page.getListContainer().isBlank()) {
|
||||
rows = doc.select(page.getListContainer());
|
||||
} else {
|
||||
// 无列表容器 → 整页提取一行
|
||||
rows = new Elements();
|
||||
rows.add(doc);
|
||||
rows = new Elements(); rows.add(doc);
|
||||
}
|
||||
|
||||
List<Map<String, Object>> results = new ArrayList<>();
|
||||
int skip = Math.max(0, page.getSkipRows());
|
||||
for (int i = skip; i < rows.size(); i++) {
|
||||
for (int i = Math.max(0, page.getSkipRows()); i < rows.size(); i++) {
|
||||
Element row = rows.get(i);
|
||||
Map<String, Object> rowData = new LinkedHashMap<>();
|
||||
boolean hasData = false;
|
||||
|
||||
for (Map.Entry<String, FieldExtractor> field : page.getFields().entrySet()) {
|
||||
String fieldName = field.getKey();
|
||||
FieldExtractor fe = field.getValue();
|
||||
for (Map.Entry<String, FieldExtractor> f : page.getFields().entrySet()) {
|
||||
FieldExtractor fe = f.getValue();
|
||||
if (fe.getSelector() == null) continue;
|
||||
|
||||
Element el = row.selectFirst(fe.getSelector());
|
||||
if (el == null) {
|
||||
if (fe.getOptional() == null || !fe.getOptional()) {
|
||||
rowData.put(fieldName, null);
|
||||
}
|
||||
if (fe.getOptional() == null || !fe.getOptional()) rowData.put(f.getKey(), null);
|
||||
continue;
|
||||
}
|
||||
|
||||
String raw;
|
||||
String type = fe.getType() != null ? fe.getType() : "text";
|
||||
switch (type) {
|
||||
case "attribute":
|
||||
raw = fe.getAttribute() != null ? el.attr(fe.getAttribute()) : el.text();
|
||||
break;
|
||||
case "html":
|
||||
raw = el.html();
|
||||
break;
|
||||
default:
|
||||
raw = el.text();
|
||||
}
|
||||
|
||||
Object value = applyTransform(raw, fe.getTransform());
|
||||
rowData.put(fieldName, value);
|
||||
String raw = switch (fe.getType() != null ? fe.getType() : "text") {
|
||||
case "attribute" -> fe.getAttribute() != null ? el.attr(fe.getAttribute()) : el.text();
|
||||
case "html" -> el.html();
|
||||
default -> el.text();
|
||||
};
|
||||
rowData.put(f.getKey(), applyTransform(raw, fe.getTransform()));
|
||||
hasData = true;
|
||||
}
|
||||
|
||||
if (hasData) results.add(rowData);
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
/**
|
||||
* 应用内置转换器
|
||||
*/
|
||||
// ==================== 转换器 ====================
|
||||
|
||||
private Object applyTransform(String raw, String transform) {
|
||||
if (raw == null || transform == null) return raw;
|
||||
|
||||
try {
|
||||
return switch (transform.toLowerCase()) {
|
||||
case "int", "parseint" -> {
|
||||
String clean = raw.replaceAll("[^\\d-]", "");
|
||||
yield clean.isEmpty() ? null : Integer.parseInt(clean);
|
||||
}
|
||||
case "long" -> {
|
||||
String clean = raw.replaceAll("[^\\d-]", "");
|
||||
yield clean.isEmpty() ? null : Long.parseLong(clean);
|
||||
}
|
||||
case "double", "float" -> {
|
||||
String clean = raw.replaceAll("[^\\d.-]", "");
|
||||
yield clean.isEmpty() ? null : Double.parseDouble(clean);
|
||||
}
|
||||
case "int", "parseint" -> { String c = raw.replaceAll("[^\\d-]", ""); yield c.isEmpty() ? null : Integer.parseInt(c); }
|
||||
case "long" -> { String c = raw.replaceAll("[^\\d-]", ""); yield c.isEmpty() ? null : Long.parseLong(c); }
|
||||
case "double", "float" -> { String c = raw.replaceAll("[^\\d.-]", ""); yield c.isEmpty() ? null : Double.parseDouble(c); }
|
||||
case "filesize" -> parseFileSize(raw);
|
||||
case "boolean", "bool" -> parseBoolean(raw);
|
||||
default -> raw;
|
||||
};
|
||||
} catch (Exception e) {
|
||||
return raw;
|
||||
}
|
||||
} catch (Exception e) { return raw; }
|
||||
}
|
||||
|
||||
private Long parseFileSize(String raw) {
|
||||
if (raw == null) return null;
|
||||
Matcher m = Pattern.compile("([\\d.]+)\\s*(GB|MB|KB|TB|B|GiB|MiB|KiB|TiB)",
|
||||
Pattern.CASE_INSENSITIVE).matcher(raw);
|
||||
Matcher m = Pattern.compile("([\\d.]+)\\s*(GB|MB|KB|TB|B|GiB|MiB|KiB|TiB)", Pattern.CASE_INSENSITIVE).matcher(raw);
|
||||
if (!m.find()) {
|
||||
// try just number
|
||||
String num = raw.replaceAll("[^\\d.]", "");
|
||||
if (!num.isEmpty()) return (long) Double.parseDouble(num);
|
||||
return null;
|
||||
String n = raw.replaceAll("[^\\d.]", "");
|
||||
return n.isEmpty() ? null : (long) Double.parseDouble(n);
|
||||
}
|
||||
double val = Double.parseDouble(m.group(1));
|
||||
String unit = m.group(2).toUpperCase();
|
||||
return (long) (val * switch (unit) {
|
||||
case "TB", "TIB" -> 1024L * 1024 * 1024 * 1024;
|
||||
case "GB", "GIB" -> 1024L * 1024 * 1024;
|
||||
case "MB", "MIB" -> 1024L * 1024;
|
||||
case "KB", "KIB" -> 1024L;
|
||||
double v = Double.parseDouble(m.group(1));
|
||||
return (long) (v * switch (m.group(2).toUpperCase()) {
|
||||
case "TB","TIB" -> 1024L*1024*1024*1024;
|
||||
case "GB","GIB" -> 1024L*1024*1024;
|
||||
case "MB","MIB" -> 1024L*1024;
|
||||
case "KB","KIB" -> 1024L;
|
||||
default -> 1L;
|
||||
});
|
||||
}
|
||||
@@ -289,7 +263,6 @@ public class AdapterEngine {
|
||||
return null;
|
||||
}
|
||||
|
||||
/** 值转展示字符串(嵌套对象序列化,null → "null") */
|
||||
private String toDisplayString(Object val) {
|
||||
if (val == null) return "null";
|
||||
if (val instanceof String s) return s;
|
||||
|
||||
Reference in New Issue
Block a user