refactor: JSON解析改为匹配页面规则+selector路径提取字段,不再dump原始结构

This commit is contained in:
mediabot-pt
2026-07-01 10:13:33 +08:00
parent 82a234a6b2
commit df6928d0cb

View File

@@ -11,10 +11,9 @@ import org.springframework.stereotype.Service;
import java.util.*;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
import java.util.stream.Collectors;
/**
* 适配器解析引擎:根据配置文件的 CSS 选择器从 HTML 中提取数据
* 适配器解析引擎:根据配置文件的 CSS 选择器(HTML)或点分路径(JSON)提取数据
*/
@Slf4j
@Service
@@ -22,9 +21,6 @@ public class AdapterEngine {
private static final ObjectMapper objectMapper = new ObjectMapper();
/**
* 解析页面内容(兼容 HTML 和 JSON)
*/
public Map<String, Object> extract(AdapterConfig config, String content, String contentType, String accessUrl) {
if ("json".equalsIgnoreCase(contentType)) {
return extractJson(config, content, accessUrl);
@@ -32,68 +28,75 @@ public class AdapterEngine {
return extractHtml(config, content, accessUrl);
}
/** 保留旧签名兼容单条解析 */
public Map<String, Object> extract(AdapterConfig config, String content, String accessUrl) {
return extract(config, content, "html", accessUrl);
}
/**
* 解析 JSON 内容:直接展示 JSON 结构
*/
// ==================== JSON ====================
private Map<String, Object> extractJson(AdapterConfig config, String jsonStr, String accessUrl) {
Map<String, Object> result = new LinkedHashMap<>();
try {
Object parsed = objectMapper.readValue(jsonStr, Object.class);
result.put("contentType", "json");
result.put("matched", true);
result.put("pageName", "JSON响应");
List<Map<String, Object>> rows = new ArrayList<>();
if (parsed instanceof Map) {
@SuppressWarnings("unchecked")
Map<String, Object> map = (Map<String, Object>) parsed;
result.put("message", "JSON 对象,共 " + map.size() + " 个顶层字段");
Map<String, Object> row = new LinkedHashMap<>();
for (String key : map.keySet()) {
Object val = map.get(key);
row.put(key, toDisplayString(val));
}
rows.add(row);
result.put("rowCount", 1);
result.put("fieldCount", map.size());
} else if (parsed instanceof List) {
@SuppressWarnings("unchecked")
List<Object> list = (List<Object>) parsed;
result.put("message", "JSON 数组,共 " + list.size() + " 条记录");
int max = Math.min(list.size(), 20);
for (int i = 0; i < max; i++) {
Object item = list.get(i);
Map<String, Object> row = new LinkedHashMap<>();
if (item instanceof Map) {
@SuppressWarnings("unchecked")
Map<String, Object> m = (Map<String, Object>) item;
for (String key : m.keySet()) {
row.put(key, toDisplayString(m.get(key)));
}
} else {
row.put("value", item);
}
rows.add(row);
}
result.put("rowCount", list.size());
result.put("fieldCount", rows.isEmpty() ? 0 : rows.get(0).size());
} else {
// 简单值(string/number/boolean/null)
result.put("message", "JSON 简单值");
Map<String, Object> row = new LinkedHashMap<>();
row.put("value", parsed);
rows.add(row);
result.put("rowCount", 1);
result.put("fieldCount", 1);
Map.Entry<String, PageConfig> matched = matchPage(config.getPages(), accessUrl);
if (matched == null) {
result.put("matched", false);
result.put("message", "未匹配到页面规则");
result.put("availablePages", new ArrayList<>(config.getPages() != null ? config.getPages().keySet() : Set.of()));
return result;
}
PageConfig page = matched.getValue();
result.put("matched", true);
result.put("pageName", matched.getKey());
if (page.getFields() == null || page.getFields().isEmpty()) {
result.put("message", "页面规则未定义字段");
result.put("extracted", List.of());
result.put("rowCount", 0);
result.put("fieldCount", 0);
return result;
}
// 列表容器
List<Object> items = new ArrayList<>();
if (page.getListContainer() != null && !page.getListContainer().isBlank()) {
Object container = resolveJsonPath(parsed, page.getListContainer());
if (container instanceof List) items = (List<Object>) container;
else items.add(parsed);
} else {
items.add(parsed);
}
// 逐行提取字段
List<Map<String, Object>> rows = new ArrayList<>();
int max = Math.min(items.size(), 50);
for (int i = 0; i < max; i++) {
Map<String, Object> row = new LinkedHashMap<>();
boolean hasData = false;
for (Map.Entry<String, FieldExtractor> field : page.getFields().entrySet()) {
String fieldName = field.getKey();
FieldExtractor fe = field.getValue();
if (fe.getSelector() == null) continue;
Object val = resolveJsonPath(items.get(i), fe.getSelector());
if (val != null) hasData = true;
if (fe.getOptional() != null && fe.getOptional() && val == null) continue;
if (val instanceof String s) {
row.put(fieldName, applyTransform(s, fe.getTransform()));
} else if (val != null) {
row.put(fieldName, toDisplayString(val));
} else {
row.put(fieldName, null);
}
}
if (hasData) rows.add(row);
}
result.put("extracted", rows);
result.put("rowCount", rows.size());
result.put("fieldCount", page.getFields().size());
result.put("message", rows.isEmpty() ? "selector 未匹配到数据" : "共 " + rows.size() + " 行");
} catch (Exception e) {
result.put("contentType", "json");
result.put("matched", false);
@@ -102,9 +105,40 @@ public class AdapterEngine {
return result;
}
/**
* 解析 HTML 内容:Jsoup + CSS 选择器
*/
@SuppressWarnings("unchecked")
private Object resolveJsonPath(Object root, String path) {
if (root == null || path == null || path.isBlank()) return null;
Object current = root;
for (String part : path.split("\\.")) {
if (current == null) return null;
String key = part;
int bracket = key.indexOf('[');
int arrIdx = -1;
if (bracket > 0) {
try {
arrIdx = Integer.parseInt(key.substring(bracket + 1, key.indexOf(']')));
key = key.substring(0, bracket);
} catch (Exception ignored) {}
}
if (current instanceof Map) {
current = ((Map<String, Object>) current).get(key);
} else if (current instanceof List && !key.isEmpty()) {
try {
int li = Integer.parseInt(key);
current = ((List<Object>) current).size() > li ? ((List<Object>) current).get(li) : null;
} catch (NumberFormatException e) { return null; }
} else {
return null;
}
if (arrIdx >= 0 && current instanceof List) {
current = ((List<Object>) current).size() > arrIdx ? ((List<Object>) current).get(arrIdx) : null;
}
}
return current;
}
// ==================== HTML ====================
private Map<String, Object> extractHtml(AdapterConfig config, String html, String accessUrl) {
Map<String, Object> result = new LinkedHashMap<>();
Document doc = Jsoup.parse(html);
@@ -115,66 +149,43 @@ public class AdapterEngine {
result.put("contentType", "html");
result.put("message", "未匹配到页面规则");
result.put("availablePages", new ArrayList<>(config.getPages() != null ? config.getPages().keySet() : Set.of()));
result.put("accessUrl", accessUrl);
return result;
}
String pageName = matched.getKey();
PageConfig page = matched.getValue();
result.put("matched", true);
result.put("contentType", "html");
result.put("pageName", pageName);
result.put("pageUrl", accessUrl);
result.put("pageName", matched.getKey());
result.put("fieldCount", page.getFields() != null ? page.getFields().size() : 0);
List<Map<String, Object>> extracted = extractRows(doc, page);
result.put("extracted", extracted);
result.put("rowCount", extracted.size());
result.put("fieldCount", page.getFields() != null ? page.getFields().size() : 0);
return result;
}
/**
* 匹配 URL 对应的页面规则
*/
// ==================== 页面匹配 ====================
private Map.Entry<String, PageConfig> matchPage(Map<String, PageConfig> pages, String accessUrl) {
if (pages == null || pages.isEmpty() || accessUrl == null) return null;
for (Map.Entry<String, PageConfig> entry : pages.entrySet()) {
PageConfig page = entry.getValue();
// 精确 URL 匹配
if (page.getUrl() != null) {
String normalizedPageUrl = page.getUrl().replaceAll("\\?.+", "");
String normalizedAccessUrl = accessUrl.replaceAll("\\?.+", "");
if (normalizedAccessUrl.contains(normalizedPageUrl)
|| normalizedPageUrl.contains(normalizedAccessUrl)) {
return entry;
}
String a = page.getUrl().replaceAll("\\?.+", "");
String b = accessUrl.replaceAll("\\?.+", "");
if (b.contains(a) || a.contains(b)) return entry;
}
// URL 正则匹配
if (page.getUrlPattern() != null) {
try {
if (Pattern.compile(page.getUrlPattern()).matcher(accessUrl).find()) {
return entry;
}
} catch (Exception ignored) {}
}
// 模糊匹配:URL 路径中包含页面 key
if (accessUrl.contains("/" + entry.getKey())) {
return entry;
try { if (Pattern.compile(page.getUrlPattern()).matcher(accessUrl).find()) return entry; }
catch (Exception ignored) {}
}
if (accessUrl.contains("/" + entry.getKey())) return entry;
}
return null;
}
/**
* 从文档中提取所有行的数据
*/
@SuppressWarnings("unchecked")
// ==================== HTML 提取 ====================
private List<Map<String, Object>> extractRows(Document doc, PageConfig page) {
if (page.getFields() == null || page.getFields().isEmpty()) return List.of();
@@ -182,101 +193,64 @@ public class AdapterEngine {
if (page.getListContainer() != null && !page.getListContainer().isBlank()) {
rows = doc.select(page.getListContainer());
} else {
// 无列表容器 → 整页提取一行
rows = new Elements();
rows.add(doc);
rows = new Elements(); rows.add(doc);
}
List<Map<String, Object>> results = new ArrayList<>();
int skip = Math.max(0, page.getSkipRows());
for (int i = skip; i < rows.size(); i++) {
for (int i = Math.max(0, page.getSkipRows()); i < rows.size(); i++) {
Element row = rows.get(i);
Map<String, Object> rowData = new LinkedHashMap<>();
boolean hasData = false;
for (Map.Entry<String, FieldExtractor> field : page.getFields().entrySet()) {
String fieldName = field.getKey();
FieldExtractor fe = field.getValue();
for (Map.Entry<String, FieldExtractor> f : page.getFields().entrySet()) {
FieldExtractor fe = f.getValue();
if (fe.getSelector() == null) continue;
Element el = row.selectFirst(fe.getSelector());
if (el == null) {
if (fe.getOptional() == null || !fe.getOptional()) {
rowData.put(fieldName, null);
}
if (fe.getOptional() == null || !fe.getOptional()) rowData.put(f.getKey(), null);
continue;
}
String raw;
String type = fe.getType() != null ? fe.getType() : "text";
switch (type) {
case "attribute":
raw = fe.getAttribute() != null ? el.attr(fe.getAttribute()) : el.text();
break;
case "html":
raw = el.html();
break;
default:
raw = el.text();
}
Object value = applyTransform(raw, fe.getTransform());
rowData.put(fieldName, value);
String raw = switch (fe.getType() != null ? fe.getType() : "text") {
case "attribute" -> fe.getAttribute() != null ? el.attr(fe.getAttribute()) : el.text();
case "html" -> el.html();
default -> el.text();
};
rowData.put(f.getKey(), applyTransform(raw, fe.getTransform()));
hasData = true;
}
if (hasData) results.add(rowData);
}
return results;
}
/**
* 应用内置转换器
*/
// ==================== 转换器 ====================
private Object applyTransform(String raw, String transform) {
if (raw == null || transform == null) return raw;
try {
return switch (transform.toLowerCase()) {
case "int", "parseint" -> {
String clean = raw.replaceAll("[^\\d-]", "");
yield clean.isEmpty() ? null : Integer.parseInt(clean);
}
case "long" -> {
String clean = raw.replaceAll("[^\\d-]", "");
yield clean.isEmpty() ? null : Long.parseLong(clean);
}
case "double", "float" -> {
String clean = raw.replaceAll("[^\\d.-]", "");
yield clean.isEmpty() ? null : Double.parseDouble(clean);
}
case "int", "parseint" -> { String c = raw.replaceAll("[^\\d-]", ""); yield c.isEmpty() ? null : Integer.parseInt(c); }
case "long" -> { String c = raw.replaceAll("[^\\d-]", ""); yield c.isEmpty() ? null : Long.parseLong(c); }
case "double", "float" -> { String c = raw.replaceAll("[^\\d.-]", ""); yield c.isEmpty() ? null : Double.parseDouble(c); }
case "filesize" -> parseFileSize(raw);
case "boolean", "bool" -> parseBoolean(raw);
default -> raw;
};
} catch (Exception e) {
return raw;
}
} catch (Exception e) { return raw; }
}
private Long parseFileSize(String raw) {
if (raw == null) return null;
Matcher m = Pattern.compile("([\\d.]+)\\s*(GB|MB|KB|TB|B|GiB|MiB|KiB|TiB)",
Pattern.CASE_INSENSITIVE).matcher(raw);
Matcher m = Pattern.compile("([\\d.]+)\\s*(GB|MB|KB|TB|B|GiB|MiB|KiB|TiB)", Pattern.CASE_INSENSITIVE).matcher(raw);
if (!m.find()) {
// try just number
String num = raw.replaceAll("[^\\d.]", "");
if (!num.isEmpty()) return (long) Double.parseDouble(num);
return null;
String n = raw.replaceAll("[^\\d.]", "");
return n.isEmpty() ? null : (long) Double.parseDouble(n);
}
double val = Double.parseDouble(m.group(1));
String unit = m.group(2).toUpperCase();
return (long) (val * switch (unit) {
case "TB", "TIB" -> 1024L * 1024 * 1024 * 1024;
case "GB", "GIB" -> 1024L * 1024 * 1024;
case "MB", "MIB" -> 1024L * 1024;
case "KB", "KIB" -> 1024L;
double v = Double.parseDouble(m.group(1));
return (long) (v * switch (m.group(2).toUpperCase()) {
case "TB","TIB" -> 1024L*1024*1024*1024;
case "GB","GIB" -> 1024L*1024*1024;
case "MB","MIB" -> 1024L*1024;
case "KB","KIB" -> 1024L;
default -> 1L;
});
}
@@ -289,7 +263,6 @@ public class AdapterEngine {
return null;
}
/** 值转展示字符串(嵌套对象序列化,null → "null") */
private String toDisplayString(Object val) {
if (val == null) return "null";
if (val instanceof String s) return s;