feat: 适配器解析引擎,用 CSS 选择器从提交的 HTML 中自动提取数据进行校验
This commit is contained in:
@@ -0,0 +1,22 @@
|
||||
package com.par.core.adapter;
|
||||
|
||||
import com.fasterxml.jackson.annotation.JsonIgnoreProperties;
|
||||
import lombok.Data;
|
||||
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* 适配器配置根对象
|
||||
*/
|
||||
@Data
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public class AdapterConfig {
|
||||
/** 站点 ID */
|
||||
private String siteId;
|
||||
/** 站点类型 */
|
||||
private String siteType;
|
||||
/** 域名列表 */
|
||||
private String[] domains;
|
||||
/** 页面解析规则 */
|
||||
private Map<String, PageConfig> pages;
|
||||
}
|
||||
211
par-core/src/main/java/com/par/core/adapter/AdapterEngine.java
Normal file
211
par-core/src/main/java/com/par/core/adapter/AdapterEngine.java
Normal file
@@ -0,0 +1,211 @@
|
||||
package com.par.core.adapter;
|
||||
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.jsoup.Jsoup;
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.jsoup.nodes.Element;
|
||||
import org.jsoup.select.Elements;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import java.util.*;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* 适配器解析引擎:根据配置文件的 CSS 选择器从 HTML 中提取数据
|
||||
*/
|
||||
@Slf4j
|
||||
@Service
|
||||
public class AdapterEngine {
|
||||
|
||||
private static final ObjectMapper objectMapper = new ObjectMapper();
|
||||
|
||||
/**
|
||||
* 解析单个 HTML 页面,匹配配置中的页面规则并提取数据
|
||||
* @return { pageName, pageUrl, extracted: [{field: value, ...}, ...] }
|
||||
*/
|
||||
public Map<String, Object> extract(AdapterConfig config, String html, String accessUrl) {
|
||||
Map<String, Object> result = new LinkedHashMap<>();
|
||||
Document doc = Jsoup.parse(html);
|
||||
|
||||
// 匹配页面
|
||||
Map.Entry<String, PageConfig> matched = matchPage(config.getPages(), accessUrl);
|
||||
if (matched == null) {
|
||||
result.put("matched", false);
|
||||
result.put("message", "未匹配到页面规则");
|
||||
result.put("availablePages", config.getPages().keySet());
|
||||
return result;
|
||||
}
|
||||
|
||||
String pageName = matched.getKey();
|
||||
PageConfig page = matched.getValue();
|
||||
result.put("matched", true);
|
||||
result.put("pageName", pageName);
|
||||
result.put("pageUrl", accessUrl);
|
||||
|
||||
// 提取数据
|
||||
List<Map<String, Object>> extracted = extractRows(doc, page);
|
||||
result.put("extracted", extracted);
|
||||
result.put("rowCount", extracted.size());
|
||||
result.put("fieldCount", page.getFields() != null ? page.getFields().size() : 0);
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* 匹配 URL 对应的页面规则
|
||||
*/
|
||||
private Map.Entry<String, PageConfig> matchPage(Map<String, PageConfig> pages, String accessUrl) {
|
||||
if (pages == null || pages.isEmpty() || accessUrl == null) return null;
|
||||
|
||||
for (Map.Entry<String, PageConfig> entry : pages.entrySet()) {
|
||||
PageConfig page = entry.getValue();
|
||||
|
||||
// 精确 URL 匹配
|
||||
if (page.getUrl() != null) {
|
||||
String normalizedPageUrl = page.getUrl().replaceAll("\\?.+", "");
|
||||
String normalizedAccessUrl = accessUrl.replaceAll("\\?.+", "");
|
||||
if (normalizedAccessUrl.contains(normalizedPageUrl)
|
||||
|| normalizedPageUrl.contains(normalizedAccessUrl)) {
|
||||
return entry;
|
||||
}
|
||||
}
|
||||
|
||||
// URL 正则匹配
|
||||
if (page.getUrlPattern() != null) {
|
||||
try {
|
||||
if (Pattern.compile(page.getUrlPattern()).matcher(accessUrl).find()) {
|
||||
return entry;
|
||||
}
|
||||
} catch (Exception ignored) {}
|
||||
}
|
||||
|
||||
// 模糊匹配:URL 路径中包含页面 key
|
||||
if (accessUrl.contains("/" + entry.getKey())) {
|
||||
return entry;
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* 从文档中提取所有行的数据
|
||||
*/
|
||||
@SuppressWarnings("unchecked")
|
||||
private List<Map<String, Object>> extractRows(Document doc, PageConfig page) {
|
||||
if (page.getFields() == null || page.getFields().isEmpty()) return List.of();
|
||||
|
||||
Elements rows;
|
||||
if (page.getListContainer() != null && !page.getListContainer().isBlank()) {
|
||||
rows = doc.select(page.getListContainer());
|
||||
} else {
|
||||
// 无列表容器 → 整页提取一行
|
||||
rows = new Elements();
|
||||
rows.add(doc);
|
||||
}
|
||||
|
||||
List<Map<String, Object>> results = new ArrayList<>();
|
||||
int skip = Math.max(0, page.getSkipRows());
|
||||
for (int i = skip; i < rows.size(); i++) {
|
||||
Element row = rows.get(i);
|
||||
Map<String, Object> rowData = new LinkedHashMap<>();
|
||||
boolean hasData = false;
|
||||
|
||||
for (Map.Entry<String, FieldExtractor> field : page.getFields().entrySet()) {
|
||||
String fieldName = field.getKey();
|
||||
FieldExtractor fe = field.getValue();
|
||||
if (fe.getSelector() == null) continue;
|
||||
|
||||
Element el = row.selectFirst(fe.getSelector());
|
||||
if (el == null) {
|
||||
if (fe.getOptional() == null || !fe.getOptional()) {
|
||||
rowData.put(fieldName, null);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
String raw;
|
||||
String type = fe.getType() != null ? fe.getType() : "text";
|
||||
switch (type) {
|
||||
case "attribute":
|
||||
raw = fe.getAttribute() != null ? el.attr(fe.getAttribute()) : el.text();
|
||||
break;
|
||||
case "html":
|
||||
raw = el.html();
|
||||
break;
|
||||
default:
|
||||
raw = el.text();
|
||||
}
|
||||
|
||||
Object value = applyTransform(raw, fe.getTransform());
|
||||
rowData.put(fieldName, value);
|
||||
hasData = true;
|
||||
}
|
||||
|
||||
if (hasData) results.add(rowData);
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
/**
|
||||
* 应用内置转换器
|
||||
*/
|
||||
private Object applyTransform(String raw, String transform) {
|
||||
if (raw == null || transform == null) return raw;
|
||||
|
||||
try {
|
||||
return switch (transform.toLowerCase()) {
|
||||
case "int", "parseint" -> {
|
||||
String clean = raw.replaceAll("[^\\d-]", "");
|
||||
yield clean.isEmpty() ? null : Integer.parseInt(clean);
|
||||
}
|
||||
case "long" -> {
|
||||
String clean = raw.replaceAll("[^\\d-]", "");
|
||||
yield clean.isEmpty() ? null : Long.parseLong(clean);
|
||||
}
|
||||
case "double", "float" -> {
|
||||
String clean = raw.replaceAll("[^\\d.-]", "");
|
||||
yield clean.isEmpty() ? null : Double.parseDouble(clean);
|
||||
}
|
||||
case "filesize" -> parseFileSize(raw);
|
||||
case "boolean", "bool" -> parseBoolean(raw);
|
||||
default -> raw;
|
||||
};
|
||||
} catch (Exception e) {
|
||||
return raw;
|
||||
}
|
||||
}
|
||||
|
||||
private Long parseFileSize(String raw) {
|
||||
if (raw == null) return null;
|
||||
Matcher m = Pattern.compile("([\\d.]+)\\s*(GB|MB|KB|TB|B|GiB|MiB|KiB|TiB)",
|
||||
Pattern.CASE_INSENSITIVE).matcher(raw);
|
||||
if (!m.find()) {
|
||||
// try just number
|
||||
String num = raw.replaceAll("[^\\d.]", "");
|
||||
if (!num.isEmpty()) return (long) Double.parseDouble(num);
|
||||
return null;
|
||||
}
|
||||
double val = Double.parseDouble(m.group(1));
|
||||
String unit = m.group(2).toUpperCase();
|
||||
return (long) (val * switch (unit) {
|
||||
case "TB", "TIB" -> 1024L * 1024 * 1024 * 1024;
|
||||
case "GB", "GIB" -> 1024L * 1024 * 1024;
|
||||
case "MB", "MIB" -> 1024L * 1024;
|
||||
case "KB", "KIB" -> 1024L;
|
||||
default -> 1L;
|
||||
});
|
||||
}
|
||||
|
||||
private Boolean parseBoolean(String raw) {
|
||||
if (raw == null) return null;
|
||||
String s = raw.trim().toLowerCase();
|
||||
if (s.equals("true") || s.equals("yes") || s.equals("1") || s.equals("是")) return true;
|
||||
if (s.equals("false") || s.equals("no") || s.equals("0") || s.equals("否")) return false;
|
||||
return null;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,22 @@
|
||||
package com.par.core.adapter;
|
||||
|
||||
import com.fasterxml.jackson.annotation.JsonIgnoreProperties;
|
||||
import lombok.Data;
|
||||
|
||||
/**
|
||||
* 字段提取器定义
|
||||
*/
|
||||
@Data
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public class FieldExtractor {
|
||||
/** CSS 选择器 */
|
||||
private String selector;
|
||||
/** 提取类型: text | attribute | html */
|
||||
private String type = "text";
|
||||
/** type=attribute 时指定属性名 */
|
||||
private String attribute;
|
||||
/** 转换器名称 */
|
||||
private String transform;
|
||||
/** 是否可选 */
|
||||
private Boolean optional;
|
||||
}
|
||||
26
par-core/src/main/java/com/par/core/adapter/PageConfig.java
Normal file
26
par-core/src/main/java/com/par/core/adapter/PageConfig.java
Normal file
@@ -0,0 +1,26 @@
|
||||
package com.par.core.adapter;
|
||||
|
||||
import com.fasterxml.jackson.annotation.JsonIgnoreProperties;
|
||||
import lombok.Data;
|
||||
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* 页面解析配置
|
||||
*/
|
||||
@Data
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public class PageConfig {
|
||||
/** 页面 URL 或 URL 模板 */
|
||||
private String url;
|
||||
/** URL 正则匹配模式 */
|
||||
private String urlPattern;
|
||||
/** 请求方法 */
|
||||
private String method = "GET";
|
||||
/** 列表容器 CSS 选择器 */
|
||||
private String listContainer;
|
||||
/** 跳过行数 */
|
||||
private int skipRows;
|
||||
/** 字段定义 map */
|
||||
private Map<String, FieldExtractor> fields;
|
||||
}
|
||||
Reference in New Issue
Block a user