feat: 适配器解析引擎,用 CSS 选择器从提交的 HTML 中自动提取数据进行校验

This commit is contained in:
mediabot-pt
2026-07-01 01:25:16 +08:00
parent e27a5b59ed
commit e56f1d4b92
9 changed files with 439 additions and 0 deletions

View File

@@ -96,5 +96,11 @@
<artifactId>qiniu-java-sdk</artifactId>
<version>7.15.0</version>
</dependency>
<!-- Jsoup HTML 解析 -->
<dependency>
<groupId>org.jsoup</groupId>
<artifactId>jsoup</artifactId>
</dependency>
</dependencies>
</project>

View File

@@ -0,0 +1,22 @@
package com.par.core.adapter;
import com.fasterxml.jackson.annotation.JsonIgnoreProperties;
import lombok.Data;
import java.util.Map;
/**
* 适配器配置根对象
*/
@Data
@JsonIgnoreProperties(ignoreUnknown = true)
public class AdapterConfig {
/** 站点 ID */
private String siteId;
/** 站点类型 */
private String siteType;
/** 域名列表 */
private String[] domains;
/** 页面解析规则 */
private Map<String, PageConfig> pages;
}

View File

@@ -0,0 +1,211 @@
package com.par.core.adapter;
import com.fasterxml.jackson.databind.ObjectMapper;
import lombok.extern.slf4j.Slf4j;
import org.jsoup.Jsoup;
import org.jsoup.nodes.Document;
import org.jsoup.nodes.Element;
import org.jsoup.select.Elements;
import org.springframework.stereotype.Service;
import java.util.*;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
import java.util.stream.Collectors;
/**
* 适配器解析引擎:根据配置文件的 CSS 选择器从 HTML 中提取数据
*/
@Slf4j
@Service
public class AdapterEngine {
private static final ObjectMapper objectMapper = new ObjectMapper();
/**
* 解析单个 HTML 页面,匹配配置中的页面规则并提取数据
* @return { pageName, pageUrl, extracted: [{field: value, ...}, ...] }
*/
public Map<String, Object> extract(AdapterConfig config, String html, String accessUrl) {
Map<String, Object> result = new LinkedHashMap<>();
Document doc = Jsoup.parse(html);
// 匹配页面
Map.Entry<String, PageConfig> matched = matchPage(config.getPages(), accessUrl);
if (matched == null) {
result.put("matched", false);
result.put("message", "未匹配到页面规则");
result.put("availablePages", config.getPages().keySet());
return result;
}
String pageName = matched.getKey();
PageConfig page = matched.getValue();
result.put("matched", true);
result.put("pageName", pageName);
result.put("pageUrl", accessUrl);
// 提取数据
List<Map<String, Object>> extracted = extractRows(doc, page);
result.put("extracted", extracted);
result.put("rowCount", extracted.size());
result.put("fieldCount", page.getFields() != null ? page.getFields().size() : 0);
return result;
}
/**
* 匹配 URL 对应的页面规则
*/
private Map.Entry<String, PageConfig> matchPage(Map<String, PageConfig> pages, String accessUrl) {
if (pages == null || pages.isEmpty() || accessUrl == null) return null;
for (Map.Entry<String, PageConfig> entry : pages.entrySet()) {
PageConfig page = entry.getValue();
// 精确 URL 匹配
if (page.getUrl() != null) {
String normalizedPageUrl = page.getUrl().replaceAll("\\?.+", "");
String normalizedAccessUrl = accessUrl.replaceAll("\\?.+", "");
if (normalizedAccessUrl.contains(normalizedPageUrl)
|| normalizedPageUrl.contains(normalizedAccessUrl)) {
return entry;
}
}
// URL 正则匹配
if (page.getUrlPattern() != null) {
try {
if (Pattern.compile(page.getUrlPattern()).matcher(accessUrl).find()) {
return entry;
}
} catch (Exception ignored) {}
}
// 模糊匹配:URL 路径中包含页面 key
if (accessUrl.contains("/" + entry.getKey())) {
return entry;
}
}
return null;
}
/**
* 从文档中提取所有行的数据
*/
@SuppressWarnings("unchecked")
private List<Map<String, Object>> extractRows(Document doc, PageConfig page) {
if (page.getFields() == null || page.getFields().isEmpty()) return List.of();
Elements rows;
if (page.getListContainer() != null && !page.getListContainer().isBlank()) {
rows = doc.select(page.getListContainer());
} else {
// 无列表容器 → 整页提取一行
rows = new Elements();
rows.add(doc);
}
List<Map<String, Object>> results = new ArrayList<>();
int skip = Math.max(0, page.getSkipRows());
for (int i = skip; i < rows.size(); i++) {
Element row = rows.get(i);
Map<String, Object> rowData = new LinkedHashMap<>();
boolean hasData = false;
for (Map.Entry<String, FieldExtractor> field : page.getFields().entrySet()) {
String fieldName = field.getKey();
FieldExtractor fe = field.getValue();
if (fe.getSelector() == null) continue;
Element el = row.selectFirst(fe.getSelector());
if (el == null) {
if (fe.getOptional() == null || !fe.getOptional()) {
rowData.put(fieldName, null);
}
continue;
}
String raw;
String type = fe.getType() != null ? fe.getType() : "text";
switch (type) {
case "attribute":
raw = fe.getAttribute() != null ? el.attr(fe.getAttribute()) : el.text();
break;
case "html":
raw = el.html();
break;
default:
raw = el.text();
}
Object value = applyTransform(raw, fe.getTransform());
rowData.put(fieldName, value);
hasData = true;
}
if (hasData) results.add(rowData);
}
return results;
}
/**
* 应用内置转换器
*/
private Object applyTransform(String raw, String transform) {
if (raw == null || transform == null) return raw;
try {
return switch (transform.toLowerCase()) {
case "int", "parseint" -> {
String clean = raw.replaceAll("[^\\d-]", "");
yield clean.isEmpty() ? null : Integer.parseInt(clean);
}
case "long" -> {
String clean = raw.replaceAll("[^\\d-]", "");
yield clean.isEmpty() ? null : Long.parseLong(clean);
}
case "double", "float" -> {
String clean = raw.replaceAll("[^\\d.-]", "");
yield clean.isEmpty() ? null : Double.parseDouble(clean);
}
case "filesize" -> parseFileSize(raw);
case "boolean", "bool" -> parseBoolean(raw);
default -> raw;
};
} catch (Exception e) {
return raw;
}
}
private Long parseFileSize(String raw) {
if (raw == null) return null;
Matcher m = Pattern.compile("([\\d.]+)\\s*(GB|MB|KB|TB|B|GiB|MiB|KiB|TiB)",
Pattern.CASE_INSENSITIVE).matcher(raw);
if (!m.find()) {
// try just number
String num = raw.replaceAll("[^\\d.]", "");
if (!num.isEmpty()) return (long) Double.parseDouble(num);
return null;
}
double val = Double.parseDouble(m.group(1));
String unit = m.group(2).toUpperCase();
return (long) (val * switch (unit) {
case "TB", "TIB" -> 1024L * 1024 * 1024 * 1024;
case "GB", "GIB" -> 1024L * 1024 * 1024;
case "MB", "MIB" -> 1024L * 1024;
case "KB", "KIB" -> 1024L;
default -> 1L;
});
}
private Boolean parseBoolean(String raw) {
if (raw == null) return null;
String s = raw.trim().toLowerCase();
if (s.equals("true") || s.equals("yes") || s.equals("1") || s.equals("是")) return true;
if (s.equals("false") || s.equals("no") || s.equals("0") || s.equals("否")) return false;
return null;
}
}

View File

@@ -0,0 +1,22 @@
package com.par.core.adapter;
import com.fasterxml.jackson.annotation.JsonIgnoreProperties;
import lombok.Data;
/**
* 字段提取器定义
*/
@Data
@JsonIgnoreProperties(ignoreUnknown = true)
public class FieldExtractor {
/** CSS 选择器 */
private String selector;
/** 提取类型: text | attribute | html */
private String type = "text";
/** type=attribute 时指定属性名 */
private String attribute;
/** 转换器名称 */
private String transform;
/** 是否可选 */
private Boolean optional;
}

View File

@@ -0,0 +1,26 @@
package com.par.core.adapter;
import com.fasterxml.jackson.annotation.JsonIgnoreProperties;
import lombok.Data;
import java.util.Map;
/**
* 页面解析配置
*/
@Data
@JsonIgnoreProperties(ignoreUnknown = true)
public class PageConfig {
/** 页面 URL 或 URL 模板 */
private String url;
/** URL 正则匹配模式 */
private String urlPattern;
/** 请求方法 */
private String method = "GET";
/** 列表容器 CSS 选择器 */
private String listContainer;
/** 跳过行数 */
private int skipRows;
/** 字段定义 map */
private Map<String, FieldExtractor> fields;
}