Java中如何处理Google自定义搜索API单请求32关键词上限
解决Google Custom Search API关键词数量限制问题
核心思路
Google Custom Search API单请求最多允许32个关键词(包含site:、location:、带引号的短语等各类查询元素)。要处理超过32个关键词的场景,需:
- 将完整查询条件拆分为多个不超过32个关键词的子查询
- 执行所有子查询
- 合并结果并去重,合理统计总结果数
重构后的代码实现
以下是修改后的GoogleSearchApiServiceImpl,新增查询拆分、批量执行和结果合并逻辑:
import com.google.api.client.http.javanet.NetHttpTransport; import com.google.api.client.json.jackson2.JacksonFactory; import com.google.api.services.customsearch.Customsearch; import com.google.api.services.customsearch.model.Result; import com.google.api.services.customsearch.model.Search; import com.arun.googleSearch.Master.Group.domain.Group; import com.arun.googleSearch.Master.Group.domain.GroupRepository; import com.arun.googleSearch.Master.category.domain.Category; import com.arun.googleSearch.Master.category.domain.CategoryRepository; import com.arun.googleSearch.Master.googesearch.data.GoogleSearchResponseData; import com.arun.googleSearch.Master.googesearch.data.GoogleSearchResultData; import com.arun.googleSearch.Master.googesearch.request.CreateGoogleSearchRequest; import com.ponsun.googleSearch.config.GoogleSearchApiConfig; import com.arun.googleSearch.infrastructure.exceptions.googleSearch_ApplicationException; import lombok.RequiredArgsConstructor; import org.springframework.stereotype.Service; import org.springframework.transaction.annotation.Transactional; import java.io.IOException; import java.text.SimpleDateFormat; import java.util.*; import java.util.concurrent.CompletableFuture; import java.util.concurrent.ExecutionException; import java.util.stream.Collectors; import java.util.stream.IntStream; import java.util.stream.Stream; @RequiredArgsConstructor @Service public class GoogleSearchApiServiceImpl implements GoogleSearchApiService { private final GoogleSearchApiConfig googleSearchConfig; private final CategoryRepository categoryRepository; private final GroupRepository groupRepository; // API允许的最大关键词数 private static final int MAX_KEYWORDS_PER_QUERY = 32; @Override @Transactional(readOnly = true) public GoogleSearchResponseData googleSearchRequest(String q, CreateGoogleSearchRequest createGoogleSearchRequest) { String apiKey = googleSearchConfig.getGoogleApiKey(); String cx = googleSearchConfig.getGoogleCx(); int reqTimeOut = googleSearchConfig.getHttpReqTimeOut(); long perPage = Optional.ofNullable(createGoogleSearchRequest.getPerPage()).orElse(10); String dateRange = createGoogleSearchRequest.getDateRestrict(); String geoCode = createGoogleSearchRequest.getMedia().equals("news") ? createGoogleSearchRequest.getCountry() : ""; String cr = geoCode.isEmpty() ? null : "country" + geoCode; String gl = geoCode.isEmpty() ? null : geoCode; String includOrExclude = findIncludeOrExclude(createGoogleSearchRequest.getMedia()); // 收集所有查询元素(每个元素算一个关键词) List<String> queryElements = new ArrayList<>(); // 添加主查询词 queryElements.add("\"" + q + "\""); // 添加公司、位置条件 if (createGoogleSearchRequest.getLocation() != null && !createGoogleSearchRequest.getLocation().isEmpty()) { queryElements.add("location:" + createGoogleSearchRequest.getLocation()); } if (createGoogleSearchRequest.getCompany() != null && !createGoogleSearchRequest.getCompany().isEmpty()) { queryElements.add("company:" + createGoogleSearchRequest.getCompany()); } // 添加站点相关条件 String siteSearch = constructSites(createGoogleSearchRequest); if (siteSearch != null && !siteSearch.isEmpty()) { // 拆分siteSearch中的每个site:xxx元素 Collections.addAll(queryElements, siteSearch.split(" OR ")); } // 添加自定义查询中的排除站点、违规分类、关键词 String customQuery = buildCustomQuery(createGoogleSearchRequest); if (customQuery != null && !customQuery.isEmpty()) { // 拆分排除站点(每个-xxx算一个元素) List<String> excludedSites = Arrays.stream(customQuery.split(" -")) .filter(s -> !s.isEmpty()) .map(s -> "-" + s) .collect(Collectors.toList()); queryElements.addAll(excludedSites); // 拆分违规分类和关键词组(每个()内的OR组算多个元素) String filteredCustom = customQuery.replaceAll("-\\S+", "").trim(); if (!filteredCustom.isEmpty()) { // 提取括号内的内容 String[] groups = filteredCustom.split("\\(|\\)"); for (String group : groups) { if (group.contains(" OR ")) { Collections.addAll(queryElements, group.split(" OR ")); } else if (!group.trim().isEmpty()) { queryElements.add(group.trim()); } } } } // 添加日期和排序条件(这些不算关键词,直接拼在每个子查询后) String dateAndSortQuery = finalQuery(createGoogleSearchRequest); // 拆分查询元素为多个子查询组 List<List<String>> queryGroups = splitIntoGroups(queryElements, MAX_KEYWORDS_PER_QUERY); Customsearch customsearch = initializeCustomSearch(reqTimeOut); GoogleSearchResponseData responseData = new GoogleSearchResponseData(); Set<GoogleSearchResultData> uniqueResults = new HashSet<>(); long maxTotalResults = 0; try { // 并行执行所有子查询 List<CompletableFuture<Void>> futures = queryGroups.stream() .map(group -> { String subQuery = String.join(" ", group) + dateAndSortQuery; return CompletableFuture.runAsync(() -> { try { Search result = executeSearch(customsearch, subQuery, apiKey, cx, 1, perPage, dateRange, gl, cr, includOrExclude); // 更新最大总结果数 synchronized (this) { if (result.getSearchInformation().getTotalResults() > maxTotalResults) { maxTotalResults = result.getSearchInformation().getTotalResults(); } } // 转换结果并添加到去重集合 List<GoogleSearchResultData> resultItems = result.getItems().stream() .map(this::createGoogleSearchResultData) .collect(Collectors.toList()); synchronized (uniqueResults) { uniqueResults.addAll(resultItems); } } catch (IOException e) { throw new googleSearch_ApplicationException("子查询执行失败: " + e.getMessage()); } }); }) .collect(Collectors.toList()); // 等待所有子查询完成 CompletableFuture.allOf(futures.toArray(new CompletableFuture[0])).get(); responseData.setTotalSearchResults(maxTotalResults); responseData.setItems(new ArrayList<>(uniqueResults)); } catch (InterruptedException | ExecutionException e) { e.printStackTrace(); throw new googleSearch_ApplicationException("批量查询执行失败: " + e.getMessage()); } return responseData; } // 将列表拆分为指定大小的分组 private List<List<String>> splitIntoGroups(List<String> items, int groupSize) { return IntStream.range(0, (items.size() + groupSize - 1) / groupSize) .mapToObj(i -> items.subList(i * groupSize, Math.min((i + 1) * groupSize, items.size()))) .collect(Collectors.toList()); } private Search executeSearch(Customsearch customsearch, String qry, String apiKey, String cx, long startIndex, long perPage, String dateRestrict, String gl, String cr, String includOrExclude) throws IOException { System.out.println("执行子查询: " + qry); Customsearch.Cse.List list = customsearch.cse().list(qry); list.setKey(apiKey); list.setCx(cx); list.setSiteSearchFilter(includOrExclude); list.setStart(startIndex); list.setNum(perPage); list.setDateRestrict(dateRestrict); list.setGl(gl); list.setCr(cr); return list.execute(); } // 以下方法保持原逻辑不变 private String findIncludeOrExclude(String value){ switch (value){ case "include": return "i"; case "exclude": return "e"; case "news": return "i"; case "sitesOnly": return "i"; default: return "i"; } } private String constructSites(CreateGoogleSearchRequest createGoogleSearchRequest){ String value = createGoogleSearchRequest.getMedia(); StringBuilder searchQuery = new StringBuilder(); //news if(value.equals("news")) { List<String> newsCategories = Stream.of(3) .flatMap(gid -> categoryRepository.findByGroupId(gid).stream()) .map(Category::getName) .collect(Collectors.toList()); if (newsCategories.size() > 15) { newsCategories = newsCategories.stream().limit(15).collect(Collectors.toList()); } String prefixedSites = newsCategories.stream() .map(site -> "site:" + site) .collect(Collectors.joining(" OR ")); searchQuery.append(prefixedSites); } //include or exclude social media List<String> socialMediaSites = Stream.of(1) .flatMap(gid -> categoryRepository.findByGroupId(gid).stream()) .map(Category::getName) .collect(Collectors.toList()); if (socialMediaSites.size() > 15) { socialMediaSites = socialMediaSites.stream().limit(15).collect(Collectors.toList()); } if (value.equals("include")){ String prefixedSites = socialMediaSites.stream() .map(site -> "site:" + site) .collect(Collectors.joining(" OR ")); searchQuery.append(prefixedSites); } if (value.equals("exclude")){ String prefixedSites = socialMediaSites.stream() .map(site -> "-" + site) .collect(Collectors.joining(" ")); searchQuery.append(prefixedSites); } //sites only if(value.equals("sitesOnly")) { List<String> includedSites = new ArrayList<>(); if (!createGoogleSearchRequest.getOnlyFromTheseSites().isEmpty()) { for (String site : createGoogleSearchRequest.getOnlyFromTheseSites()) { if (!site.contains(".")) { site += ".com"; } includedSites.add("site:" + site); } } if (!includedSites.isEmpty()) { searchQuery.append(String.join(" OR ", includedSites)); } } return searchQuery.length() > 0 ? searchQuery.toString() : null; } private Customsearch initializeCustomSearch(int reqTimeOut) { try { return new Customsearch(new NetHttpTransport(), new JacksonFactory(), httpRequest -> { httpRequest.setConnectTimeout(reqTimeOut); httpRequest.setReadTimeout(reqTimeOut); }); } catch (Exception e) { e.printStackTrace(); throw new RuntimeException("Error initializing Customsearch", e); } } private GoogleSearchResultData createGoogleSearchResultData(Result item) { String cseImageSrc = null; if (item.getPagemap() != null) { cseImageSrc = extractCseImageSrc(item.getPagemap()); } String translateLink = "https://translate.google.com/translate?hl=&sl=auto&tl=en&u=" +item.getLink(); List<GoogleSearchResultData.InnerResult> details = new ArrayList<>(); if (item.getSnippet() != null){ for (String paragraph : item.getSnippet().split("\n")) { GoogleSearchResultData.InnerResult innerResult = new GoogleSearchResultData.InnerResult(); innerResult.setParagraphs(paragraph); details.add(innerResult); } } System.out.println(item.getTitle() + " - " + item.getLink()); return new GoogleSearchResultData(item.getTitle(), item.getLink(), item.getDisplayLink(), cseImageSrc, translateLink, details); } private String extractCseImageSrc(Map<String, List<Map<String, Object>>> pagemap) { Object cseImageObject = pagemap.get("cse_image"); if (cseImageObject != null && cseImageObject instanceof List) { List cseImageList = (List) cseImageObject; if (!cseImageList.isEmpty()) { Map cseImageMap = (Map) cseImageList.get(0); return (String) cseImageMap.get("src"); } } return null; } private String buildCustomQuery(CreateGoogleSearchRequest createGoogleSearchRequest) { long startTime = System.currentTimeMillis(); Map<Boolean, List<Integer>> groupedIds = createGoogleSearchRequest.getGroupIds().stream() .flatMap(gid -> groupRepository.findById(gid.getGroupId()).stream()) .collect(Collectors.partitioningBy(Group::getIsOffence, Collectors.mapping(Group::getId, Collectors.toList()))); List<Integer> offensiveGroupIds = groupedIds.get(true); StringBuilder searchQuery = new StringBuilder(); //exclude these sites List<String> excludedSites = new ArrayList<>(); if (!createGoogleSearchRequest.getExcludeTheseSites().isEmpty()) { for (String site : createGoogleSearchRequest.getExcludeTheseSites()) { if (!site.contains(".")) { site += ".com"; } excludedSites.add("-" + site); } } if (!excludedSites.isEmpty()) { searchQuery.append(String.join(" ", excludedSites)).append(" "); } // predicate offence List<String> offensiveCategories = offensiveGroupIds.stream() .flatMap(gid -> categoryRepository.findByGroupId(gid).stream() .map(Category::getName) .findFirst() .stream() ) .toList(); if (!offensiveCategories.isEmpty()) { searchQuery.append("(").append( offensiveCategories.stream() .map(category -> category.contains(" ") ? "\"" + category + "\"" : category) .collect(Collectors.joining(" OR ")) ).append(") "); } // keywords List<String> keyWords = createGoogleSearchRequest.getKeywords(); if(!keyWords.isEmpty()){ searchQuery.append("(").append( keyWords.stream() .map(keyWord -> keyWord.contains(" ") ? "\"" + keyWord + "\"" : keyWord) .collect(Collectors.joining(" OR ")) ).append(")"); } long endTime = System.currentTimeMillis(); long elapsedTime = endTime - startTime; System.out.println("Time taken to process the buildCustomQuery: " + elapsedTime + " milliseconds"); return searchQuery.length() > 0 ? searchQuery.toString().trim() : null; } private String finalQuery(CreateGoogleSearchRequest createGoogleSearchRequest) { StringBuilder searchQuery = new StringBuilder(); // before , after dates if (createGoogleSearchRequest.getAfterDate() != null) { SimpleDateFormat formatter = new SimpleDateFormat("yyyy-MM-dd"); String formattedDate = formatter.format(createGoogleSearchRequest.getAfterDate()); searchQuery.append(" after:").append(formattedDate); } if (createGoogleSearchRequest.getBeforeDate() != null) { SimpleDateFormat formatter = new SimpleDateFormat("yyyy-MM-dd"); String formattedDate = formatter.format(createGoogleSearchRequest.getBeforeDate()); searchQuery.append(" before:").append(formattedDate); } // data sorting if (createGoogleSearchRequest.getAfterDate() == null && createGoogleSearchRequest.getBeforeDate() == null) { if (Objects.equals(createGoogleSearchRequest.getSort(), "date")) { searchQuery.append(" sort=").append(createGoogleSearchRequest.getSort()); } } return searchQuery.toString(); } }
关键逻辑说明
- 查询元素拆分:将所有查询条件(主关键词、站点、位置、分类等)拆分为单个元素,每个元素视为一个关键词,确保每组不超过32个
- 并行执行子查询:使用
CompletableFuture并行执行多个子查询,提高处理效率,注意控制并发数避免触发API频率限制 - 结果去重与合并:用
HashSet存储结果自动去重,总结果数取所有子查询的最大值(Google的总结果为估算值,求和无意义) - 兼容性处理:日期、排序等非关键词条件直接附加到每个子查询后,不占用关键词配额
内容的提问来源于stack exchange,提问作者Arun Mozhi
相关产品推荐
相关产品推荐

