summaryrefslogtreecommitdiff
path: root/htroot/api
diff options
context:
space:
mode:
authorluccioman <luccioman@users.noreply.github.com>2017-07-08 09:04:03 +0200
committerluccioman <luccioman@users.noreply.github.com>2017-07-08 09:04:03 +0200
commitbf55f1d6e582eb126e74d18e0bea2be542efda68 (patch)
tree890bafc9b681e0e3736f8e33bb7eb35842aaa117 /htroot/api
parent2a87b08cea67f8f2ae46e318c1c3945e8520ec53 (diff)
Started support of partial parsing on large streamed resources.
Thus enable getpageinfo_p API to return something in a reasonable amount of time on resources over MegaBytes size range. Support added first with the generic XML parser, for other formats regular crawler limits apply as usual.
Diffstat (limited to 'htroot/api')
-rw-r--r--htroot/api/getpageinfo_p.java19
1 files changed, 15 insertions, 4 deletions
diff --git a/htroot/api/getpageinfo_p.java b/htroot/api/getpageinfo_p.java
index a24f3ccc5..309421a63 100644
--- a/htroot/api/getpageinfo_p.java
+++ b/htroot/api/getpageinfo_p.java
@@ -87,7 +87,8 @@ public class getpageinfo_p {
* </ul>
* </li>
* <li>agentName (optional) : the string identifying the agent used to fetch the resource. Example : "YaCy Internet (cautious)"</li>
- * <li>maxLinks (optional) : the maximum number of links, sitemap URLs or icons to return</li>
+ * <li>maxLinks (optional integer value) : the maximum number of links, sitemap URLs or icons to return on 'title' action</li>
+ * <li>maxBytes (optional long integer value) : the maximum number of bytes to load and parse from the url on 'title' action</li>
* </ul>
* @param env
* server environment
@@ -139,7 +140,17 @@ public class getpageinfo_p {
net.yacy.document.Document scraper = null;
if (u != null) try {
ClientIdentification.Agent agent = ClientIdentification.getAgent(post.get("agentName", ClientIdentification.yacyInternetCrawlerAgentName));
- scraper = sb.loader.loadDocumentAsStream(u, CacheStrategy.IFEXIST, BlacklistType.CRAWLER, agent);
+
+ if(post.containsKey("maxBytes")) {
+ /* A maxBytes limit is specified : let's try to parse only the amount of bytes given */
+ final long maxBytes = post.getLong("maxBytes", sb.loader.protocolMaxFileSize(u));
+ scraper = sb.loader.loadDocumentAsLimitedStream(u, CacheStrategy.IFEXIST, BlacklistType.CRAWLER, agent, maxLinks, maxBytes);
+ } else {
+ /* No maxBytes limit : apply regular parsing with default crawler limits.
+ * Eventual maxLinks limit will apply after loading and parsing the document. */
+ scraper = sb.loader.loadDocumentAsStream(u, CacheStrategy.IFEXIST, BlacklistType.CRAWLER, agent);
+ }
+
} catch (final IOException e) {
ConcurrentLog.logException(e);
// bad things are possible, i.e. that the Server responds with "403 Bad Behavior"
@@ -151,7 +162,7 @@ public class getpageinfo_p {
// put the icons that belong to the document
Set<DigestURL> iconURLs = scraper.getIcons().keySet();
- int count = 0;
+ long count = 0;
for (DigestURL iconURL : iconURLs) {
if(count >= maxLinks) {
break;
@@ -199,7 +210,7 @@ public class getpageinfo_p {
count++;
}
prop.put("links", count);
- prop.put("hasMoreLinks", (count >= maxLinks && urisIt.hasNext()) ? "1" : "0");
+ prop.put("hasMoreLinks", scraper.isPartiallyParsed() || (count >= maxLinks && urisIt.hasNext()) ? "1" : "0");
prop.putXML("sitelist", links.length() > 0 ? links.substring(1) : "");
prop.putXML("filter", filter.length() > 0 ? filter.substring(1) : ".*");
}