diff options
| author | Michael Peter Christen <mc@yacy.net> | 2023-01-16 14:50:30 +0100 |
|---|---|---|
| committer | Michael Peter Christen <mc@yacy.net> | 2023-01-16 14:50:30 +0100 |
| commit | 9fcd8f1bdac38beb18e00e7ed37c6532ea5bd416 (patch) | |
| tree | d2328e22e15adde4b67cf40a0a880fdf107a20a1 /htroot/CrawlProfileEditor_p.xml | |
| parent | 5a52b01c0995c439aee36c63ad0c75dca5ad20e1 (diff) | |
added canonical filter
attention: this is on by default!
(it should do the right thing)
Diffstat (limited to 'htroot/CrawlProfileEditor_p.xml')
| -rw-r--r-- | htroot/CrawlProfileEditor_p.xml | 87 |
1 files changed, 44 insertions, 43 deletions
diff --git a/htroot/CrawlProfileEditor_p.xml b/htroot/CrawlProfileEditor_p.xml index 0b880ac3f..58036b48e 100644 --- a/htroot/CrawlProfileEditor_p.xml +++ b/htroot/CrawlProfileEditor_p.xml @@ -1,48 +1,49 @@ <?xml version="1.0" encoding="UTF-8" standalone="yes"?> <crawlProfiles> #{crawlProfiles}# - <crawlProfile> - <handle>#[handle]#</handle> - <name>#[name]#</name> - <collections>#[collections]#</collections> - <agentName>#[agentName]#</agentName> - <userAgent>#[userAgent]#</userAgent> - <depth>#[depth]#</depth> - <directDocByURL>#(directDocByURL)#false::true#(/directDocByURL)#</directDocByURL> - <recrawlIfOlder>#[recrawlIfOlder]#</recrawlIfOlder> - <domMaxPages>#[domMaxPages]#</domMaxPages> - <crawlingQ>#(crawlingQ)#false::true#(/crawlingQ)#</crawlingQ> - <followFrames>#(followFrames)#false::true#(/followFrames)#</followFrames> - <obeyHtmlRobotsNoindex>#(obeyHtmlRobotsNoindex)#false::true#(/obeyHtmlRobotsNoindex)#</obeyHtmlRobotsNoindex> - <obeyHtmlRobotsNofollow>#(obeyHtmlRobotsNofollow)#false::true#(/obeyHtmlRobotsNofollow)#</obeyHtmlRobotsNofollow> - <indexText>#(indexText)#false::true#(/indexText)#</indexText> - <indexMedia>#(indexMedia)#false::true#(/indexMedia)#</indexMedia> - <storeHTCache>#(storeHTCache)#false::true#(/storeHTCache)#</storeHTCache> - <remoteIndexing>#(remoteIndexing)#false::true#(/remoteIndexing)#</remoteIndexing> - <cacheStrategy>#[cacheStrategy]#</cacheStrategy> - <crawlerAlwaysCheckMediaType>#(crawlerAlwaysCheckMediaType)#false::true#(/crawlerAlwaysCheckMediaType)#</crawlerAlwaysCheckMediaType> - <crawlerURLMustMatch>#[crawlerURLMustMatch]#</crawlerURLMustMatch> - <crawlerURLMustNotMatch>#[crawlerURLMustNotMatch]#</crawlerURLMustNotMatch> - <crawlerOriginURLMustMatch>#[crawlerOriginURLMustMatch]#</crawlerOriginURLMustMatch> - <crawlerOriginURLMustNotMatch>#[crawlerOriginURLMustNotMatch]#</crawlerOriginURLMustNotMatch> - <crawlerIPMustMatch>#[crawlerIPMustMatch]#</crawlerIPMustMatch> - <crawlerIPMustNotMatch>#[crawlerIPMustNotMatch]#</crawlerIPMustNotMatch> - <crawlerCountryMustMatch>#[crawlerCountryMustMatch]#</crawlerCountryMustMatch> - <crawlerNoLimitURLMustMatch>#[crawlerNoLimitURLMustMatch]#</crawlerNoLimitURLMustMatch> - <indexURLMustMatch>#[indexURLMustMatch]#</indexURLMustMatch> - <indexURLMustNotMatch>#[indexURLMustNotMatch]#</indexURLMustNotMatch> - <indexContentMustMatch>#[indexContentMustMatch]#</indexContentMustMatch> - <indexContentMustNotMatch>#[indexContentMustNotMatch]#</indexContentMustNotMatch> - <indexMediaTypeMustMatch>#[indexMediaTypeMustMatch]#</indexMediaTypeMustMatch> - <indexMediaTypeMustNotMatch>#[indexMediaTypeMustNotMatch]#</indexMediaTypeMustNotMatch> - <indexSolrQueryMustMatch>#[indexSolrQueryMustMatch]#</indexSolrQueryMustMatch> - <indexSolrQueryMustNotMatch>#[indexSolrQueryMustNotMatch]#</indexSolrQueryMustNotMatch> - <status>#(status)#terminated::active::system#(/status)#</status> - <crawlingDomFilterContent> - #{crawlingDomFilterContent}# - <item>#[item]#</item> - #{/crawlingDomFilterContent}# - </crawlingDomFilterContent> - </crawlProfile> + <crawlProfile> + <handle>#[handle]#</handle> + <name>#[name]#</name> + <collections>#[collections]#</collections> + <agentName>#[agentName]#</agentName> + <userAgent>#[userAgent]#</userAgent> + <depth>#[depth]#</depth> + <directDocByURL>#(directDocByURL)#false::true#(/directDocByURL)#</directDocByURL> + <recrawlIfOlder>#[recrawlIfOlder]#</recrawlIfOlder> + <domMaxPages>#[domMaxPages]#</domMaxPages> + <crawlingQ>#(crawlingQ)#false::true#(/crawlingQ)#</crawlingQ> + <followFrames>#(followFrames)#false::true#(/followFrames)#</followFrames> + <obeyHtmlRobotsNoindex>#(obeyHtmlRobotsNoindex)#false::true#(/obeyHtmlRobotsNoindex)#</obeyHtmlRobotsNoindex> + <obeyHtmlRobotsNofollow>#(obeyHtmlRobotsNofollow)#false::true#(/obeyHtmlRobotsNofollow)#</obeyHtmlRobotsNofollow> + <indexText>#(indexText)#false::true#(/indexText)#</indexText> + <indexMedia>#(indexMedia)#false::true#(/indexMedia)#</indexMedia> + <storeHTCache>#(storeHTCache)#false::true#(/storeHTCache)#</storeHTCache> + <remoteIndexing>#(remoteIndexing)#false::true#(/remoteIndexing)#</remoteIndexing> + <cacheStrategy>#[cacheStrategy]#</cacheStrategy> + <crawlerAlwaysCheckMediaType>#(crawlerAlwaysCheckMediaType)#false::true#(/crawlerAlwaysCheckMediaType)#</crawlerAlwaysCheckMediaType> + <crawlerURLMustMatch>#[crawlerURLMustMatch]#</crawlerURLMustMatch> + <crawlerURLMustNotMatch>#[crawlerURLMustNotMatch]#</crawlerURLMustNotMatch> + <crawlerOriginURLMustMatch>#[crawlerOriginURLMustMatch]#</crawlerOriginURLMustMatch> + <crawlerOriginURLMustNotMatch>#[crawlerOriginURLMustNotMatch]#</crawlerOriginURLMustNotMatch> + <crawlerIPMustMatch>#[crawlerIPMustMatch]#</crawlerIPMustMatch> + <crawlerIPMustNotMatch>#[crawlerIPMustNotMatch]#</crawlerIPMustNotMatch> + <crawlerCountryMustMatch>#[crawlerCountryMustMatch]#</crawlerCountryMustMatch> + <crawlerNoLimitURLMustMatch>#[crawlerNoLimitURLMustMatch]#</crawlerNoLimitURLMustMatch> + <indexURLMustMatch>#[indexURLMustMatch]#</indexURLMustMatch> + <indexURLMustNotMatch>#[indexURLMustNotMatch]#</indexURLMustNotMatch> + <indexContentMustMatch>#[indexContentMustMatch]#</indexContentMustMatch> + <indexContentMustNotMatch>#[indexContentMustNotMatch]#</indexContentMustNotMatch> + <indexMediaTypeMustMatch>#[indexMediaTypeMustMatch]#</indexMediaTypeMustMatch> + <indexMediaTypeMustNotMatch>#[indexMediaTypeMustNotMatch]#</indexMediaTypeMustNotMatch> + <indexSolrQueryMustMatch>#[indexSolrQueryMustMatch]#</indexSolrQueryMustMatch> + <indexSolrQueryMustNotMatch>#[indexSolrQueryMustNotMatch]#</indexSolrQueryMustNotMatch> + <noindexWhenCanonicalUnequalURL>#(noindexWhenCanonicalUnequalURL)#false::true#(/noindexWhenCanonicalUnequalURL)#</noindexWhenCanonicalUnequalURL> + <status>#(status)#terminated::active::system#(/status)#</status> + <crawlingDomFilterContent> + #{crawlingDomFilterContent}# + <item>#[item]#</item> + #{/crawlingDomFilterContent}# + </crawlingDomFilterContent> + </crawlProfile> #{/crawlProfiles}# </crawlProfiles> |
