-
-
Notifications
You must be signed in to change notification settings - Fork 13
Simple crawler
kant2002 edited this page Jan 1, 2015
·
2 revisions
Simple crawler
Below is minimal crawler application
internal class Program
{
public static IFilter[] ExtensionsToSkip = new[]
{
(RegexFilter)new Regex(@"(\.jpg|\.css|\.js|\.gif|\.jpeg|\.png|\.ico)",
RegexOptions.Compiled | RegexOptions.CultureInvariant | RegexOptions.IgnoreCase)
};
void Main()
{
NCrawlerModule.Setup();
Console.Out.WriteLine("Simple crawl demo");
// Setup crawler to crawl http://ncrawler.codeplex.com
// with 1 thread adhering to robot rules, and maximum depth
// of 2 with 4 pipeline steps:
// * Step 1 - The Html Processor, parses and extracts links, text and more from html
// * Step 2 - Processes PDF files, extracting text
// * Step 3 - Try to determine language based on page, based on text extraction, using google language detection
// * Step 4 - Dump the information to the console, this is a custom step, see the DumperStep class
using (Crawler c = new Crawler(new Uri("http://ncrawler.codeplex.com"),
new HtmlDocumentProcessor(), // Process html
new iTextSharpPdfProcessor.iTextSharpPdfProcessor(),
new GoogleLanguageDetection(),
new DumperStep())
{
// Custom step to visualize crawl
MaximumThreadCount = 2,
MaximumCrawlDepth = 10,
ExcludeFilter = Program.ExtensionsToSkip,
})
{
// Begin crawl
c.Crawl();
}
}
}