mirror of
https://github.com/muraenateam/muraena.git
synced 2026-07-23 18:17:53 +00:00
225 lines
5.5 KiB
Go
225 lines
5.5 KiB
Go
package crawler
|
|
|
|
import (
|
|
"fmt"
|
|
"net/url"
|
|
"strings"
|
|
"sync"
|
|
"time"
|
|
|
|
"github.com/ditashi/jsbeautifier-go/jsbeautifier"
|
|
"github.com/evilsocket/islazy/tui"
|
|
"github.com/gocolly/colly"
|
|
"gopkg.in/resty.v1"
|
|
"mvdan.cc/xurls"
|
|
|
|
"github.com/muraenateam/muraena/proxy"
|
|
"github.com/muraenateam/muraena/session"
|
|
)
|
|
|
|
const (
|
|
// Name of this module
|
|
Name = "crawler"
|
|
|
|
// Description of this module
|
|
Description = "Crawls the target domain in order to retrieve most of the target external origins"
|
|
|
|
// Author of this module
|
|
Author = "Muraena Team"
|
|
)
|
|
|
|
// Crawler module
|
|
type Crawler struct {
|
|
session.SessionModule
|
|
|
|
Enabled bool
|
|
Depth int
|
|
UpTo int
|
|
|
|
Domains []string
|
|
}
|
|
|
|
var (
|
|
subdomains, uniqueDomains []string
|
|
discoveredJsUrls []string
|
|
waitGroup sync.WaitGroup
|
|
)
|
|
|
|
// Name returns the module name
|
|
func (module *Crawler) Name() string {
|
|
return Name
|
|
}
|
|
|
|
// Description returns the module description
|
|
func (module *Crawler) Description() string {
|
|
return Description
|
|
}
|
|
|
|
// Author returns the module author
|
|
func (module *Crawler) Author() string {
|
|
return Author
|
|
}
|
|
|
|
// Prompt prints module status based on the provided parameters
|
|
func (module *Crawler) Prompt() {
|
|
module.Raw("No options are available for this module")
|
|
}
|
|
|
|
// Load configures the module by initializing its main structure and variables
|
|
func Load(s *session.Session) (m *Crawler, err error) {
|
|
|
|
config := s.Config.Crawler
|
|
m = &Crawler{
|
|
SessionModule: session.NewSessionModule(Name, s),
|
|
Enabled: config.Enabled,
|
|
UpTo: config.UpTo,
|
|
Depth: config.Depth,
|
|
}
|
|
|
|
// Armor domains
|
|
config.ExternalOrigins = proxy.ArmorDomain(config.ExternalOrigins)
|
|
if !m.Enabled {
|
|
m.Debug("is disabled")
|
|
return
|
|
}
|
|
|
|
m.explore()
|
|
waitGroup.Wait()
|
|
config.ExternalOrigins = proxy.ArmorDomain(m.Domains)
|
|
|
|
m.Info("Domain crawling stats:")
|
|
err = s.UpdateConfiguration(&config.ExternalOrigins, &subdomains, &uniqueDomains)
|
|
return
|
|
}
|
|
|
|
func (module *Crawler) explore() {
|
|
|
|
var config *session.Configuration
|
|
config = module.Session.Config
|
|
|
|
module.Info("Starting exploration of %s (crawlDepth:%d crawlMaxReq: %d), just a few seconds...",
|
|
config.Proxy.Target, module.Depth, module.UpTo)
|
|
|
|
c := colly.NewCollector(
|
|
colly.UserAgent("Mozilla/5.0 (Macintosh; Intel Mac OS X x.y; rv:10.0) Gecko/20100101 Firefox/10.0"),
|
|
|
|
// MaxDepth is by default 1, so only the links on the scraped page are visited,
|
|
// and no further links are followed
|
|
colly.MaxDepth(module.Depth),
|
|
)
|
|
|
|
numVisited := 0
|
|
c.OnRequest(func(r *colly.Request) {
|
|
numVisited++
|
|
if numVisited > module.UpTo {
|
|
r.Abort()
|
|
return
|
|
}
|
|
})
|
|
|
|
c.OnHTML("script[src]", func(e *colly.HTMLElement) {
|
|
res := e.Attr("src")
|
|
if module.appendExternalDomain(res) {
|
|
// if it is a script from an external domain, make sure to fetch it
|
|
// beautify it and see it we need to replace things
|
|
waitGroup.Add(1)
|
|
go module.fetchJS(&waitGroup, res, config.Proxy.Target)
|
|
}
|
|
|
|
})
|
|
|
|
// all other tags with src attribute (img/video/iframe/etc..)
|
|
c.OnHTML("[src]", func(e *colly.HTMLElement) {
|
|
res := e.Attr("src")
|
|
module.appendExternalDomain(res)
|
|
})
|
|
|
|
c.OnHTML("link[href]", func(e *colly.HTMLElement) {
|
|
res := e.Attr("href")
|
|
module.appendExternalDomain(res)
|
|
})
|
|
|
|
c.OnHTML("meta[content]", func(e *colly.HTMLElement) {
|
|
res := e.Attr("content")
|
|
module.appendExternalDomain(res)
|
|
})
|
|
|
|
// Callback for links on scraped pages
|
|
c.OnHTML("a[href]", func(e *colly.HTMLElement) {
|
|
res := e.Attr("href")
|
|
module.appendExternalDomain(res)
|
|
|
|
// crawl
|
|
if err := c.Visit(e.Request.AbsoluteURL(res)); err != nil {
|
|
// module.Debug("[Colly Visit]%s", err)
|
|
}
|
|
})
|
|
|
|
if err := c.Limit(&colly.LimitRule{DomainGlob: "*", RandomDelay: 500 * time.Millisecond}); err != nil {
|
|
module.Warning("[Colly Limit]%s", err)
|
|
}
|
|
|
|
c.OnResponse(func(r *colly.Response) {})
|
|
|
|
c.OnRequest(func(r *colly.Request) {})
|
|
|
|
dest := fmt.Sprintf("%s%s", config.Protocol, config.Proxy.Target)
|
|
err := c.Visit(dest)
|
|
if err != nil {
|
|
module.Info("Exploration error visiting %s: %s", dest, tui.Red(err.Error()))
|
|
}
|
|
}
|
|
|
|
func (module *Crawler) fetchJS(waitGroup *sync.WaitGroup, res string, dest string) {
|
|
|
|
defer waitGroup.Done()
|
|
|
|
u, _ := url.Parse(res)
|
|
if u.Scheme == "" {
|
|
u.Scheme = "https://"
|
|
res = "https:" + res
|
|
}
|
|
nu := fmt.Sprintf("%s%s", u.Host, u.Path)
|
|
if !Contains(&discoveredJsUrls, nu) {
|
|
discoveredJsUrls = append(discoveredJsUrls, nu)
|
|
module.Debug("Fetching new JS URL: %s", nu)
|
|
|
|
resp, err := resty.R().Get(res)
|
|
if err != nil {
|
|
module.Error("Error fetching JS at %s: %s", res, err)
|
|
}
|
|
body := string(resp.Body())
|
|
|
|
opts := jsbeautifier.DefaultOptions()
|
|
beautyBody, err := jsbeautifier.Beautify(&body, opts)
|
|
if err != nil {
|
|
module.Error("Error beautifying JS at %s", res)
|
|
}
|
|
|
|
jsUrls := xurls.Strict.FindAllString(beautyBody, -1)
|
|
if len(jsUrls) > 0 && len(jsUrls) < 100 { // prevent cases where we have a lots of domains
|
|
for _, jsURL := range jsUrls {
|
|
module.appendExternalDomain(jsURL)
|
|
}
|
|
module.Info("%d domain(s) found in JS at %s", len(jsUrls), res)
|
|
}
|
|
}
|
|
}
|
|
|
|
func (module *Crawler) appendExternalDomain(res string) bool {
|
|
if strings.HasPrefix(res, "//") || strings.HasPrefix(res, "https://") || strings.HasPrefix(res, "http://") {
|
|
u, err := url.Parse(res)
|
|
if err != nil {
|
|
module.Error("url.Parse error, skipping external domain %s: %s", res, err)
|
|
return false
|
|
}
|
|
// update the crawledDomains after doing some minimal checks that might happen from xurls when
|
|
// parsing urls from JS files
|
|
if len(u.Host) > 2 && (strings.Contains(u.Host, ".") || strings.Contains(u.Host, ":")) {
|
|
module.Domains = append(module.Domains, u.Host)
|
|
}
|
|
|
|
return true
|
|
}
|
|
return false
|
|
}
|