mirror of
https://github.com/edoardottt/cariddi.git
synced 2026-09-29 13:05:01 +02:00
330 lines
9.3 KiB
Go
330 lines
9.3 KiB
Go
/*
|
|
==========
|
|
Cariddi
|
|
==========
|
|
|
|
This program is free software: you can redistribute it and/or modify
|
|
it under the terms of the GNU General Public License as published by
|
|
the Free Software Foundation, either version 3 of the License, or
|
|
(at your option) any later version.
|
|
|
|
This program is distributed in the hope that it will be useful,
|
|
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
GNU General Public License for more details.
|
|
|
|
You should have received a copy of the GNU General Public License
|
|
along with this program. If not, see http://www.gnu.org/licenses/.
|
|
|
|
@Repository: https://github.com/edoardottt/cariddi
|
|
|
|
@Author: edoardottt, https://www.edoardoottavianelli.it
|
|
*/
|
|
|
|
package crawler
|
|
|
|
import (
|
|
"fmt"
|
|
"net/url"
|
|
"os"
|
|
"regexp"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/edoardottt/cariddi/output"
|
|
"github.com/edoardottt/cariddi/scanner"
|
|
"github.com/edoardottt/cariddi/utils"
|
|
"github.com/gocolly/colly"
|
|
)
|
|
|
|
//Crawler it's the actual crawler core
|
|
func Crawler(target string, txt string, html string, delayTime int, concurrency int, ignore string, ignoreTxt string,
|
|
cache bool, timeout int, secrets bool, secretsFile []string, plain bool, endpoints bool, endpointsFile []string,
|
|
fileType int) ([]string, []scanner.SecretMatched, []scanner.EndpointMatched, []scanner.FileTypeMatched) {
|
|
|
|
// This is to avoid to insert into the crawler target regular
|
|
// expression directories passed as input.
|
|
var targetTemp string
|
|
if !utils.HasScheme(target) {
|
|
targetTemp = utils.GetHost("http://" + target)
|
|
} else {
|
|
targetTemp = utils.GetHost(target)
|
|
}
|
|
|
|
if targetTemp == "" {
|
|
fmt.Println("The URL provided is not built in a proper way: " + target)
|
|
os.Exit(1)
|
|
}
|
|
|
|
//clean target input
|
|
target = utils.RemoveProtocol(target)
|
|
|
|
var ignoreSlice []string
|
|
ignoreBool := false
|
|
//if ignore -> produce the slice
|
|
if ignore != "" {
|
|
ignoreBool = true
|
|
ignoreSlice = utils.CheckInputArray(ignore)
|
|
}
|
|
|
|
//if ignoreTxt -> produce the slice
|
|
if ignoreTxt != "" {
|
|
ignoreBool = true
|
|
ignoreSlice = utils.ReadFile(ignoreTxt)
|
|
}
|
|
|
|
var FinalResults []string
|
|
var FinalSecrets []scanner.SecretMatched
|
|
var FinalEndpoints []scanner.EndpointMatched
|
|
var FinalExtensions []scanner.FileTypeMatched
|
|
|
|
// Instantiate collector using the cache if needed
|
|
c := colly.NewCollector()
|
|
if cache {
|
|
c = colly.NewCollector(
|
|
colly.AllowedDomains(targetTemp),
|
|
colly.Async(true),
|
|
colly.CacheDir("./.cariddi_cache"),
|
|
)
|
|
} else {
|
|
c = colly.NewCollector(
|
|
colly.AllowedDomains(targetTemp),
|
|
colly.Async(true),
|
|
)
|
|
}
|
|
|
|
c.Limit(
|
|
&colly.LimitRule{
|
|
Parallelism: concurrency,
|
|
Delay: time.Duration(delayTime) * time.Second,
|
|
},
|
|
)
|
|
c.AllowURLRevisit = false
|
|
if timeout != 10 {
|
|
c.SetRequestTimeout(time.Second * time.Duration(timeout))
|
|
}
|
|
|
|
// On every a element which has href attribute call callback
|
|
c.OnHTML("a[href]", func(e *colly.HTMLElement) {
|
|
link := e.Attr("href")
|
|
// Visit link found on page
|
|
// Only those links are visited which are in AllowedDomains
|
|
if ignoreBool {
|
|
if !IgnoreMatch(link, ignoreSlice) {
|
|
c.Visit(e.Request.AbsoluteURL(link))
|
|
}
|
|
} else {
|
|
c.Visit(e.Request.AbsoluteURL(link))
|
|
}
|
|
})
|
|
|
|
// On every script element which has src attribute call callback
|
|
c.OnHTML("script[src]", func(e *colly.HTMLElement) {
|
|
link := e.Attr("src")
|
|
// Visit link found on page
|
|
// Only those links are visited which are in AllowedDomains
|
|
if ignoreBool {
|
|
if !IgnoreMatch(link, ignoreSlice) {
|
|
c.Visit(e.Request.AbsoluteURL(link))
|
|
}
|
|
} else {
|
|
c.Visit(e.Request.AbsoluteURL(link))
|
|
}
|
|
})
|
|
|
|
// On every link element which has href attribute call callback
|
|
c.OnHTML("link[href]", func(e *colly.HTMLElement) {
|
|
link := e.Attr("href")
|
|
rel := e.Attr("rel")
|
|
// Visit link found on page
|
|
// Only those links are visited which are in AllowedDomains
|
|
if rel != "alternate" && rel != "stylesheet" {
|
|
if ignoreBool {
|
|
if !IgnoreMatch(link, ignoreSlice) {
|
|
c.Visit(e.Request.AbsoluteURL(link))
|
|
}
|
|
} else {
|
|
c.Visit(e.Request.AbsoluteURL(link))
|
|
}
|
|
}
|
|
})
|
|
|
|
// On every iframe element which has src attribute call callback
|
|
c.OnHTML("iframe[src]", func(e *colly.HTMLElement) {
|
|
link := e.Attr("src")
|
|
// Visit link found on page
|
|
// Only those links are visited which are in AllowedDomains
|
|
if ignoreBool {
|
|
if !IgnoreMatch(link, ignoreSlice) {
|
|
c.Visit(e.Request.AbsoluteURL(link))
|
|
}
|
|
} else {
|
|
c.Visit(e.Request.AbsoluteURL(link))
|
|
}
|
|
})
|
|
|
|
c.OnResponse(func(r *colly.Response) {
|
|
|
|
fmt.Println(r.Request.URL.String())
|
|
|
|
lengthOk := len(string(r.Body)) > 10
|
|
|
|
FinalResults = append(FinalResults, r.Request.URL.String())
|
|
|
|
//if endpoints or secrets or filetype: scan
|
|
if endpoints || secrets || (1 <= fileType && fileType <= 7) {
|
|
// HERE SCAN FOR SECRETS
|
|
if secrets && lengthOk {
|
|
secretsSlice := huntSecrets(secretsFile, r.Request.URL.String(), string(r.Body))
|
|
FinalSecrets = append(FinalSecrets, secretsSlice...)
|
|
}
|
|
// HERE SCAN FOR ENDPOINTS
|
|
if endpoints {
|
|
endpointsSlice := huntEndpoints(endpointsFile, r.Request.URL.String())
|
|
for _, elem := range endpointsSlice {
|
|
if len(elem.Parameters) != 0 {
|
|
FinalEndpoints = append(FinalEndpoints, elem)
|
|
}
|
|
}
|
|
}
|
|
// HERE SCAN FOR EXTENSIONS
|
|
if 1 <= fileType && fileType <= 7 {
|
|
extension := huntExtensions(r.Request.URL.String(), fileType)
|
|
if extension.Url != "" {
|
|
FinalExtensions = append(FinalExtensions, extension)
|
|
}
|
|
}
|
|
}
|
|
})
|
|
|
|
// Start scraping on target
|
|
c.Visit("http://" + target)
|
|
c.Visit("https://" + target)
|
|
c.Wait()
|
|
if html != "" {
|
|
output.FooterHTML(html)
|
|
}
|
|
return FinalResults, FinalSecrets, FinalEndpoints, FinalExtensions
|
|
}
|
|
|
|
//huntSecrets hunts for secrets
|
|
func huntSecrets(secretsFile []string, target string, body string) []scanner.SecretMatched {
|
|
secrets := SecretsMatch(target, body, secretsFile)
|
|
return secrets
|
|
}
|
|
|
|
//SecretsMatch checks if a body matches some secrets
|
|
func SecretsMatch(url string, body string, secretsFile []string) []scanner.SecretMatched {
|
|
var secrets []scanner.SecretMatched
|
|
if len(secretsFile) == 0 {
|
|
for _, secret := range scanner.GetRegexes() {
|
|
if matched, err := regexp.Match(secret.Regex, []byte(body)); err == nil && matched {
|
|
re := regexp.MustCompile(secret.Regex)
|
|
match := re.FindStringSubmatch(body)
|
|
// Avoiding false positives
|
|
var isFalsePositive = false
|
|
for _, falsePositive := range secret.FalsePositives {
|
|
if strings.Contains(strings.ToLower(match[0]), falsePositive) {
|
|
isFalsePositive = true
|
|
break
|
|
}
|
|
}
|
|
if !isFalsePositive {
|
|
secretFound := scanner.SecretMatched{Secret: secret, Url: url, Match: match[0]}
|
|
secrets = append(secrets, secretFound)
|
|
}
|
|
}
|
|
}
|
|
} else {
|
|
for _, secret := range secretsFile {
|
|
if matched, err := regexp.Match(secret, []byte(body)); err == nil && matched {
|
|
re := regexp.MustCompile(secret)
|
|
match := re.FindStringSubmatch(body)
|
|
secretScanned := scanner.Secret{Name: "CustomFromFile", Description: "", Regex: secret, Poc: ""}
|
|
secretFound := scanner.SecretMatched{Secret: secretScanned, Url: url, Match: match[0]}
|
|
secrets = append(secrets, secretFound)
|
|
}
|
|
}
|
|
}
|
|
return secrets
|
|
}
|
|
|
|
//huntEndpoints hunts for juicy endpoints
|
|
func huntEndpoints(endpointsFile []string, target string) []scanner.EndpointMatched {
|
|
endpoints := EndpointsMatch(target, endpointsFile)
|
|
return endpoints
|
|
}
|
|
|
|
//EndpointsMatch check if an endpoint matches a juicy parameter
|
|
func EndpointsMatch(target string, endpointsFile []string) []scanner.EndpointMatched {
|
|
var endpoints []scanner.EndpointMatched
|
|
matched := []scanner.Parameter{}
|
|
parameters := utils.RetrieveParameters(target)
|
|
if len(endpointsFile) == 0 {
|
|
for _, parameter := range scanner.GetJuicyParameters() {
|
|
for _, param := range parameters {
|
|
if strings.ToLower(param) == parameter.Parameter {
|
|
matched = append(matched, parameter)
|
|
}
|
|
endpoints = append(endpoints, scanner.EndpointMatched{Parameters: matched, Url: target})
|
|
}
|
|
}
|
|
} else {
|
|
|
|
for _, parameter := range endpointsFile {
|
|
for _, param := range parameters {
|
|
if param == parameter {
|
|
matched = append(matched, scanner.Parameter{Parameter: parameter, Attacks: []string{}})
|
|
}
|
|
endpoints = append(endpoints, scanner.EndpointMatched{Parameters: matched, Url: target})
|
|
}
|
|
}
|
|
}
|
|
return endpoints
|
|
}
|
|
|
|
//huntExtensions hunts for extensions
|
|
func huntExtensions(target string, severity int) scanner.FileTypeMatched {
|
|
var extension scanner.FileTypeMatched
|
|
copyTarget := target
|
|
for _, ext := range scanner.GetExtensions() {
|
|
if ext.Severity <= severity {
|
|
firstIndex := strings.Index(target, "?")
|
|
if firstIndex > -1 {
|
|
target = target[:firstIndex]
|
|
}
|
|
i := strings.LastIndex(target, ".")
|
|
if i >= 0 && strings.ToLower(target[i:]) == "."+ext.Extension {
|
|
extension = scanner.FileTypeMatched{Filetype: ext, Url: copyTarget}
|
|
}
|
|
}
|
|
}
|
|
return extension
|
|
}
|
|
|
|
//RetrieveBody retrieves the body of a url
|
|
func RetrieveBody(target string) string {
|
|
sb, err := GetRequest(target)
|
|
if err == nil && sb != "" {
|
|
return sb
|
|
}
|
|
return ""
|
|
}
|
|
|
|
//isLinkOkay checks if a link is buit in a proper way
|
|
func isLinkOkay(input string) bool {
|
|
_, err := url.Parse(input)
|
|
return err == nil
|
|
}
|
|
|
|
//IgnoreMatch checks if the URL is not in
|
|
//the ignored ones.
|
|
func IgnoreMatch(url string, ignoreSlice []string) bool {
|
|
for _, ignore := range ignoreSlice {
|
|
if strings.Contains(url, ignore) {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|