// Package urlutil provides helpers for pulling URLs out of arbitrary text. package urlutil import ( "net/url" "regexp" ) // urlPattern matches http(s) URLs embedded in free-form text. var urlPattern = regexp.MustCompile(`https?://[^\s]+`) // ExtractURLs returns the http(s) URLs found in text, in order of appearance // and without duplicates. Trailing punctuation that is unlikely to be part of // a URL (e.g. a sentence-ending period or a wrapping parenthesis) is trimmed, // and each candidate is validated with url.Parse. func ExtractURLs(text string) []string { matches := urlPattern.FindAllString(text, -1) seen := make(map[string]struct{}, len(matches)) urls := make([]string, 0, len(matches)) for _, match := range matches { candidate := trimURL(match) parsed, err := url.Parse(candidate) if err != nil || parsed.Host == "" { continue } if _, ok := seen[candidate]; ok { continue } seen[candidate] = struct{}{} urls = append(urls, candidate) } return urls } // trimURL strips common trailing characters that are almost never part of the // URL itself when it appears inside a sentence. func trimURL(s string) string { for len(s) > 0 { last := s[len(s)-1] switch last { case '.', ',', ';', ':', '!', '?', ')', ']', '}', '"', '\'', '>': s = s[:len(s)-1] default: return s } } return s }